@tangle-network/agent-eval 0.179.0 → 0.181.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +66 -0
- package/README.md +119 -146
- package/dist/adapters/http.d.ts +2 -2
- package/dist/{agent-profile-B7yErX0q.d.ts → agent-profile-CivaSsSy.d.ts} +4 -4
- package/dist/{agent-profile-B7yErX0q.d.ts.map → agent-profile-CivaSsSy.d.ts.map} +1 -1
- package/dist/{agent-profile-cell-0gSi5ffD.js → agent-profile-cell-Cv6UA-W_.js} +20 -57
- package/dist/agent-profile-cell-Cv6UA-W_.js.map +1 -0
- package/dist/{agent-profile-cell-CTOZJUuE.d.ts → agent-profile-cell-s__adRnK.d.ts} +3 -3
- package/dist/agent-profile-cell-s__adRnK.d.ts.map +1 -0
- package/dist/analyst/index.d.ts +10 -10
- package/dist/analyst/index.js +4 -4
- package/dist/ast-CP9ae9B0.js +557 -0
- package/dist/ast-CP9ae9B0.js.map +1 -0
- package/dist/ast-hI-vjW6J.d.ts +457 -0
- package/dist/ast-hI-vjW6J.d.ts.map +1 -0
- package/dist/{benchmark-command-CY6Dg5t5.js → benchmark-command-B57n9vjz.js} +7 -6
- package/dist/{benchmark-command-CY6Dg5t5.js.map → benchmark-command-B57n9vjz.js.map} +1 -1
- package/dist/benchmarks/index.d.ts +4 -4
- package/dist/benchmarks/index.js +3 -3
- package/dist/campaign/index.d.ts +6 -6
- package/dist/campaign/index.js +8 -8
- package/dist/{campaign-BGEurASO.js → campaign-4_ppJW5X.js} +12 -12
- package/dist/{campaign-BGEurASO.js.map → campaign-4_ppJW5X.js.map} +1 -1
- package/dist/{campaign-evidence-D8DBLqLI.js → campaign-evidence-B8oF9xQ6.js} +515 -471
- package/dist/campaign-evidence-B8oF9xQ6.js.map +1 -0
- package/dist/cli.js +5 -8
- package/dist/cli.js.map +1 -1
- package/dist/{client-BlLY6o2w.js → client-CXE-U1SA.js} +3 -1
- package/dist/client-CXE-U1SA.js.map +1 -0
- package/dist/{client-CuQgX33c.d.ts → client-kh2jOjTK.d.ts} +4 -4
- package/dist/{client-CuQgX33c.d.ts.map → client-kh2jOjTK.d.ts.map} +1 -1
- package/dist/contract/index.d.ts +13 -13
- package/dist/contract/index.js +11 -10
- package/dist/contract/index.js.map +1 -1
- package/dist/{default-registry-IGDE9XIC.d.ts → default-registry-BwDSWVzg.d.ts} +6 -6
- package/dist/{default-registry-IGDE9XIC.d.ts.map → default-registry-BwDSWVzg.d.ts.map} +1 -1
- package/dist/{default-registry-BryMEmr8.js → default-registry-aL7xUrUz.js} +2 -2
- package/dist/{default-registry-BryMEmr8.js.map → default-registry-aL7xUrUz.js.map} +1 -1
- package/dist/{define-agent-eval-Cx4Ls9ta.d.ts → define-agent-eval-CwOWWQt_.d.ts} +33 -12
- package/dist/define-agent-eval-CwOWWQt_.d.ts.map +1 -0
- package/dist/{define-agent-eval-Dzidv34q.js → define-agent-eval-Ddu33JH9.js} +134 -67
- package/dist/define-agent-eval-Ddu33JH9.js.map +1 -0
- package/dist/{dspy-rlm-engine-xKiWmj_G.js → dspy-rlm-engine-S53V0HhE.js} +2 -2
- package/dist/{dspy-rlm-engine-xKiWmj_G.js.map → dspy-rlm-engine-S53V0HhE.js.map} +1 -1
- package/dist/{engine-CX8ReXkn.d.ts → engine-DS1cysJy.d.ts} +10 -7
- package/dist/engine-DS1cysJy.d.ts.map +1 -0
- package/dist/{eval-campaign-Cs-7MiCs.js → eval-campaign-aYdtjtJR.js} +4 -4
- package/dist/{eval-campaign-Cs-7MiCs.js.map → eval-campaign-aYdtjtJR.js.map} +1 -1
- package/dist/{exact-types-B7LC1EyX.d.ts → exact-types-BZDe0W2D.d.ts} +2 -2
- package/dist/{exact-types-B7LC1EyX.d.ts.map → exact-types-BZDe0W2D.d.ts.map} +1 -1
- package/dist/experiment/index.d.ts +27 -477
- package/dist/experiment/index.d.ts.map +1 -1
- package/dist/experiment/index.js +95 -559
- package/dist/experiment/index.js.map +1 -1
- package/dist/{experiment-tracker-B3TiF5-u.d.ts → experiment-tracker-C7PfnF4b.d.ts} +2 -2
- package/dist/{experiment-tracker-B3TiF5-u.d.ts.map → experiment-tracker-C7PfnF4b.d.ts.map} +1 -1
- package/dist/{external-optimizer-process-Dlz8YxrT.js → external-optimizer-process-QDRURJAM.js} +3 -3
- package/dist/{external-optimizer-process-Dlz8YxrT.js.map → external-optimizer-process-QDRURJAM.js.map} +1 -1
- package/dist/{external-optimizer-subprocess-q3VzlGAO.js → external-optimizer-subprocess-D4dzUBZI.js} +3 -2
- package/dist/{external-optimizer-subprocess-q3VzlGAO.js.map → external-optimizer-subprocess-D4dzUBZI.js.map} +1 -1
- package/dist/{feedback-trajectory-eHWNv5Aj.d.ts → feedback-trajectory-CXmtITBo.d.ts} +3 -3
- package/dist/{feedback-trajectory-eHWNv5Aj.d.ts.map → feedback-trajectory-CXmtITBo.d.ts.map} +1 -1
- package/dist/hosted/index.d.ts +2 -2
- package/dist/hosted/index.d.ts.map +1 -1
- package/dist/hosted/index.js +1 -1
- package/dist/{index-BxWvILU8.d.ts → index-Bp_6sj3x.d.ts} +109 -56
- package/dist/index-Bp_6sj3x.d.ts.map +1 -0
- package/dist/{index-e7LXeRVa.d.ts → index-CJ3LhKIX.d.ts} +2 -2
- package/dist/{index-e7LXeRVa.d.ts.map → index-CJ3LhKIX.d.ts.map} +1 -1
- package/dist/{index-CbLmrWCa.d.ts → index-DNntP4ch.d.ts} +8 -8
- package/dist/{index-CbLmrWCa.d.ts.map → index-DNntP4ch.d.ts.map} +1 -1
- package/dist/{index-DxNYmx4a.d.ts → index-DoykkxW0.d.ts} +11 -11
- package/dist/{index-DxNYmx4a.d.ts.map → index-DoykkxW0.d.ts.map} +1 -1
- package/dist/index.d.ts +28 -28
- package/dist/index.js +25 -16
- package/dist/index.js.map +1 -1
- package/dist/{insight-report-DETqPc_A.d.ts → insight-report-D1qa0HWs.d.ts} +9 -5
- package/dist/{insight-report-DETqPc_A.d.ts.map → insight-report-D1qa0HWs.d.ts.map} +1 -1
- package/dist/{integrity-DsHWCebQ.js → integrity-DH5ng72x.js} +2 -2
- package/dist/{integrity-DsHWCebQ.js.map → integrity-DH5ng72x.js.map} +1 -1
- package/dist/{integrity-BKTcA-HP.d.ts → integrity-rGOfSUle.d.ts} +2 -2
- package/dist/{integrity-BKTcA-HP.d.ts.map → integrity-rGOfSUle.d.ts.map} +1 -1
- package/dist/{ledger-core-Cs9f7385.js → journal-Cs9f7385.js} +1 -1
- package/dist/journal-Cs9f7385.js.map +1 -0
- package/dist/{judge-calibration-C5CbMYce.d.ts → judge-calibration-DFtEMlde.d.ts} +31 -2
- package/dist/judge-calibration-DFtEMlde.d.ts.map +1 -0
- package/dist/{judge-calibration-BnpVKtnb.js → judge-calibration-DYmaBtJr.js} +48 -2
- package/dist/{judge-calibration-BnpVKtnb.js.map → judge-calibration-DYmaBtJr.js.map} +1 -1
- package/dist/ledger-core/index.d.ts +1 -1
- package/dist/ledger-core/index.js +1 -1
- package/dist/{llm-judge-v80Kmu9g.js → llm-judge-DEFZeSiu.js} +645 -456
- package/dist/llm-judge-DEFZeSiu.js.map +1 -0
- package/dist/{matrix-DeMmnWrP.d.ts → matrix-CyhW-vgJ.d.ts} +2 -2
- package/dist/{matrix-DeMmnWrP.d.ts.map → matrix-CyhW-vgJ.d.ts.map} +1 -1
- package/dist/meta-eval/index.d.ts +138 -7
- package/dist/meta-eval/index.d.ts.map +1 -1
- package/dist/meta-eval/index.js +245 -97
- package/dist/meta-eval/index.js.map +1 -1
- package/dist/{mint-Cc1_zwRQ.js → mint-ySIIkKlV.js} +2 -2
- package/dist/{mint-Cc1_zwRQ.js.map → mint-ySIIkKlV.js.map} +1 -1
- package/dist/multishot/golden/index.d.ts +1 -1
- package/dist/multishot/index.d.ts +2 -2
- package/dist/openapi.json +1 -1
- package/dist/outcome-store-BXlkwMPR.js +131 -0
- package/dist/outcome-store-BXlkwMPR.js.map +1 -0
- package/dist/{outcome-store-BYHIuO0e.d.ts → outcome-store-CNt4iZ67.d.ts} +18 -25
- package/dist/outcome-store-CNt4iZ67.d.ts.map +1 -0
- package/dist/{paired-promotion-decision-CGzg0cI_.d.ts → paired-promotion-decision-DPsMQm-0.d.ts} +13 -7
- package/dist/{paired-promotion-decision-CGzg0cI_.d.ts.map → paired-promotion-decision-DPsMQm-0.d.ts.map} +1 -1
- package/dist/pipelines/index.js +1 -1
- package/dist/{produced-state-Cv0kJJuP.js → produced-state-BHboMaab.js} +3 -3
- package/dist/{produced-state-Cv0kJJuP.js.map → produced-state-BHboMaab.js.map} +1 -1
- package/dist/profile-cell.d.ts +1 -1
- package/dist/profile-cell.js +1 -1
- package/dist/{promotion-policy-DWOm70gx.js → promotion-policy-CDMMxzb6.js} +28 -40
- package/dist/promotion-policy-CDMMxzb6.js.map +1 -0
- package/dist/{registry-ByVld1-5.d.ts → registry-BRbB6Y0v.d.ts} +4 -4
- package/dist/{registry-ByVld1-5.d.ts.map → registry-BRbB6Y0v.d.ts.map} +1 -1
- package/dist/{release-confidence-BAcNYOf1.d.ts → release-confidence-BcqeQTHW.d.ts} +3 -3
- package/dist/{release-confidence-BAcNYOf1.d.ts.map → release-confidence-BcqeQTHW.d.ts.map} +1 -1
- package/dist/{release-confidence-BcGCclTB.js → release-confidence-DMg8n18l.js} +2 -2
- package/dist/{release-confidence-BcGCclTB.js.map → release-confidence-DMg8n18l.js.map} +1 -1
- package/dist/{report-command-DKlXfU5r.js → report-command-V1ecVgAv.js} +27 -3
- package/dist/report-command-V1ecVgAv.js.map +1 -0
- package/dist/reporting.d.ts +4 -4
- package/dist/reporting.js +3 -3
- package/dist/{researcher-jsW1X94L.d.ts → researcher-64T49THL.d.ts} +6 -6
- package/dist/{researcher-jsW1X94L.d.ts.map → researcher-64T49THL.d.ts.map} +1 -1
- package/dist/{reward-hacking-ZXEi9VCq.d.ts → reward-hacking-uzO_ihep.d.ts} +2 -2
- package/dist/{reward-hacking-ZXEi9VCq.d.ts.map → reward-hacking-uzO_ihep.d.ts.map} +1 -1
- package/dist/rl.d.ts +53 -99
- package/dist/rl.d.ts.map +1 -1
- package/dist/rl.js +182 -169
- package/dist/rl.js.map +1 -1
- package/dist/rollout/index.d.ts +1 -1
- package/dist/rollout/index.js +2 -2
- package/dist/{rollout-DmoJVqrF.js → rollout-B-UF5R6w.js} +2 -2
- package/dist/{rollout-DmoJVqrF.js.map → rollout-B-UF5R6w.js.map} +1 -1
- package/dist/rubric-predictive-validity-Bmj2_cll.d.ts +79 -0
- package/dist/rubric-predictive-validity-Bmj2_cll.d.ts.map +1 -0
- package/dist/rubric-predictive-validity-CCK-1B7w.js +178 -0
- package/dist/rubric-predictive-validity-CCK-1B7w.js.map +1 -0
- package/dist/{run-record-DTv1MdjK.d.ts → run-record-BiTWauyO.d.ts} +2 -2
- package/dist/{run-record-DTv1MdjK.d.ts.map → run-record-BiTWauyO.d.ts.map} +1 -1
- package/dist/run-record-Br-Yzt_k.js +464 -0
- package/dist/run-record-Br-Yzt_k.js.map +1 -0
- package/dist/{run-record-DQpSf7t-.js → run-record-DualPTn2.js} +2 -2
- package/dist/{run-record-DQpSf7t-.js.map → run-record-DualPTn2.js.map} +1 -1
- package/dist/{semantic-concept-judge-Bi6_iGqg.js → semantic-concept-judge-Bm5JDEKO.js} +3 -3
- package/dist/{semantic-concept-judge-Bi6_iGqg.js.map → semantic-concept-judge-Bm5JDEKO.js.map} +1 -1
- package/dist/{sequential-B5gXgcyp.js → sequential-DAsyV2T9.js} +42 -25
- package/dist/sequential-DAsyV2T9.js.map +1 -0
- package/dist/{series-convergence-DeG33RpC.d.ts → series-convergence-BnMs_uAr.d.ts} +3 -3
- package/dist/{series-convergence-DeG33RpC.d.ts.map → series-convergence-BnMs_uAr.d.ts.map} +1 -1
- package/dist/{skillopt-optimization-method-C3oYul8v.js → skillopt-optimization-method-CL_0aArC.js} +5 -5
- package/dist/{skillopt-optimization-method-C3oYul8v.js.map → skillopt-optimization-method-CL_0aArC.js.map} +1 -1
- package/dist/{statistical-heldout-0La5ZTlv.d.ts → statistical-heldout-CpVd6FmY.d.ts} +207 -144
- package/dist/statistical-heldout-CpVd6FmY.d.ts.map +1 -0
- package/dist/{store-tool-spans-4J1EDElP.d.ts → store-tool-spans-Dt-YdAuE.d.ts} +6 -6
- package/dist/{store-tool-spans-4J1EDElP.d.ts.map → store-tool-spans-Dt-YdAuE.d.ts.map} +1 -1
- package/dist/{summary-report-gMrbYawB.d.ts → summary-report-D1h4dlrK.d.ts} +3 -3
- package/dist/{summary-report-gMrbYawB.d.ts.map → summary-report-D1h4dlrK.d.ts.map} +1 -1
- package/dist/{summary-report-B16xy9Kd.js → summary-report-e-MaOAHV.js} +2 -2
- package/dist/{summary-report-B16xy9Kd.js.map → summary-report-e-MaOAHV.js.map} +1 -1
- package/dist/supervisor-run/index.d.ts +4 -2
- package/dist/supervisor-run/index.d.ts.map +1 -1
- package/dist/supervisor-run/index.js +3 -3
- package/dist/{terminal-record-Ce9_UjRz.js → terminal-record-BtPwKTSr.js} +58 -26
- package/dist/terminal-record-BtPwKTSr.js.map +1 -0
- package/dist/{tool-groups-2QA0S7dK.d.ts → tool-groups-B2bSNaJB.d.ts} +3 -3
- package/dist/tool-groups-B2bSNaJB.d.ts.map +1 -0
- package/dist/{tool-waste-B9tdWV6g.js → tool-waste-C7MU9u1e.js} +2 -2
- package/dist/{tool-waste-B9tdWV6g.js.map → tool-waste-C7MU9u1e.js.map} +1 -1
- package/dist/trace-repair/index.d.ts +2 -2
- package/dist/traces.d.ts +6 -6
- package/dist/traces.js +1 -1
- package/dist/{types-BmlkCrg0.d.ts → types-BvZoPTGa.d.ts} +3 -3
- package/dist/{types-BmlkCrg0.d.ts.map → types-BvZoPTGa.d.ts.map} +1 -1
- package/dist/{types-gvRsyJLh.d.ts → types-CBbLtr2J.d.ts} +38 -3
- package/dist/{types-gvRsyJLh.d.ts.map → types-CBbLtr2J.d.ts.map} +1 -1
- package/dist/{types-C34V4Vto.d.ts → types-CS0qk_Yp.d.ts} +4 -4
- package/dist/{types-C34V4Vto.d.ts.map → types-CS0qk_Yp.d.ts.map} +1 -1
- package/dist/{types-DzuaM493.d.ts → types-D7gEdPoQ.d.ts} +3 -3
- package/dist/{types-DzuaM493.d.ts.map → types-D7gEdPoQ.d.ts.map} +1 -1
- package/dist/{types-vUdAx2Cj.d.ts → types-lPkDQNqJ.d.ts} +20 -2
- package/dist/{types-vUdAx2Cj.d.ts.map → types-lPkDQNqJ.d.ts.map} +1 -1
- package/dist/wire/index.d.ts +2 -2
- package/docs/adapters-observability.md +14 -0
- package/docs/campaign-proposers.md +86 -128
- package/docs/charter.md +108 -112
- package/docs/concepts.md +157 -69
- package/docs/design/mlbenchmarks-book-review.md +440 -0
- package/docs/design/mlbenchmarks-review/observations.json +713 -0
- package/docs/design/mlbenchmarks-review/probes.mts +476 -0
- package/docs/design/mlbenchmarks-review/sources.json +200 -0
- package/docs/design/self-improvement-evidence-audit.md +263 -0
- package/docs/design.md +2 -1
- package/docs/eval-surface-map.md +95 -42
- package/docs/evaluation-integrity.md +220 -0
- package/docs/experiment.md +111 -55
- package/docs/feature-guide.md +5 -6
- package/docs/hosted-ingest-spec.md +4 -11
- package/docs/insight-report.md +187 -455
- package/docs/outcome-validity.md +182 -0
- package/docs/product-eval-adoption.md +1 -2
- package/docs/research-report-methodology.md +7 -7
- package/docs/search-history-receipts.md +8 -0
- package/docs/statistical-evidence.md +129 -0
- package/docs/verdicts.md +76 -49
- package/package.json +1 -1
- package/dist/agent-profile-cell-0gSi5ffD.js.map +0 -1
- package/dist/agent-profile-cell-CTOZJUuE.d.ts.map +0 -1
- package/dist/campaign-evidence-D8DBLqLI.js.map +0 -1
- package/dist/client-BlLY6o2w.js.map +0 -1
- package/dist/define-agent-eval-Cx4Ls9ta.d.ts.map +0 -1
- package/dist/define-agent-eval-Dzidv34q.js.map +0 -1
- package/dist/engine-CX8ReXkn.d.ts.map +0 -1
- package/dist/index-BxWvILU8.d.ts.map +0 -1
- package/dist/judge-calibration-C5CbMYce.d.ts.map +0 -1
- package/dist/ledger-core-Cs9f7385.js.map +0 -1
- package/dist/llm-judge-v80Kmu9g.js.map +0 -1
- package/dist/outcome-store-BYHIuO0e.d.ts.map +0 -1
- package/dist/outcome-store-ChBKlTd_.js +0 -75
- package/dist/outcome-store-ChBKlTd_.js.map +0 -1
- package/dist/promotion-policy-DWOm70gx.js.map +0 -1
- package/dist/report-command-DKlXfU5r.js.map +0 -1
- package/dist/rubric-predictive-validity-2D5Gw9z9.js +0 -131
- package/dist/rubric-predictive-validity-2D5Gw9z9.js.map +0 -1
- package/dist/rubric-predictive-validity-Dl1dvKCv.d.ts +0 -75
- package/dist/rubric-predictive-validity-Dl1dvKCv.d.ts.map +0 -1
- package/dist/run-record-CR63CpHK.js +0 -216
- package/dist/run-record-CR63CpHK.js.map +0 -1
- package/dist/sequential-B5gXgcyp.js.map +0 -1
- package/dist/statistical-heldout-0La5ZTlv.d.ts.map +0 -1
- package/dist/terminal-record-Ce9_UjRz.js.map +0 -1
- package/dist/tool-groups-2QA0S7dK.d.ts.map +0 -1
|
@@ -0,0 +1,440 @@
|
|
|
1
|
+
# What the benchmark book changes for agent-eval
|
|
2
|
+
|
|
3
|
+
Agent-eval should make the intended claim, evaluation population, and evidence lifecycle explicit before adding more autonomous search.
|
|
4
|
+
Its existing statistics and execution records provide a strong base.
|
|
5
|
+
The largest opportunity is to connect those instruments into a system that can propose, test, and revise evaluations independently of the agent being improved.
|
|
6
|
+
|
|
7
|
+
This assessment is for agent-eval maintainers deciding what to build next.
|
|
8
|
+
It reviews Moritz Hardt’s [The Emerging Science of Machine Learning Benchmarks](https://mlbenchmarks.org/) against repository revision `fe1cc5111aab5d588bf7db3a3785325635937a91`.
|
|
9
|
+
The exact inspected revision is also recorded in the [source manifest](./mlbenchmarks-review/sources.json).
|
|
10
|
+
The review date is September 12, 2026, in America/Los_Angeles.
|
|
11
|
+
Recommendations below are design proposals, not measured improvements or implemented runtime changes.
|
|
12
|
+
|
|
13
|
+
## Decision
|
|
14
|
+
|
|
15
|
+
Retain the current package boundaries and measurement primitives.
|
|
16
|
+
Prioritize three additions:
|
|
17
|
+
|
|
18
|
+
1. Bind each certification to its target population, independent observation unit, and intended use.
|
|
19
|
+
2. Track final-data exposure across campaigns, including what feedback reached candidate authors.
|
|
20
|
+
3. Admit generated evaluations through independent calibration and challenge before they can authorize agent improvements.
|
|
21
|
+
|
|
22
|
+
First repair the narrow measurement defects documented below.
|
|
23
|
+
An automated researcher that consumes the wrong outcome metric can optimize in the wrong direction faster than a human reviewer notices.
|
|
24
|
+
|
|
25
|
+
Do not start with another optimizer, runner, statistics package, or general benchmark leaderboard.
|
|
26
|
+
Do not treat a high judge agreement score as proof that a model ranking is correct.
|
|
27
|
+
Do not make an evaluator’s own approval rate its optimization objective.
|
|
28
|
+
|
|
29
|
+
## Reading scope and method
|
|
30
|
+
|
|
31
|
+
The live index exposes **16 reading pages**, totaling **109,771 whitespace-delimited body words**.
|
|
32
|
+
That count includes notes, references, tables, and equation text.
|
|
33
|
+
We read all 16: preface, prologue, chapters 1–8, and chapters 10–15.
|
|
34
|
+
Three parallel reviews covered the full text; the repository mapping checked implementations, exports, relevant callers, tests, and offline behavior.
|
|
35
|
+
HTML text omitted by paragraph extraction was separately audited and read.
|
|
36
|
+
Figures and tables supporting the conclusions were checked against their HTML or PDF context.
|
|
37
|
+
|
|
38
|
+
**Chapter 9 remains unavailable.**
|
|
39
|
+
The book references an annotation chapter, but the live index jumps from 8 to 10.
|
|
40
|
+
The checked `/09-annotation.html`, `/09-data-annotation.html`, and `/09-annotations.html` paths returned HTTP 404.
|
|
41
|
+
This review does not claim coverage of an unavailable chapter or the forthcoming print edition.
|
|
42
|
+
We read chapter references as part of the book; we did not independently reproduce every cited study or read every cited paper.
|
|
43
|
+
|
|
44
|
+
The [manifest](./mlbenchmarks-review/sources.json) records every page URL, title, body count, and downloaded HTML SHA-256.
|
|
45
|
+
It preserves source identity without vendoring the book.
|
|
46
|
+
The [offline probes](./mlbenchmarks-review/probes.mts) and [observations](./mlbenchmarks-review/observations.json) preserve the local checks behind concrete findings.
|
|
47
|
+
The probes use deterministic or synthetic data and make zero paid model calls.
|
|
48
|
+
They establish behavior and assumption boundaries, not production defect rates or expected improvement sizes.
|
|
49
|
+
|
|
50
|
+
The search covered `src`, `tests`, `docs`, `examples`, and package exports at the inspected revision.
|
|
51
|
+
It did not audit deployed consumers, production traffic, or the implementation of agent-runtime and agent-knowledge.
|
|
52
|
+
An absent connection here may already exist in a downstream application.
|
|
53
|
+
|
|
54
|
+
## The book, chapter by chapter
|
|
55
|
+
|
|
56
|
+
The book supplies several kinds of support.
|
|
57
|
+
Mathematical results depend on their assumptions; benchmark studies describe particular settings; historical interpretations suggest mechanisms worth testing.
|
|
58
|
+
The architecture proposals are our applications of those sources.
|
|
59
|
+
|
|
60
|
+
| Reading | Main contribution and qualification | Consequence for agent-eval |
|
|
61
|
+
| --- | --- | --- |
|
|
62
|
+
| [Preface](https://mlbenchmarks.org/00-preface.html) | Frames benchmarks as both development instruments and scientific institutions. The book is a synthesis, not an agent evaluation implementation guide. | Evaluate the measurement process and its incentives, alongside individual agents. |
|
|
63
|
+
| [Prologue](https://mlbenchmarks.org/00-prologue.html) | Uses a learning-rate anecdote to question the search for a universal modeling trick. Its empirical lesson unfolds in later chapters. | Preserve empirical comparison and let the researcher choose methods. |
|
|
64
|
+
| [1. Introduction](https://mlbenchmarks.org/01-introduction.html#the-iron-rule) | Explains benchmark-driven competition through the ImageNet and language-model eras. Benchmark success and explanatory scientific progress are different achievements. | Keep mechanism tests and replication alongside winner selection. |
|
|
65
|
+
| [2. Populations and predictions](https://mlbenchmarks.org/02-populations-predictions.html#errors-and-metrics) | Defines prediction risk relative to a population, loss, and decision problem. Calibration, accuracy, precision, and recall answer different questions. | Record the target distribution and error consequences; a scenario digest does not establish population validity. |
|
|
66
|
+
| [3. Detecting differences](https://mlbenchmarks.org/03-detecting-differences.html#comparing-similar-models) | Develops sample requirements, iid assumptions, multiplicity, and discordance in paired correctness outcomes. Small differences can require substantial independent evidence. | Reuse paired and discordance-aware power checks; distinguish task count from repeated execution count. |
|
|
67
|
+
| [4. Holdout method](https://mlbenchmarks.org/04-holdout-method.html#whats-the-holdout-method-for) | Separates development feedback, model ranking, and capability measurement. Valid intervals within a dataset do not establish external validity. | Give exploratory reports, fixed-roster comparisons, and deployment claims different evidence requirements. |
|
|
68
|
+
| [5. Test set reuse](https://mlbenchmarks.org/05-test-set-reuse.html#guarantees-of-the-holdout-method-under-adaptivity) | Shows why adaptive feedback changes holdout guarantees. Worst-case attacks demonstrate possibility, not the prevalence of practical overfitting. | Track exposure and adaptive claim history across calls; preserve useful development feedback. |
|
|
69
|
+
| [6. Scientific crisis](https://mlbenchmarks.org/06-scientific-crisis.html#researcher-degrees-of-freedom) | Explains selection, low power, publication incentives, and researcher flexibility. A p-value is not the probability that a claim is true. | Retain failed attempts and amended rules, practical effect sizes, controls, and unresolved outcomes. |
|
|
70
|
+
| [7. Replication in machine learning](https://mlbenchmarks.org/07-replication-machine-learning.html#measurement-versus-ranking) | Studies cases where new test sets shift absolute accuracy while preserving much of the ranking. Evidence from ImageNet does not guarantee language-agent stability. | Distinguish rerunning identical artifacts from sampling new tasks and reproducing the conclusion independently. |
|
|
71
|
+
| [8. Forces against crisis](https://mlbenchmarks.org/08-forces-against-crisis.html#biases-and-heuristics) | Examines leaderboard mechanisms, human information filtering, and shared code. These partly explain empirical robustness; they are not universal protections. | Autonomous search needs explicit feedback policies because it may exploit details that humans ignored. |
|
|
72
|
+
| [10. Generative models](https://mlbenchmarks.org/10-generative-models.html#the-limits-of-scaling-laws) | Connects language modeling, scaling, training distributions, and downstream benchmarks. Better likelihood or fitted scaling laws need not establish product capability. | Measure complete executable profiles and user outcomes; keep model-level proxies in their stated role. |
|
|
73
|
+
| [11. Evaluating language models](https://mlbenchmarks.org/11-evaluating-language-models.html#confounded-evaluations) | Covers post-training, generative judges, shortcuts, contamination, and tune-before-test. Unequal task preparation can confound claims about base-model capability. | Declare whether the comparison concerns deployed products or adaptation potential; account for preparation when that claim requires it. |
|
|
74
|
+
| [12. The problem of aggregation](https://mlbenchmarks.org/12-problem-aggregation.html#problems-of-aggregation-and-voting-systems) | Uses social choice and empirical comparisons to expose ranking tradeoffs. No theorem says every task-specific aggregate is useless. | Preserve dimensions, target weights, subgroup denominators, and sensitivity to defensible alternative aggregation policies. |
|
|
75
|
+
| [13. When the model moves the data](https://mlbenchmarks.org/13-model-moves-data.html#what-performativity-means-for-model-evaluation) | Models deployments that change future data. Stability, optimality, and welfare differ; feedback-loop stories require evidence. | Record assignment, time, exposure, and affected populations; distinguish monitoring correlation from causal deployment effects. |
|
|
76
|
+
| [14. Evaluation at the frontier](https://mlbenchmarks.org/14-evaluation-frontier.html#agreement-alone-is-not-enough) | Shows why judge agreement can coexist with wrong rankings. Discusses debiasing, verification, simulation, and live experiments, each with limits. | Calibrate ranking errors against independent labels, challenge evaluators, and connect offline decisions to later outcomes. |
|
|
77
|
+
| [15. Epilogue](https://mlbenchmarks.org/15-epilogue.html) | Returns to the social and scientific choices behind measurement. More automation does not remove judgment about desirable outcomes. | Let domain owners define value and acceptable failures; automate evidence collection and scrutiny. |
|
|
78
|
+
|
|
79
|
+
### What transfers, and what does not
|
|
80
|
+
|
|
81
|
+
The most useful distinction is **development signal versus ranking versus capability certification**.
|
|
82
|
+
These uses require progressively stronger evidence.
|
|
83
|
+
A regression suite can be useful after repeated exposure without supporting a fresh claim about unseen tasks.
|
|
84
|
+
A ranking can reproduce across populations while every absolute success rate changes.
|
|
85
|
+
A perfectly reproducible computation can measure the wrong construct.
|
|
86
|
+
|
|
87
|
+
Chapter 8 suggests a specific automation risk.
|
|
88
|
+
Human researchers often discarded most benchmark feedback through heuristics and limited attention.
|
|
89
|
+
An autonomous optimizer can retain every score, failed attempt, trace, and explanation.
|
|
90
|
+
The book motivates testing whether that additional feedback increases overfitting; it does not establish that our optimizers currently do so.
|
|
91
|
+
Its [Ladder mechanism](https://mlbenchmarks.org/08-forces-against-crisis.html#leaderboard-error) releases score updates only after sufficiently large improvements.
|
|
92
|
+
The guarantees depend on the specified mechanism and observation assumptions.
|
|
93
|
+
A minimum-effect gate alone does not reproduce them.
|
|
94
|
+
Keep complete private audit evidence even when an optimizer receives restricted feedback.
|
|
95
|
+
|
|
96
|
+
Chapter 11 also requires a careful distinction.
|
|
97
|
+
Comparing two products with their actual prompts and tools is appropriate when those products are the alternatives being deployed.
|
|
98
|
+
Comparing underlying models’ learning potential may require equal task preparation, adaptation curves, and total preparation cost.
|
|
99
|
+
Automatically tuning every model would change the first question into the second.
|
|
100
|
+
|
|
101
|
+
At the frontier, independent agreement and formal verification remain conditional evidence.
|
|
102
|
+
Two agents can share a blind spot.
|
|
103
|
+
A proof kernel checks a formal statement, leaving the connection to the intended claim as a separate obligation.
|
|
104
|
+
The package’s [verification strategy model](../verification-strategies.md) already captures this distinction.
|
|
105
|
+
|
|
106
|
+
## What agent-eval already has
|
|
107
|
+
|
|
108
|
+
These are implementation findings, not an assessment of adoption in every consumer.
|
|
109
|
+
“Partial” means the named behavior exists but leaves a specific contract or integration gap.
|
|
110
|
+
|
|
111
|
+
| Concern | Checked implementation | Assessment and remaining gap |
|
|
112
|
+
| --- | --- | --- |
|
|
113
|
+
| Statistical comparisons | [statistics](../../src/statistics/index.ts), [paired decisions](../../src/paired-promotion-decision.ts), [heldout pairing](../../src/campaign/gates/statistical-heldout.ts) | Present. Paired tests, uncertainty, exact binary methods, multiplicity, and power do not need replacement. Claim scope and independent sampling units need stronger binding. |
|
|
114
|
+
| Cluster-aware design | [power](../../src/experiment/power.ts), [registered rule AST](../../src/experiment/ast.ts) | Present. The high-level campaign path does not automatically select these methods from a declared generalization target. |
|
|
115
|
+
| Executable preregistration | [define/seal/open](../../src/experiment/define.ts), [acceptance tests](../../tests/experiment/preregistration-acceptance.test.ts) | Present. Extend this rule representation instead of creating a second experiment language. |
|
|
116
|
+
| Sequential testing | [sequential gate](../../src/campaign/gates/sequential.ts), [e-process](../../src/statistics/sequential-eprocess.ts) | Present. Optional stopping within a valid stream differs from adapting hypotheses across streams or counting dependent replicas as independent evidence. |
|
|
117
|
+
| Search/final separation | [selfImprove](../../src/contract/self-improve.ts), [method comparison](../../src/campaign/presets/compare-optimization-methods.ts) | Present within calls. Final cases are withheld from method inputs. A persistent cross-campaign exposure policy is missing from these paths. |
|
|
118
|
+
| Dataset identity and access | [Dataset](../../src/dataset.ts), [contamination helpers](../../src/contamination-guard.ts), [labeled store](../../src/campaign/labeled-store/fs-adapter.ts) | Partial. Hashes, split labels, mutation locks, temporal sampling, and access logs exist. They do not establish secrecy, lineage independence, or one-time certification use. |
|
|
119
|
+
| Complete search history | [SearchLedger](../../src/campaign/search-ledger.ts), [history receipt documentation](../search-history-receipts.md) | Present. Reuse the canonical ledger; do not create another optimizer event log. Requiring complete history remains a caller policy. |
|
|
120
|
+
| Evidence identity and authority | [EvidenceReceipt](../../src/experiment/evidence-receipt.ts), [campaign receipts](../../src/experiment/campaign-evidence.ts), [registry](../../src/experiment/evidence-record.ts) | Present. These bind identities and preserve declared authority. Hashes and authority labels alone do not prove independent execution or valid sampling. |
|
|
121
|
+
| Judge quality and drift | [calibration](../../src/judge-calibration.ts), [sentinel](../../src/meta-eval/sentinel.ts), [plants](../../src/meta-eval/plants.ts) | Present. Per-candidate residual bias, ranking validity, and independent evaluator admission need composition. Position and self-preference helpers exist internally; only verbosity has a root public export. |
|
|
122
|
+
| Multiple objectives | [promotion policy](../../src/campaign/gates/promotion-policy.ts), [production gate](../../src/campaign/gates/default-production-gate.ts) | Present. Per-dimension regression guards and evidence vectors exist. Explicit target-population weights and aggregation sensitivity remain useful additions. |
|
|
123
|
+
| Outcome validity | [correlation study](../../src/meta-eval/correlation-study.ts), [rubric validity](../../src/meta-eval/rubric-predictive-validity.ts), [outcome store](../../src/meta-eval/outcome-store.ts) | Present. Repair the metric-selection defect below; define direction and observational limits before using correlations for automated steering. |
|
|
124
|
+
| Adaptation and causal primitives | [adaptation evaluation](../../src/rl/adaptation-eval.ts), [off-policy estimators](../../src/rl/off-policy.ts) | Present. The adaptation comparison needs pairing repair. IPS, SNIPS, and doubly robust estimation still depend on supplied propensities, overlap, and identification assumptions. |
|
|
125
|
+
| Generating reusable cases | [fixtures](../../src/campaign/fixtures.ts), [feedback trajectories](../../src/feedback-trajectory.ts), [analysts](../../src/analyst/index.ts) | Partial. Cases can be authored, replayed, and scored. The inspected package lacks a complete admission protocol for an autonomously authored evaluation. |
|
|
126
|
+
| Active and adversarial case selection | [curriculum](../../src/rl/active-curriculum.ts), [adversarial scenarios](../../src/rl/adversarial.ts), [fuzzing](../../src/fuzz/fuzz-agent.ts), [discrimination](../../src/campaign/scenario-selection.ts) | Present. Extend population and exposure accounting around these primitives; do not propose a first automatic case-generation loop. |
|
|
127
|
+
| Automated improvement | [contract](../../src/contract/index.ts), [Researcher](../../src/researcher.ts), [predictive-validity researcher](../../src/rl/predictive-validity-researcher.ts) | Partial. Optimizer adapters and inspect/propose/apply/evaluate contracts exist. The predictive-validity researcher recommends changes but does not execute plans. |
|
|
128
|
+
| Verification without answer keys | [strategy/checker port](../../src/verification-strategy.ts), [verdicts](../../src/verdict.ts), [repair grading](../../src/trace-repair/index.ts) | Present as contracts and applicable execution paths. Domain checkers remain injected; a universal verifier is neither provided nor justified. |
|
|
129
|
+
|
|
130
|
+
The [charter](../charter.md) is a dated inventory with some later additions described beneath its original missing list.
|
|
131
|
+
Current source already supplies cluster-aware power, sealed rules, funnels, and evidence receipts.
|
|
132
|
+
Treat those as foundations to connect, not unbuilt modules.
|
|
133
|
+
|
|
134
|
+
## Concrete findings from offline checks
|
|
135
|
+
|
|
136
|
+
The observations below are intentionally narrower than production reliability claims.
|
|
137
|
+
They use the actual library functions at the inspected revision.
|
|
138
|
+
The [probe source](./mlbenchmarks-review/probes.mts) contains the complete inputs and invocation paths.
|
|
139
|
+
The archived probes target review snapshot `dda9941437190c9c541b3f54946bfeeb153366fe`, whose implementation matches the inspected source.
|
|
140
|
+
Run `pnpm exec tsx docs/design/mlbenchmarks-review/probes.mts` there after installing the locked dependencies.
|
|
141
|
+
Use an isolated checkout without concurrent source edits.
|
|
142
|
+
Current APIs have breaking changes, so these historical probes cannot run unchanged against current implementation code.
|
|
143
|
+
Current regressions and the [integrity example](../../examples/evaluation-integrity/) verify the replacement behavior.
|
|
144
|
+
It prints current observations without asserting that the recorded defects must persist.
|
|
145
|
+
The source identity hashes actual files under `src`, plus `package.json`, `pnpm-lock.yaml`, and `tsconfig.json`.
|
|
146
|
+
A separate hash identifies the diagnostic itself.
|
|
147
|
+
These identities survive documentation commits and change with local source edits, including untracked files.
|
|
148
|
+
They assume dependencies were installed from the lockfile; they do not fingerprint installed packages or the host environment.
|
|
149
|
+
|
|
150
|
+
| Finding | Observed result | Consequence and bounded correction |
|
|
151
|
+
| --- | --- | --- |
|
|
152
|
+
| Final evidence can be reused across independent calls | Two `selfImprove()` calls each dispatched baseline and candidate on the same six final cases: 12 final dispatches per call. Both returned `ship`, with the same final-set digest. | Confirms no cross-call consumption guard on this path. Add exposure-aware certification policy; this probe does not demonstrate empirical overfitting. |
|
|
153
|
+
| Outcome correlation can select the wrong metric | Ten synthetic runs had `csat = score` and an earlier object key with the opposite trend. Default `latest` returned correlation −1 for `csat`; `mean` returned +1. | [The reducer](../../src/meta-eval/correlation-study.ts) does not receive the requested metric name. Preserve metric identity when selecting the latest eligible outcome. |
|
|
154
|
+
| Adaptation comparison accepts unrelated tasks | Curves with disjoint scenario IDs can return `a_better`. The implementation computes separate marginal intervals and compares point summaries. | [The comparison](../../src/rl/adaptation-eval.ts) claims pairing without joining IDs. Use the existing paired machinery, report missing pairs, and distinguish descriptive curves from release evidence. |
|
|
155
|
+
| Exchangeability alone cannot validate the sequential gate | Choose one fair sign per experiment and repeat it for 100 cells. The positive state promotes after 15 observations; the negative state does not. | [The gate commentary](../../src/campaign/gates/sequential.ts) overstates exchangeability and shuffling. Under this marginal-zero construction, false promotion is 50%; the required conditional-mean assumption fails. |
|
|
156
|
+
|
|
157
|
+
The sequential example is an exact two-state counterexample, not a Monte Carlo estimate.
|
|
158
|
+
It challenges the stated assumption boundary, not the conditional-mean theorem underlying an e-process.
|
|
159
|
+
Shuffling correlated replicas does not manufacture independent tasks.
|
|
160
|
+
|
|
161
|
+
Two additional source findings matter before automating interpretation.
|
|
162
|
+
In [rubric predictive validity](../../src/meta-eval/rubric-predictive-validity.ts), the `load_bearing` classification uses absolute correlation.
|
|
163
|
+
Strong negative association can therefore earn that label; tests make magnitude-based behavior intentional despite contradictory interface prose.
|
|
164
|
+
Require an explicit outcome direction before treating this label as a recommendation to increase rubric weight.
|
|
165
|
+
This already affects [PredictiveValidityResearcher](../../src/rl/predictive-validity-researcher.ts).
|
|
166
|
+
Given nonempty failures, it can recommend up-weighting the top `load_bearing` rubric without checking the correlation’s sign.
|
|
167
|
+
The correction must reach that consumer as well as the report vocabulary.
|
|
168
|
+
|
|
169
|
+
In the [contamination probe](../../src/rl/contamination.ts), per-item `qValue` is derived from `1 - abs(delta)` before adjustment.
|
|
170
|
+
That quantity has no demonstrated p-value calibration.
|
|
171
|
+
The global paired Wilcoxon calculation is separate and should not be conflated with these display values.
|
|
172
|
+
Remove inferential naming from the heuristic or introduce a justified repeated-sample model.
|
|
173
|
+
Perturbation sensitivity can reflect changed difficulty as well as contamination.
|
|
174
|
+
|
|
175
|
+
The open safety-floor change was a separate worktree and PR during this review.
|
|
176
|
+
This assessment does not duplicate ownership of that gate or assume its changes were present in the inspected base.
|
|
177
|
+
|
|
178
|
+
## Prioritized improvements
|
|
179
|
+
|
|
180
|
+
### 1. Bind the claim to the design
|
|
181
|
+
|
|
182
|
+
**Recommendation: adapt; highest architectural priority.**
|
|
183
|
+
Extend the existing experiment definition with claim metadata and mechanically checked obligations.
|
|
184
|
+
Do not create a new runner or parallel estimator family.
|
|
185
|
+
|
|
186
|
+
The record should name the intended use, target population, sampling frame, observation unit, clustering, timeframe, outcome direction, and practical effect threshold.
|
|
187
|
+
It should identify whether the population is fixed, sampled independently, or affected by deployment.
|
|
188
|
+
It should retain exclusions, selection probabilities when known, and reasons when they are unknown.
|
|
189
|
+
|
|
190
|
+
For a frozen task roster, repeated executions can estimate execution variability conditional on that roster.
|
|
191
|
+
For unseen-task claims, resample independent tasks or task families through the existing cluster-aware methods.
|
|
192
|
+
Variants derived from one incident should retain their common source identity.
|
|
193
|
+
Twenty rewrites of one failure are not twenty independent examples of customer demand.
|
|
194
|
+
The method-comparison path reduces repetitions to scenario means; the primary heldout gate counts `scenario:rep` cells.
|
|
195
|
+
Make that choice follow the claim instead of treating either convention as universally correct.
|
|
196
|
+
|
|
197
|
+
Preflight power at the minimum worthwhile effect using the actual registered decision procedure.
|
|
198
|
+
The existing `power-floor` gate asks whether maximum power anywhere on its supplied effect grid reaches the target.
|
|
199
|
+
A fixture with power 0.1 at effect 0.01 and power 1 at effect 1 passes its target-0.8 check.
|
|
200
|
+
Those values are supplied curve points, not measured operating characteristics.
|
|
201
|
+
The result matches the gate’s structural-feasibility semantics; it does not establish adequacy for detecting an effect of 0.01.
|
|
202
|
+
|
|
203
|
+
**Smallest decisive check:** compare one task repeated 100 times with 100 independent tasks.
|
|
204
|
+
Both designs should expose the same execution count and different independent-unit counts.
|
|
205
|
+
A claim about unseen tasks must not acquire precision merely by duplicating the first task.
|
|
206
|
+
Reject this addition if the same safety and clarity can be obtained by composing existing typed fields without a new contract.
|
|
207
|
+
|
|
208
|
+
### 2. Account for evidence exposure across campaigns
|
|
209
|
+
|
|
210
|
+
**Recommendation: adapt; prerequisite for autonomous certification.**
|
|
211
|
+
Bind final-set commitments, task-family lineage, claim identity, and released feedback to the canonical ledger.
|
|
212
|
+
The host should reserve and consume final evidence through a durable operation with retry identity.
|
|
213
|
+
The record must distinguish replaying one completed measurement from opening evidence for a new adaptive decision.
|
|
214
|
+
|
|
215
|
+
Keep ordinary regression and exploratory reuse available and labeled.
|
|
216
|
+
For certification, support a frozen comparison, registered sequential collection of new valid observations, or refreshed final data.
|
|
217
|
+
A seed change or new run directory does not make an exposed task fresh.
|
|
218
|
+
Concurrent hosts must not each mint an apparently unused reservation for the same claim and evidence.
|
|
219
|
+
|
|
220
|
+
The runtime owns file permissions, model context, storage credentials, and separation between authors and final evaluators.
|
|
221
|
+
Eval owns the portable access/consumption record and the certification refusal.
|
|
222
|
+
An append-only record cannot prove secrecy if the host lets the author read the answer files.
|
|
223
|
+
|
|
224
|
+
**Smallest decisive check:** reproduce the two-call probe with one final-set reservation shared across restarts and two competing workers.
|
|
225
|
+
Require an explicit reuse policy before another adaptive certification can consume it.
|
|
226
|
+
Then compare full-trace, score-only, and thresholded development feedback on a null benchmark and an untouched replica at equal total spend.
|
|
227
|
+
That experiment tests whether formal information restrictions are worth their operational cost.
|
|
228
|
+
Defer a differential-privacy or reusable-holdout implementation until this measurement supports it.
|
|
229
|
+
|
|
230
|
+
### 3. Treat evaluation authoring as a measured task
|
|
231
|
+
|
|
232
|
+
**Recommendation: adapt the existing composition; largest automation opportunity.**
|
|
233
|
+
An evaluation candidate should be a versioned bundle of fixture references, checker identities, rubric, population description, and sealed decision rules.
|
|
234
|
+
The bundle should reuse `Scenario`, `JudgeConfig`, `VerificationStrategy`, `SealedExperiment`, and `EvidenceReceipt`.
|
|
235
|
+
Introduce only the missing admission and lineage fields.
|
|
236
|
+
|
|
237
|
+
The evaluator’s objective is detecting consequential defects while accepting independently verified good behavior.
|
|
238
|
+
High agent scores, high judge agreement, and large test counts are insufficient objectives.
|
|
239
|
+
A checker that rejects every output has excellent defect recall and no useful decision quality.
|
|
240
|
+
|
|
241
|
+
**Smallest decisive check:** give an eval author one real failure trace and a bounded budget.
|
|
242
|
+
Require a reproducible fixture, a correct reference, realistic negative controls, and an independently held audit set.
|
|
243
|
+
Compare its selected checker with a maintained human checker and a simple deterministic baseline.
|
|
244
|
+
Measure false acceptance and rejection, coverage by defect family, unknowns, flakiness, cost, and decision changes.
|
|
245
|
+
The winning author must improve audited decisions without weakening the accepted behavior.
|
|
246
|
+
|
|
247
|
+
### 4. Measure ranking validity and aggregation sensitivity
|
|
248
|
+
|
|
249
|
+
**Recommendation: adapt the calibration and reporting modules.**
|
|
250
|
+
Keep the evidence vector and per-dimension gates.
|
|
251
|
+
Add reports for per-candidate judge residuals against independent labels and uncertainty in pairwise ranking differences.
|
|
252
|
+
Good global agreement can hide a small directional bias that reverses a close comparison.
|
|
253
|
+
|
|
254
|
+
Retain subgroup counts and report the result under a few domain-approved population weights and normalization choices.
|
|
255
|
+
Record rank reversals and subgroup regressions rather than automatically choosing favorable weights.
|
|
256
|
+
Any data-driven choice of aggregation belongs to development and needs a later independent assessment.
|
|
257
|
+
|
|
258
|
+
**Smallest decisive check:** construct a panel with high aggregate agreement and known candidate-specific bias.
|
|
259
|
+
The ranking audit should catch the reversal while the existing agreement summary remains high.
|
|
260
|
+
For aggregation, change irrelevant alternatives and approved weights while keeping the focal models’ raw scores unchanged.
|
|
261
|
+
Report sensitivity without declaring that every aggregate is invalid.
|
|
262
|
+
|
|
263
|
+
[Prediction-powered inference](https://mlbenchmarks.org/14-evaluation-frontier.html#prediction-powered-inference) is a promising later experiment.
|
|
264
|
+
It combines many inexpensive predictions with a smaller independent labeled sample to correct measurement bias.
|
|
265
|
+
Compare its interval coverage and cost against an equally funded human-only estimator under candidate-specific and shifting bias.
|
|
266
|
+
The reference/proxy pairs and proxy-only sample must represent the same target population.
|
|
267
|
+
A bias correction does not resolve an undefined target construct.
|
|
268
|
+
The chapter derives a factor-two effective-sample-size ceiling for its considered unbiased estimators under a specified binary-score regime.
|
|
269
|
+
That regime requires agreement between 0.5 and the candidate’s reference score.
|
|
270
|
+
It does not bound every form of judge assistance or tool-based verification.
|
|
271
|
+
Measure coverage and total cost instead of assuming inexpensive proxy labels produce large savings.
|
|
272
|
+
|
|
273
|
+
### 5. Distinguish reproducibility from replication and transfer
|
|
274
|
+
|
|
275
|
+
**Recommendation: adapt existing receipts and experiment comparisons.**
|
|
276
|
+
Record whether a result reuses exact data, samples new tasks, changes the implementation, or tests another environment or population.
|
|
277
|
+
Use those distinctions in evidence records and reports.
|
|
278
|
+
Replicas should state which conclusion must reproduce: absolute performance, pairwise lift, ranking, or a proposed mechanism.
|
|
279
|
+
|
|
280
|
+
For claims about adaptation potential, compose repaired adaptation curves with method comparison and complete preparation costs.
|
|
281
|
+
Pin what each arm may change and which demonstrations it sees.
|
|
282
|
+
For product selection, preserve the actual deployed profile as the treatment being compared.
|
|
283
|
+
|
|
284
|
+
**Smallest decisive check:** rerun a fixed comparison on a newly sampled task cohort with preserved inclusion rules.
|
|
285
|
+
Check absolute score movement and paired/ranking movement separately.
|
|
286
|
+
Do not call a cache replay an independent replication.
|
|
287
|
+
Measure selection regret: the target loss incurred by choosing a candidate from the source benchmark.
|
|
288
|
+
A mostly preserved ranking can still select the wrong winner or leave every candidate below an operational reliability threshold.
|
|
289
|
+
|
|
290
|
+
### 6. Make deployment feedback interpretable
|
|
291
|
+
|
|
292
|
+
**Recommendation: adapt; execution remains downstream.**
|
|
293
|
+
After fixing outcome selection, extend outcome provenance with assignment, eligibility, exposure, observation window, and censoring information where the host can supply it.
|
|
294
|
+
Distinguish missing outcomes from users who experienced no event.
|
|
295
|
+
Record when the agent changes which tasks arrive, which users remain, or which feedback is observed.
|
|
296
|
+
|
|
297
|
+
Use outcome correlation as a diagnostic hypothesis.
|
|
298
|
+
Use randomized deployment comparisons when feasible, or existing off-policy estimators when their assumptions and propensities are defensible.
|
|
299
|
+
Neither correlation nor a stable feedback loop establishes that a change caused an improvement.
|
|
300
|
+
|
|
301
|
+
**Smallest decisive check:** build a two-cohort example where aggregate satisfaction rises because difficult users disappear while both cohorts worsen.
|
|
302
|
+
The report should expose changed denominators and within-cohort effects.
|
|
303
|
+
A production experiment should then measure whether the suspected selection mechanism actually occurs.
|
|
304
|
+
|
|
305
|
+
## A design for automated evaluation engineering
|
|
306
|
+
|
|
307
|
+
Here, “Software 3.0” means agents authoring executable evaluation assets and learning how to improve them.
|
|
308
|
+
This is a proposed system design derived from the review, not terminology or an architecture prescribed by the book.
|
|
309
|
+
|
|
310
|
+
Use three connected loops with separately versioned objectives and evidence.
|
|
311
|
+
Keep the evaluator fixed during each agent comparison.
|
|
312
|
+
Keep the independent evaluator audit fixed during each evaluation-design comparison.
|
|
313
|
+
Use deployment observations to challenge whether either comparison still represents the intended outcome.
|
|
314
|
+
|
|
315
|
+
```mermaid
|
|
316
|
+
flowchart TD
|
|
317
|
+
O[Production traces and independent outcomes] --> D[Diagnose missing behaviors]
|
|
318
|
+
D --> A[Host authors evaluation candidate]
|
|
319
|
+
A --> C[Calibrate on good and bad controls]
|
|
320
|
+
C --> H[Development challenge]
|
|
321
|
+
H --> S[Seal evaluation candidate and audit rules]
|
|
322
|
+
S --> U[Independent final audit]
|
|
323
|
+
U --> V[Admit evaluation version]
|
|
324
|
+
V --> P[Optimize agent against development cases]
|
|
325
|
+
P --> F[Measure selected agent on reserved final cases]
|
|
326
|
+
F --> R[Evidence receipt and release decision]
|
|
327
|
+
R --> O
|
|
328
|
+
H --> E[Reject or revise evaluation candidate]
|
|
329
|
+
E --> A
|
|
330
|
+
U --> X[Do not admit]
|
|
331
|
+
X -->|Fresh audit evidence or valid registered reuse| A
|
|
332
|
+
```
|
|
333
|
+
|
|
334
|
+
The host chooses methods and coordinates workers.
|
|
335
|
+
The diagram does not move agent execution into this package.
|
|
336
|
+
Final-audit feedback consumes its independence, whether the candidate passes or fails.
|
|
337
|
+
Another revision requires fresh audit evidence or a registered protocol that justifies the proposed reuse.
|
|
338
|
+
|
|
339
|
+
| Stage | Required artifact | Existing building block | New obligation |
|
|
340
|
+
| --- | --- | --- | --- |
|
|
341
|
+
| Diagnose | Cited failure hypothesis and affected population | Trace analysts, feedback trajectories, replay | Explain why the case matters beyond being easy to generate. |
|
|
342
|
+
| Author | Fixture/checker/rubric bundle with source identities | Eval fixtures, datasets, checker ports | Separate author-visible material from sealed labels and final cases. |
|
|
343
|
+
| Calibrate | Known-good and known-bad outcomes, with execution evidence | Plants, golden calibration, verifiers | Reject inert checks, always-reject checks, and checks that reward the author’s own wording. |
|
|
344
|
+
| Challenge | Independently constructed development counterexamples and adjudicated disagreements | Blind equivalence protocol, repair execution, judge calibration | Measure defect-family coverage, false decisions, and source independence. |
|
|
345
|
+
| Seal and audit | Frozen candidate, registered rules, and independent final audit | Sealed experiments, power checks, funnels | Bind population, observation unit, exposure policy, and audit authority; consume final-audit evidence on disclosure. |
|
|
346
|
+
| Improve agent | Complete candidate history and detached selected artifact | `selfImprove`, optimizer adapters, SearchLedger | Prevent evaluator or final-case changes inside the comparison. |
|
|
347
|
+
| Certify | Paired final results and exact evidence bindings | Campaigns, gates, EvidenceReceipt | Verify final-evidence reservation and required evaluator health. |
|
|
348
|
+
| Revalidate | Later outcomes and an explicit replication decision | Outcome stores, sentinel, evidence registry | Distinguish drift, changed populations, and candidate-caused feedback. |
|
|
349
|
+
|
|
350
|
+
### How to evaluate the eval author
|
|
351
|
+
|
|
352
|
+
The unit of observation should be an independent task or incident family with independently adjudicated outcomes.
|
|
353
|
+
Split entire families and generator sources across authoring, selection, and final audit.
|
|
354
|
+
Do not let the author choose which of its failures disappear from the denominator.
|
|
355
|
+
|
|
356
|
+
Give every author the same source evidence, tool access, labeling allowance, and total measured budget.
|
|
357
|
+
Account for generation, calibration, challenge, failed executions, review, and final scoring.
|
|
358
|
+
Equal numbers of generated tests are not equal resources.
|
|
359
|
+
|
|
360
|
+
Compare three approaches initially: a maintained human procedure, a simple fixture-and-mutation procedure, and one autonomous author.
|
|
361
|
+
Use existing method comparison and sealed experiment machinery wherever their contracts fit.
|
|
362
|
+
Prefer one bounded pilot domain with executable ground truth before semantic or frontier research tasks.
|
|
363
|
+
For optimizing the target agent, include simple sample-and-verify selection as a baseline when outcomes are executable.
|
|
364
|
+
Charge its sampling and verification against the same total resource allowance.
|
|
365
|
+
|
|
366
|
+
The primary decision should concern false acceptance of consequential defects under a predeclared acceptable false-rejection limit.
|
|
367
|
+
Report both error rates with denominators and uncertainty, along with coverage, inconclusive outcomes, run failures, latency, and complete cost.
|
|
368
|
+
Report the distribution across task families; an average can hide a completely untested failure class.
|
|
369
|
+
Use an independently chosen practical effect threshold and cluster-aware power calculation to set the final sample size.
|
|
370
|
+
Calibrate the complete workflow on known nulls and known improvements across repeated independent experiments.
|
|
371
|
+
Include case authoring, candidate selection, stopping, and final confirmation in that measurement.
|
|
372
|
+
An isolated estimator’s error rate does not certify the larger adaptive procedure.
|
|
373
|
+
|
|
374
|
+
Test transfer to new repositories, task families, and later failure incidents before claiming a general eval engineer.
|
|
375
|
+
Measure whether the admitted evaluator changes actual release decisions correctly.
|
|
376
|
+
Generating plausible tests is only an intermediate artifact.
|
|
377
|
+
|
|
378
|
+
### Stop the recursive trust problem at explicit authority
|
|
379
|
+
|
|
380
|
+
The same agent may propose changes to the system and to its evaluator in separate experiments.
|
|
381
|
+
It must not certify its own change by altering the standard during the comparison.
|
|
382
|
+
When a rubric or checker changes, version it and remeasure both arms under that version.
|
|
383
|
+
A prompt edit to the evaluator cannot retroactively improve the candidate’s recorded result.
|
|
384
|
+
Before combining both changes, score old and new agent outputs under both old and new evaluators.
|
|
385
|
+
Independent references in this crossed comparison distinguish an agent improvement from a changed measuring instrument.
|
|
386
|
+
|
|
387
|
+
The final audit should use a separately controlled evidence source: executable ground truth, independent labels, replication, or an appropriate checker.
|
|
388
|
+
Different model families can reduce one source of shared bias; they do not prove independence or correctness.
|
|
389
|
+
The host must enforce separation and retain actual access evidence.
|
|
390
|
+
An authority field records the declaration; it does not authenticate the declaration by itself.
|
|
391
|
+
|
|
392
|
+
For tasks without answer keys, retain the checker’s stated assumptions and unresolved obligations.
|
|
393
|
+
Request a better observation or narrower claim when no available check can discriminate success from failure.
|
|
394
|
+
Do not resolve that limit by adding more mutually agreeing agents.
|
|
395
|
+
|
|
396
|
+
### Allocate work by the uncertainty that changes the decision
|
|
397
|
+
|
|
398
|
+
A downstream controller can choose between collecting new tasks, adding repetitions, buying independent labels, challenging a checker, or testing another mechanism.
|
|
399
|
+
Eval should provide the evidence and cost estimates for that choice.
|
|
400
|
+
It should not hard-code a research strategy into the substrate.
|
|
401
|
+
Compose existing curriculum, discriminative selection, and adversarial exploration where they fit.
|
|
402
|
+
The proposed addition concerns valid allocation and comparison of those choices, rather than their initial implementation.
|
|
403
|
+
|
|
404
|
+
If uncertainty comes from task diversity, collect new task families.
|
|
405
|
+
If it comes from execution noise on a fixed roster, add repetitions.
|
|
406
|
+
If the judge is biased, buy independent labels or execute a stronger check.
|
|
407
|
+
If every candidate ties, inspect case discrimination and metric resolution before proposing another architecture.
|
|
408
|
+
Retain a representative final sample when development uses deliberately difficult or discriminative cases.
|
|
409
|
+
|
|
410
|
+
This controller is worth implementing only after a pilot compares it with a fixed allocation at equal actual cost.
|
|
411
|
+
Measure correct decisions per budget, not the amount of activity it schedules.
|
|
412
|
+
|
|
413
|
+
## Build order and decisions to defer
|
|
414
|
+
|
|
415
|
+
| Order | Deliverable | Evidence required before continuing |
|
|
416
|
+
| --- | --- | --- |
|
|
417
|
+
| First | Repair outcome metric selection and adaptation pairing; correct sequential assumption guidance and ambiguous metric labels. | Focused regressions using the reproduced counterexamples and valid controls. |
|
|
418
|
+
| Next | One claim contract and final-exposure integration through existing seals, ledgers, and gates. | Independent-unit, restart, duplicate-use, concurrency, and information-access boundary checks. |
|
|
419
|
+
| Pilot | One runtime-owned eval author that emits ordinary fixtures and registered evaluations. | Independent final audit against human and simple baselines at equal measured resources. |
|
|
420
|
+
| Expand | Ranking-bias audits, aggregation sensitivity, and fresh-cohort replication. | Show that each addition changes a previously wrong or unresolved decision. |
|
|
421
|
+
| Conditional | Prediction-powered inference and adaptive allocation of labels or compute. | Demonstrated coverage, reduced decision error, and measured cost benefit under relevant bias and drift. |
|
|
422
|
+
|
|
423
|
+
Retain official optimizers and the single campaign execution path.
|
|
424
|
+
Reject duplicating the ledger, experiment language, receipt format, or generic researcher loop merely to package this proposal.
|
|
425
|
+
Defer a universal scalar capability score, an automatically learned definition of user value, and unrestricted evaluator self-modification.
|
|
426
|
+
The book supplies reasons to distrust those shortcuts, not evidence that a larger autonomous loop will overcome them.
|
|
427
|
+
|
|
428
|
+
## Verification and limits
|
|
429
|
+
|
|
430
|
+
Local typechecking, build, and package verification passed at the inspected revision.
|
|
431
|
+
The diagnostic passed a separate strict TypeScript check, and all seven outputs reproduced exactly on a second execution.
|
|
432
|
+
An isolated archive of the reviewed source reproduced the complete diagnostic output, including source identity.
|
|
433
|
+
Eleven focused provenance checks covered that reproduction, local changes, documentation stability, and refusal of symlinks and special files.
|
|
434
|
+
The Vitest run completed with **399 passed files, 2 skipped files; 5,876 passed tests, 3 skipped tests**.
|
|
435
|
+
The recorded test invocation expanded to the full suite; the exact command is preserved with the observations.
|
|
436
|
+
No production evaluation campaign, paid optimization experiment, or deployment was performed.
|
|
437
|
+
|
|
438
|
+
Source inspection and the offline probes support the gaps and defects identified here.
|
|
439
|
+
They do not establish downstream prevalence, adoption cost, or expected performance gains.
|
|
440
|
+
The proposed experiments state what evidence would justify implementation or reject the proposal.
|