@tangle-network/agent-eval 0.144.5 → 0.144.7
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +30 -0
- package/README.md +2 -0
- package/dist/{benchmark-Fmo42QVE.d.ts → agent-profile-CgDTo40f.d.ts} +91 -4
- package/dist/agent-profile-CgDTo40f.d.ts.map +1 -0
- package/dist/{agent-profile-cell-Cw0PVwDr.d.ts → agent-profile-cell-BOP-iA9Q.d.ts} +2 -2
- package/dist/{agent-profile-cell-Cw0PVwDr.d.ts.map → agent-profile-cell-BOP-iA9Q.d.ts.map} +1 -1
- package/dist/analyst/index.d.ts +471 -89
- package/dist/analyst/index.d.ts.map +1 -1
- package/dist/analyst/index.js +24 -5
- package/dist/analyst/index.js.map +1 -1
- package/dist/{analyze-runs-DWIvOAGk.js → analyze-runs-BNNK7irB.js} +3 -3
- package/dist/{analyze-runs-DWIvOAGk.js.map → analyze-runs-BNNK7irB.js.map} +1 -1
- package/dist/{baseline-D_fT6277.d.ts → baseline-CavEbRyH.d.ts} +2 -23
- package/dist/baseline-CavEbRyH.d.ts.map +1 -0
- package/dist/{benchmark-command-CA_NFOmy.js → benchmark-command-BCafwNrf.js} +664 -605
- package/dist/benchmark-command-BCafwNrf.js.map +1 -0
- package/dist/benchmarks/index.d.ts +1 -1
- package/dist/benchmarks/index.js +1 -1
- package/dist/{benchmarks-CWbsj0t4.js → benchmarks-CDSolHq7.js} +4 -4
- package/dist/{benchmarks-CWbsj0t4.js.map → benchmarks-CDSolHq7.js.map} +1 -1
- package/dist/campaign/index.d.ts +8 -6
- package/dist/campaign/index.js +5 -3
- package/dist/{campaign-ClpnD7Ug.js → campaign-Tdy3h62h.js} +17 -301
- package/dist/campaign-Tdy3h62h.js.map +1 -0
- package/dist/cli.js +2 -2
- package/dist/{client-DAb7MWtL.d.ts → client-DjXROWpx.d.ts} +4 -4
- package/dist/{client-DAb7MWtL.d.ts.map → client-DjXROWpx.d.ts.map} +1 -1
- package/dist/{completion-verifier-VvpHRu78.d.ts → completion-verifier-foUCLif_.d.ts} +6 -6
- package/dist/{completion-verifier-VvpHRu78.d.ts.map → completion-verifier-foUCLif_.d.ts.map} +1 -1
- package/dist/contract/index.d.ts +11 -10
- package/dist/contract/index.d.ts.map +1 -1
- package/dist/contract/index.js +7 -6
- package/dist/contract/index.js.map +1 -1
- package/dist/control.d.ts +2 -2
- package/dist/control.js +1 -1
- package/dist/{cost-ledger-FuQvHxPm.d.ts → cost-ledger-Bv_e8XHY.d.ts} +2 -2
- package/dist/{cost-ledger-FuQvHxPm.d.ts.map → cost-ledger-Bv_e8XHY.d.ts.map} +1 -1
- package/dist/counterfactual-CWPTrMH7.js +126 -0
- package/dist/counterfactual-CWPTrMH7.js.map +1 -0
- package/dist/counterfactual-CxmxAONP.d.ts +72 -0
- package/dist/counterfactual-CxmxAONP.d.ts.map +1 -0
- package/dist/{dataset-v_Y5902-.d.ts → dataset-C8xaLXdY.d.ts} +2 -2
- package/dist/{dataset-v_Y5902-.d.ts.map → dataset-C8xaLXdY.d.ts.map} +1 -1
- package/dist/{default-registry-BwbZ9N9v.d.ts → default-registry-BZhStdjl.d.ts} +6 -5
- package/dist/default-registry-BZhStdjl.d.ts.map +1 -0
- package/dist/{default-registry-RLNNoeEP.js → default-registry-BaQXW1Ow.js} +2 -2
- package/dist/{default-registry-RLNNoeEP.js.map → default-registry-BaQXW1Ow.js.map} +1 -1
- package/dist/{dspy-rlm-engine-19FQEMBK.js → dspy-rlm-engine-BiN49gK6.js} +4 -3
- package/dist/{dspy-rlm-engine-19FQEMBK.js.map → dspy-rlm-engine-BiN49gK6.js.map} +1 -1
- package/dist/{tool-groups-ByZiqpVk.d.ts → engine-nB64f48I.d.ts} +21 -34
- package/dist/engine-nB64f48I.d.ts.map +1 -0
- package/dist/{errors-DkfjIDvD.d.ts → errors-CKPfb2aH.d.ts} +2 -2
- package/dist/{errors-DkfjIDvD.d.ts.map → errors-CKPfb2aH.d.ts.map} +1 -1
- package/dist/errors-D-LKuDhb.js.map +1 -1
- package/dist/{eval-campaign-lZcDIwQM.js → eval-campaign-DNjCvAm-.js} +7 -6
- package/dist/{eval-campaign-lZcDIwQM.js.map → eval-campaign-DNjCvAm-.js.map} +1 -1
- package/dist/{exact-types-BQ7W90C4.d.ts → exact-types-Djvzosly.d.ts} +2 -2
- package/dist/{exact-types-BQ7W90C4.d.ts.map → exact-types-Djvzosly.d.ts.map} +1 -1
- package/dist/exec-BLtYZdWo.js +49 -0
- package/dist/exec-BLtYZdWo.js.map +1 -0
- package/dist/experiment/index.d.ts +802 -0
- package/dist/experiment/index.d.ts.map +1 -0
- package/dist/experiment/index.js +1108 -0
- package/dist/experiment/index.js.map +1 -0
- package/dist/experiment-tracker-CnRICnMl.js +500 -0
- package/dist/experiment-tracker-CnRICnMl.js.map +1 -0
- package/dist/experiment-tracker-IMntXr6J.d.ts +299 -0
- package/dist/experiment-tracker-IMntXr6J.d.ts.map +1 -0
- package/dist/{external-optimizer-contracts-CdmX2K2S.d.ts → external-optimizer-contracts-CZuJNcT5.d.ts} +3 -3
- package/dist/{external-optimizer-contracts-CdmX2K2S.d.ts.map → external-optimizer-contracts-CZuJNcT5.d.ts.map} +1 -1
- package/dist/{feedback-trajectory-WK7x4mhy.d.ts → feedback-trajectory-Rh280oXo.d.ts} +4 -4
- package/dist/{feedback-trajectory-WK7x4mhy.d.ts.map → feedback-trajectory-Rh280oXo.d.ts.map} +1 -1
- package/dist/fuzz.d.ts +2 -2
- package/dist/hosted/index.d.ts +3 -3
- package/dist/{index-BZUe-ODI.d.ts → index-BZ3-y4YL.d.ts} +3 -3
- package/dist/{index-BZUe-ODI.d.ts.map → index-BZ3-y4YL.d.ts.map} +1 -1
- package/dist/{index-DSC51roc.d.ts → index-C3ssXVLv.d.ts} +2 -2
- package/dist/{index-DSC51roc.d.ts.map → index-C3ssXVLv.d.ts.map} +1 -1
- package/dist/{index-CsuAo2-J.d.ts → index-C5HOo4ZF2.d.ts} +5 -5
- package/dist/index-C5HOo4ZF2.d.ts.map +1 -0
- package/dist/{index-DEb46kc6.d.ts → index-CvSN3IG1.d.ts} +2 -2
- package/dist/{index-DEb46kc6.d.ts.map → index-CvSN3IG1.d.ts.map} +1 -1
- package/dist/{index-CNOCxBLh.d.ts → index-CvXXlyz7.d.ts} +3 -3
- package/dist/{index-CNOCxBLh.d.ts.map → index-CvXXlyz7.d.ts.map} +1 -1
- package/dist/{index-DGIzNtRv.d.ts → index-CwDrUMe0.d.ts} +3 -3
- package/dist/{index-DGIzNtRv.d.ts.map → index-CwDrUMe0.d.ts.map} +1 -1
- package/dist/{index-CpxZSlB7.d.ts → index-Sh2I0DRc.d.ts} +15 -649
- package/dist/index-Sh2I0DRc.d.ts.map +1 -0
- package/dist/index.d.ts +252 -451
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +271 -751
- package/dist/index.js.map +1 -1
- package/dist/{insight-report-D5m1z0_n.d.ts → insight-report-C6h6F_4L.d.ts} +4 -4
- package/dist/{insight-report-D5m1z0_n.d.ts.map → insight-report-C6h6F_4L.d.ts.map} +1 -1
- package/dist/{integrity-DRXobPEs.d.ts → integrity-BuqEKu-x.d.ts} +3 -3
- package/dist/{integrity-DRXobPEs.d.ts.map → integrity-BuqEKu-x.d.ts.map} +1 -1
- package/dist/integrity-MLzHOfV9.js +141 -0
- package/dist/integrity-MLzHOfV9.js.map +1 -0
- package/dist/kind-factory-BHIgPmzS.js.map +1 -1
- package/dist/ledger-core/index.d.ts +1 -1
- package/dist/{llm-client-D3EoChAU.js → llm-client-DzvMUsS_.js} +310 -10
- package/dist/llm-client-DzvMUsS_.js.map +1 -0
- package/dist/matrix/index.d.ts +2 -2
- package/dist/meta-eval/index.d.ts +2 -2
- package/dist/{mint-DD-0oQTA.js → mint-CGEkzPLf.js} +2 -2
- package/dist/{mint-DD-0oQTA.js.map → mint-CGEkzPLf.js.map} +1 -1
- package/dist/{multi-layer-verifier-BHY1gWAc.d.ts → multi-layer-verifier-DnAqwl0h.d.ts} +2 -2
- package/dist/{multi-layer-verifier-BHY1gWAc.d.ts.map → multi-layer-verifier-DnAqwl0h.d.ts.map} +1 -1
- package/dist/multishot/index.d.ts +3 -3
- package/dist/openapi.json +1 -1
- package/dist/paired-promotion-decision-B6zJ3gYM.d.ts +114 -0
- package/dist/paired-promotion-decision-B6zJ3gYM.d.ts.map +1 -0
- package/dist/pipelines/index.d.ts +2 -1
- package/dist/pipelines/index.d.ts.map +1 -1
- package/dist/pre-registration-DakwTRXk.js +96 -0
- package/dist/pre-registration-DakwTRXk.js.map +1 -0
- package/dist/prime-bridge-transport-6feEglLf.d.ts +49 -0
- package/dist/prime-bridge-transport-6feEglLf.d.ts.map +1 -0
- package/dist/prime-protocol-BfSalTfR.js +453 -0
- package/dist/prime-protocol-BfSalTfR.js.map +1 -0
- package/dist/profile-cell.d.ts +1 -1
- package/dist/profile-cell.js +242 -1
- package/dist/profile-cell.js.map +1 -0
- package/dist/promotion-policy-ChWhTDBH.d.ts +289 -0
- package/dist/promotion-policy-ChWhTDBH.d.ts.map +1 -0
- package/dist/promotion-policy-CrLrmys8.js +682 -0
- package/dist/promotion-policy-CrLrmys8.js.map +1 -0
- package/dist/{propose-review-control-BciUCZoh.js → propose-review-control-qgWLA6E9.js} +2 -2
- package/dist/{propose-review-control-BciUCZoh.js.map → propose-review-control-qgWLA6E9.js.map} +1 -1
- package/dist/{release-report-BZRWdq_t.d.ts → release-report-CI8uisI1.d.ts} +4 -4
- package/dist/{release-report-BZRWdq_t.d.ts.map → release-report-CI8uisI1.d.ts.map} +1 -1
- package/dist/{release-report-Sl0xfkFv.js → release-report-Dy39cbFF.js} +3 -3
- package/dist/{release-report-Sl0xfkFv.js.map → release-report-Dy39cbFF.js.map} +1 -1
- package/dist/{replay-B7S7Pdbw.d.ts → replay-DFf-teiC.d.ts} +8 -7
- package/dist/replay-DFf-teiC.d.ts.map +1 -0
- package/dist/{replay-GW61ezMW.js → replay-DyBLaKFc.js} +3 -3
- package/dist/{replay-GW61ezMW.js.map → replay-DyBLaKFc.js.map} +1 -1
- package/dist/reporting.d.ts +4 -4
- package/dist/reporting.js +2 -2
- package/dist/{researcher-BSiCoM1s.d.ts → researcher-BoaxeCzP.d.ts} +6 -6
- package/dist/{researcher-BSiCoM1s.d.ts.map → researcher-BoaxeCzP.d.ts.map} +1 -1
- package/dist/{reward-hacking-CyuzxKly.js → reward-hacking-B2tPL9a3.js} +12 -3
- package/dist/reward-hacking-B2tPL9a3.js.map +1 -0
- package/dist/{reward-hacking-DFgkEY4p.d.ts → reward-hacking-Cf1PtEOz.d.ts} +34 -4
- package/dist/reward-hacking-Cf1PtEOz.d.ts.map +1 -0
- package/dist/rl.d.ts +19 -9
- package/dist/rl.d.ts.map +1 -1
- package/dist/rl.js +16 -7
- package/dist/rl.js.map +1 -1
- package/dist/rollout/index.d.ts +2 -2
- package/dist/rollout/index.js +2 -2
- package/dist/{rollout-C2fD1cf4.js → rollout-2ECTXb2N.js} +2 -2
- package/dist/{rollout-C2fD1cf4.js.map → rollout-2ECTXb2N.js.map} +1 -1
- package/dist/{rubric-predictive-validity-C4r4Y-q8.d.ts → rubric-predictive-validity-9qAwzkZm.d.ts} +2 -2
- package/dist/{rubric-predictive-validity-C4r4Y-q8.d.ts.map → rubric-predictive-validity-9qAwzkZm.d.ts.map} +1 -1
- package/dist/{run-evidence-j5Ynww6L.d.ts → run-evidence-BDFFai9R.d.ts} +3 -3
- package/dist/{run-evidence-j5Ynww6L.d.ts.map → run-evidence-BDFFai9R.d.ts.map} +1 -1
- package/dist/{run-record-ooo9FWns.d.ts → run-record-DdSa93_W.d.ts} +4 -4
- package/dist/{run-record-ooo9FWns.d.ts.map → run-record-DdSa93_W.d.ts.map} +1 -1
- package/dist/{run-record-CWN8-VsV.js → run-record-DqOw5X6_.js} +2 -2
- package/dist/{run-record-CWN8-VsV.js.map → run-record-DqOw5X6_.js.map} +1 -1
- package/dist/{schema-Cef2cFmb.d.ts → schema-Cef2cFmb2.d.ts} +1 -1
- package/dist/schema-Cef2cFmb2.d.ts.map +1 -0
- package/dist/{semantic-concept-judge-DKRtp2sY.js → semantic-concept-judge-Bmrq6yqU.js} +2 -2
- package/dist/{semantic-concept-judge-DKRtp2sY.js.map → semantic-concept-judge-Bmrq6yqU.js.map} +1 -1
- package/dist/sequential-D-BLJBKU.js +299 -0
- package/dist/sequential-D-BLJBKU.js.map +1 -0
- package/dist/{server-Df00sdwz.js → server-iu0ede49.js} +2 -2
- package/dist/{server-Df00sdwz.js.map → server-iu0ede49.js.map} +1 -1
- package/dist/{single-run-lock-D5iN0Xzb.js → single-run-lock-BMQEv1wG.js} +7 -161
- package/dist/single-run-lock-BMQEv1wG.js.map +1 -0
- package/dist/{skill-usage-DUvvudWR.d.ts → skill-usage-CJlWEUFt.d.ts} +11 -11
- package/dist/{skill-usage-DUvvudWR.d.ts.map → skill-usage-CJlWEUFt.d.ts.map} +1 -1
- package/dist/{skillopt-optimization-method-CGz9ywhM.d.ts → skillopt-optimization-method-BO7NAl3b.d.ts} +11 -297
- package/dist/skillopt-optimization-method-BO7NAl3b.d.ts.map +1 -0
- package/dist/{skillopt-optimization-method-C4FX42dy.js → skillopt-optimization-method-CQwZ-ZX8.js} +8 -652
- package/dist/skillopt-optimization-method-CQwZ-ZX8.js.map +1 -0
- package/dist/statistical-heldout-Dn9ruizm.d.ts +555 -0
- package/dist/statistical-heldout-Dn9ruizm.d.ts.map +1 -0
- package/dist/{statistics-B4u_CiFd.d.ts → statistics-D6Uebe_4.d.ts} +2 -2
- package/dist/{statistics-B4u_CiFd.d.ts.map → statistics-D6Uebe_4.d.ts.map} +1 -1
- package/dist/steps-BArUxhna.d.ts +51 -0
- package/dist/steps-BArUxhna.d.ts.map +1 -0
- package/dist/{integrity-fdt8XPAv.js → store-DNe_Uv1Q.js} +2 -140
- package/dist/store-DNe_Uv1Q.js.map +1 -0
- package/dist/{summary-report-BOM6dfP7.d.ts → summary-report-DuUS_i7W.d.ts} +4 -115
- package/dist/summary-report-DuUS_i7W.d.ts.map +1 -0
- package/dist/{summary-report-9A5y7EsK.js → summary-report-Lf-5I7xh.js} +2 -2
- package/dist/{summary-report-9A5y7EsK.js.map → summary-report-Lf-5I7xh.js.map} +1 -1
- package/dist/supervisor-run/index.d.ts +2 -2
- package/dist/supervisor-run/index.js +1 -1
- package/dist/{supervisor-run-DiyQVczd.js → supervisor-run-D_sokXcO.js} +32 -8
- package/dist/{supervisor-run-DiyQVczd.js.map → supervisor-run-D_sokXcO.js.map} +1 -1
- package/dist/tool-groups-DIVBnJyl.d.ts +28 -0
- package/dist/tool-groups-DIVBnJyl.d.ts.map +1 -0
- package/dist/trace-repair/index.d.ts +2102 -0
- package/dist/trace-repair/index.d.ts.map +1 -0
- package/dist/trace-repair/index.js +3878 -0
- package/dist/trace-repair/index.js.map +1 -0
- package/dist/traces.d.ts +6 -6
- package/dist/traces.js +3 -2
- package/dist/trajectory-YC15QDYQ.d.ts +24 -0
- package/dist/trajectory-YC15QDYQ.d.ts.map +1 -0
- package/dist/trajectory-replay/index.d.ts +781 -0
- package/dist/trajectory-replay/index.d.ts.map +1 -0
- package/dist/trajectory-replay/index.js +2103 -0
- package/dist/trajectory-replay/index.js.map +1 -0
- package/dist/{types-BjMFz88h.d.ts → types-D216SgwM.d.ts} +228 -9
- package/dist/types-D216SgwM.d.ts.map +1 -0
- package/dist/{types-yLK8gXE9.d.ts → types-D4mog56g.d.ts} +2 -2
- package/dist/{types-yLK8gXE9.d.ts.map → types-D4mog56g.d.ts.map} +1 -1
- package/dist/{types-CZt1PBIk.d.ts → types-DF_Udrp-.d.ts} +54 -5
- package/dist/{types-CZt1PBIk.d.ts.map → types-DF_Udrp-.d.ts.map} +1 -1
- package/dist/{types-Dcoaqcsc.d.ts → types-DYuNHo9R.d.ts} +5 -5
- package/dist/{types-Dcoaqcsc.d.ts.map → types-DYuNHo9R.d.ts.map} +1 -1
- package/dist/usage-receipt-EVI8B8Xu.js.map +1 -1
- package/dist/verdict-DExhxfgR.d.ts +201 -0
- package/dist/verdict-DExhxfgR.d.ts.map +1 -0
- package/dist/verdict-cache-BCcOh0kF.js +159 -0
- package/dist/verdict-cache-BCcOh0kF.js.map +1 -0
- package/dist/wire/index.d.ts +3 -3
- package/dist/wire/index.js +1 -1
- package/docs/building-doctrine.md +15 -0
- package/docs/charter.md +112 -0
- package/docs/experiment.md +104 -0
- package/docs/prime-analyst.md +1 -0
- package/docs/trace-analysis.md +26 -0
- package/docs/trace-repair-admission.md +194 -0
- package/docs/trace-repair-analyst-arms.md +121 -0
- package/docs/trace-repair-continuation.md +107 -0
- package/docs/trace-repair-grader.md +163 -0
- package/docs/trajectory-replay.md +110 -0
- package/docs/verification-strategies.md +103 -0
- package/package.json +21 -4
- package/dist/agent-profile-cell-CbfBm2g6.js +0 -335
- package/dist/agent-profile-cell-CbfBm2g6.js.map +0 -1
- package/dist/baseline-D_fT6277.d.ts.map +0 -1
- package/dist/benchmark-Fmo42QVE.d.ts.map +0 -1
- package/dist/benchmark-command-CA_NFOmy.js.map +0 -1
- package/dist/campaign-ClpnD7Ug.js.map +0 -1
- package/dist/default-registry-BwbZ9N9v.d.ts.map +0 -1
- package/dist/index-CpxZSlB7.d.ts.map +0 -1
- package/dist/index-CsuAo2-J.d.ts.map +0 -1
- package/dist/integrity-fdt8XPAv.js.map +0 -1
- package/dist/llm-client-D3EoChAU.js.map +0 -1
- package/dist/replay-B7S7Pdbw.d.ts.map +0 -1
- package/dist/reward-hacking-CyuzxKly.js.map +0 -1
- package/dist/reward-hacking-DFgkEY4p.d.ts.map +0 -1
- package/dist/schema-Cef2cFmb.d.ts.map +0 -1
- package/dist/single-run-lock-D5iN0Xzb.js.map +0 -1
- package/dist/skillopt-optimization-method-C4FX42dy.js.map +0 -1
- package/dist/skillopt-optimization-method-CGz9ywhM.d.ts.map +0 -1
- package/dist/summary-report-BOM6dfP7.d.ts.map +0 -1
- package/dist/tool-groups-ByZiqpVk.d.ts.map +0 -1
- package/dist/types-BjMFz88h.d.ts.map +0 -1
- package/dist/verdict-Dps8_okt.d.ts +0 -37
- package/dist/verdict-Dps8_okt.d.ts.map +0 -1
|
@@ -0,0 +1,104 @@
|
|
|
1
|
+
# The experiment subpath
|
|
2
|
+
|
|
3
|
+
`@tangle-network/agent-eval/experiment` turns an experiment's registration into the object that runs it.
|
|
4
|
+
|
|
5
|
+
## The covenant
|
|
6
|
+
|
|
7
|
+
1. **The registered rule is the executed rule.**
|
|
8
|
+
Every rule — row admission, subset selection, estimand, interval, decision table, validity gate, halt, budget, matched budget, reissue — is a typed data node, never a closure or prose.
|
|
9
|
+
`sealExperiment` canonicalizes and hashes the whole tree into one digest.
|
|
10
|
+
Every interpreter takes only a sealed node plus evidence records.
|
|
11
|
+
No execution surface has a parameter for alpha, threshold, metric, or stopping rule, so registered-vs-ran drift is unrepresentable rather than checked.
|
|
12
|
+
2. **Refusals live inside artifacts.**
|
|
13
|
+
An inadequate cluster count, a mismatched arm budget, a non-monotone funnel stage, a non-total decision table — each produces a typed verdict object (or a typed error), never a warning sentence beside a number.
|
|
14
|
+
3. **A change is a re-seal.**
|
|
15
|
+
`amendExperiment` verifies the current seal, validates the new spec, and appends a `{at, reason, blind[], digest}` entry.
|
|
16
|
+
The digest history is the audit trail; changing what is decided without a new digest is impossible.
|
|
17
|
+
|
|
18
|
+
## The objects
|
|
19
|
+
|
|
20
|
+
| object | entry point | what it closes |
|
|
21
|
+
| --- | --- | --- |
|
|
22
|
+
| Registered-rule AST | `src/experiment/ast.ts` (14 node families) | prose rules; every registered condition compiles to data the seal covers |
|
|
23
|
+
| Define / seal / execute | `defineExperiment`, `sealExperiment`, `amendExperiment`, `openSealedExperiment` | hand-written PREREG.md files; the runner executes the sealed rule itself |
|
|
24
|
+
| Cluster-aware power | `clusteredPower`, `assertDesignAdequate` | "4 clusters cannot certify any effect size, including 1.0" — learned by running the experiment, now refused before a dollar is spent |
|
|
25
|
+
| Denominator chain | `buildFunnel`, `executeAdmissionRule`, `composeFunnels`, `renderFunnelTable` | hand-assembled `20 → 15 → 14 → admitted` chains, formatted differently each run |
|
|
26
|
+
| Matched budgets | `verifyMatchedBudgets`, `assertMatchedBudgets` | "realized tokens must agree within 5%" verified by hand |
|
|
27
|
+
|
|
28
|
+
### Sealing and execution
|
|
29
|
+
|
|
30
|
+
```ts
|
|
31
|
+
import {
|
|
32
|
+
openSealedExperiment,
|
|
33
|
+
sealExperiment,
|
|
34
|
+
} from '@tangle-network/agent-eval/experiment'
|
|
35
|
+
|
|
36
|
+
const sealed = await sealExperiment(spec) // canonicalize + sha256 over the whole tree
|
|
37
|
+
const registered = await openSealedExperiment(sealed) // verifies the digest first
|
|
38
|
+
|
|
39
|
+
const admission = registered.admit(rows) // funnel + survivors, from the sealed rule
|
|
40
|
+
const gate = registered.gate('power-floor', { kind: 'power-floor', curve })
|
|
41
|
+
const halt = registered.halt([gate]) // refuse-spend fires before any contrast
|
|
42
|
+
const outcome = registered.decide(quantities) // the sealed table; non-total tables throw
|
|
43
|
+
```
|
|
44
|
+
|
|
45
|
+
`openSealedExperiment` is the only execution surface.
|
|
46
|
+
A rule that is not in the sealed spec cannot run; a rule that is cannot run differently.
|
|
47
|
+
|
|
48
|
+
### Cluster-aware power refusal
|
|
49
|
+
|
|
50
|
+
Two floors, one simulation:
|
|
51
|
+
|
|
52
|
+
- **Closed form, zero spend.** With `C` independent clusters, the exact whole-cluster sign-flip test can never produce a two-sided p below `2^(1-C)`.
|
|
53
|
+
Four clusters give 0.125 and three give 0.25 — both above alpha 0.05, so those designs are refused at any effect size.
|
|
54
|
+
Six clusters is the smallest certifiable count at 0.05.
|
|
55
|
+
- **Seeded simulation.** Per-row paired contrasts are drawn under a registered effect model (base win/loss rates, optional noisy clusters), each trial takes a whole-cluster percentile bootstrap, and power is the fraction of trials whose interval excludes zero.
|
|
56
|
+
|
|
57
|
+
The refusal is a verdict inside the returned artifact (`result.refusal`), with `assertDesignAdequate` as the throwing form.
|
|
58
|
+
The `power-floor` validity gate consumes a power curve as evidence and fails when the curve tops out under the registered target.
|
|
59
|
+
|
|
60
|
+
### The funnel
|
|
61
|
+
|
|
62
|
+
`buildFunnel` refuses a stage that gains rows, named exclusions that do not sum, and partitions that overdraw their source stage.
|
|
63
|
+
`executeAdmissionRule` runs a sealed admission rule over rows and returns the funnel, the survivors, and the partition rows in one object — the chain and the rows can never disagree.
|
|
64
|
+
Partitions carry `pooling: 'never'`: a secondary set is reported beside the primary chain and cannot be pooled into it.
|
|
65
|
+
The object is its own JSON render; `renderFunnelTable` prints the text table with the reconciliation line (`input = surviving + excluded`).
|
|
66
|
+
|
|
67
|
+
### Matched budgets
|
|
68
|
+
|
|
69
|
+
`verifyMatchedBudgets` compares realized per-arm tokens under the registered tolerance and returns a verdict whose `refusal` field carries `onFail: 'refuse-contrast'` when arms diverge.
|
|
70
|
+
A contrast between arms that spent differently is not a contrast; the refusal is the artifact that says so.
|
|
71
|
+
|
|
72
|
+
## Acceptance: the three preregistrations
|
|
73
|
+
|
|
74
|
+
The module's acceptance suite (`tests/experiment/preregistration-acceptance.test.ts`) re-derives the week's three hand-written preregistrations as sealed specs and reproduces each recorded decision by executing the sealed rules against the recorded evidence:
|
|
75
|
+
|
|
76
|
+
- **killtest-20260810** — all four validity gates fail on the recorded evidence (the rep-4 oracle flip, the 2-row population drift, the zero-call control, the 0.692 power ceiling) and the halt rule refuses the spend, matching the recorded `$0.00, contrast never run`.
|
|
77
|
+
The obligation node routes a positive interval without the registered control to `blocked-pending-registered-control`, never to `thesis-survives`.
|
|
78
|
+
- **freelunch-20260810** — the admission funnel reproduces the recorded `48 > 43 > 35 > 35 > 32` chain with the 3-row secondary partition; the uniform-pass budget reproduces the recorded uniform n=2; the amendment-6 ledger under the same sealed rule refuses pass 2 — the registered-vs-ran divergence the seal makes unrepresentable; the report-only decision reproduces `3/64` and `2/32`.
|
|
79
|
+
- **tbench-20260808 milestone 2** — the round-robin selection reproduces the recorded 20-row subset in pick order; the m3 subset filters the SEALED m2 draw (16 rows); the decision table on the recorded interval reproduces `not-certified-at-this-n`.
|
|
80
|
+
|
|
81
|
+
## What is composed, not duplicated
|
|
82
|
+
|
|
83
|
+
The statistical machinery underneath is re-exported from its existing homes; this subpath adds registration and refusal, not estimator forks.
|
|
84
|
+
|
|
85
|
+
| family | home |
|
|
86
|
+
| --- | --- |
|
|
87
|
+
| `pairedBootstrap`, `mcnemar`/`mcnemarPower`/`mcnemarRequiredN`, `pairedRiskDifference*`, `holm`, `benjaminiHochberg`, `eProcess`, `wilson`, `mulberry32`, sample-size helpers | `src/statistics.ts` |
|
|
88
|
+
| `clusteredPairedBinary` (cluster bootstrap + sign flip) | `src/clustered-paired-binary.ts` |
|
|
89
|
+
| `pairedEvalueSequence` (anytime-valid) | `src/sequential.ts` |
|
|
90
|
+
| `powerPreflight` (variance-based MDE refusal) | `src/campaign/gates/power-preflight.ts` |
|
|
91
|
+
| `sequentialPairedGate`, `sequentialDecide` (manifest-bound) | `src/campaign/gates/sequential.ts` |
|
|
92
|
+
| `heldoutSignificance`, `pairHoldout` | `src/campaign/gates/statistical-heldout.ts` |
|
|
93
|
+
| `paretoSignificanceGate`, `buildEvidenceVector` | `src/campaign/gates/promotion-policy.ts` |
|
|
94
|
+
| `pairArms`, `comparePairedArms`, `pairRunRecords` | `src/paired-arms.ts` |
|
|
95
|
+
| `canonicalize`, `hashJson`, `signManifest`, `verifyManifest`, `HypothesisManifest` | `src/pre-registration.ts` |
|
|
96
|
+
| `ExperimentTracker` (run ledger with KEEP/ITERATE/NOISE/REGRESSION) | `src/experiment-tracker.ts` |
|
|
97
|
+
|
|
98
|
+
`HypothesisManifest` stays as the lightweight single-metric registration; `sealExperiment` is the full-design registration.
|
|
99
|
+
The trace-repair admission machinery (`buildDenominatorChain`, oracle determinism, control policy) keeps its repair vocabulary in `./trace-repair`; this module is the general form new experiments should register against.
|
|
100
|
+
|
|
101
|
+
## Where this sits
|
|
102
|
+
|
|
103
|
+
This is Wave 2 of the [charter](./charter.md): the experiment subpath, built after the kill test that re-derived the three preregistrations as decision-rule objects (verdict: extended — ten node families beyond the seed AST, no opaque node, no rule dropped).
|
|
104
|
+
Wave 3 wires these objects to the live-sandbox seam; the improvement receipt (Wave 4) serializes a sealed experiment's digest, gates, and refusal outcomes into one attested file.
|
package/docs/prime-analyst.md
CHANGED
|
@@ -5,6 +5,7 @@ It is the third scored arm of `agent-eval analyst-benchmark`, beside the recursi
|
|
|
5
5
|
It speaks the CodeTraceBench failure-block contract only; `--analyst prime` with `--dataset agentrx` is rejected.
|
|
6
6
|
|
|
7
7
|
Implementation: `src/analyst/benchmark-runner-prime.ts` (`createPrimeBenchmarkRunner`) binds the CodeTraceBench block grammar to the shared protocol in `src/analyst/prime-protocol.ts` and `src/analyst/prime-bridge-transport.ts`.
|
|
8
|
+
The arm is declared as `primeCodeTraceAnalystDefinition()` and the creator is a thin shell over it; see the Analyst Definitions section in [trace-analysis.md](./trace-analysis.md).
|
|
8
9
|
Wiring: `--analyst prime` in `src/analyst/benchmark-command.ts`.
|
|
9
10
|
|
|
10
11
|
## What the runner does
|
package/docs/trace-analysis.md
CHANGED
|
@@ -215,6 +215,32 @@ Unknown, transformed, or fabricated identifiers are rejected.
|
|
|
215
215
|
An empty findings array means no submitted claim passed the evidence rules.
|
|
216
216
|
It does not prove the run was correct.
|
|
217
217
|
|
|
218
|
+
## Analyst Definitions
|
|
219
|
+
|
|
220
|
+
`AnalystDefinition` (`src/analyst/definition.ts`, exported from `@tangle-network/agent-eval/analyst`) is the declarative unit behind an analyst arm.
|
|
221
|
+
One definition value declares everything the arm can say to a model: the question, the task text, the `ReplyContract` row grammar, the `EvidenceProjection` (`inline` | `chunked` | `repl-variable` | `agent-tools`), a profile fragment (pinned model and reasoning-effort hints), the budget, and the repair-turn count.
|
|
222
|
+
`bindAnalyst(definition, transports)` compiles a definition plus a transport binding (`prime-bridge` or `model-owner`) into a runnable `AnalystBenchmarkRunner`.
|
|
223
|
+
|
|
224
|
+
The three benchmark arms are expressed this way:
|
|
225
|
+
|
|
226
|
+
| Arm | Definition builder | Projection | Repair turns |
|
|
227
|
+
|---|---|---|---|
|
|
228
|
+
| `direct` | `publicDirectAnalystDefinition(dataset, args)` | `chunked` (descending per-attribute byte caps) | 0 |
|
|
229
|
+
| `dspy-rlm` | `publicRlmAnalystDefinition(dataset, args)` | `repl-variable` (store bound as an engine REPL variable) | 1 (engine-internal typed repair) |
|
|
230
|
+
| `prime` | `primeCodeTraceAnalystDefinition(args)` | `inline` (serialized JSON with a capped refetch) | 0 or 1 |
|
|
231
|
+
|
|
232
|
+
`createPublicBenchmarkDirectRunner`, `createPublicBenchmarkRlmRunner`, and `createPrimeBenchmarkRunner` are thin shells over those builders, so no consumer changes.
|
|
233
|
+
The parity suite (`src/analyst/definition-parity.test.ts`) runs each compiled definition and its entry point over the same fixture rows with a fake transport and asserts byte-identical request bodies plus equal protocol digests.
|
|
234
|
+
A mismatch fails CI: expression loss between the declarative layer and the executing arm is caught by construction.
|
|
235
|
+
Do not loosen those assertions; report the construct that cannot be expressed and extend the definition slots instead.
|
|
236
|
+
|
|
237
|
+
`analystDefinitionProtocolSha256(definition)` digests the definition's protocol content; for an inline definition it equals the digest the prime arm records (`primeAnalystProtocolSha256()`).
|
|
238
|
+
`analystDefinitionAsymmetries(definitions)` compares arms on equal terms: it refuses a set whose definitions declare unequal repair turns (a retry is a second sample) and renders the declared differences — projection mode, reasoning effort, budget — beside each arm.
|
|
239
|
+
A definition a strategy cannot compile fails loud with `AnalystExpressivenessError` naming the construct.
|
|
240
|
+
|
|
241
|
+
`AnalystContext.probe` (`ExecutionProbe`) is the optional live-execution port: a runtime that owns a sandbox or checkout fills it so an analyst can run a bounded command against the run's produced state and read a typed outcome.
|
|
242
|
+
This package defines only the port; an absent probe means the analyst works from recorded evidence.
|
|
243
|
+
|
|
218
244
|
## Measure Analyst Quality
|
|
219
245
|
|
|
220
246
|
Measure the analyst on labeled traces before using its findings for automated changes.
|
|
@@ -0,0 +1,194 @@
|
|
|
1
|
+
# TB-Repair admission
|
|
2
|
+
|
|
3
|
+
Admission decides which corpus rows enter a campaign.
|
|
4
|
+
It runs before any analyst reads a row, and it publishes the denominator it produced.
|
|
5
|
+
|
|
6
|
+
`Delta-repair = P(tests pass | intervention) - P(tests pass | no-fix control)` is an average over a set of rows.
|
|
7
|
+
An analyst that declines the rows it cannot solve shrinks that set and raises its own score.
|
|
8
|
+
So the set is fixed first, by checks no analyst can influence, and every row that leaves is counted with the reason it left.
|
|
9
|
+
|
|
10
|
+
Source: [`src/trace-repair/`](../src/trace-repair/).
|
|
11
|
+
The policy the controls run under is in [trace-repair-continuation.md](./trace-repair-continuation.md).
|
|
12
|
+
|
|
13
|
+
## The five conditions
|
|
14
|
+
|
|
15
|
+
A row is admitted only when all five hold.
|
|
16
|
+
|
|
17
|
+
| # | condition | what it removes |
|
|
18
|
+
| --- | --- | --- |
|
|
19
|
+
| 0 | the task's own suite returns one verdict on byte-identical state | tasks whose grader is not a function of the state, where every condition below is a draw |
|
|
20
|
+
| 1 | the recorded prefix replays with divergence at or below 10% | rows whose recording does not reproduce, so no claim about step `k` can be executed |
|
|
21
|
+
| 2 | the task's held-out tests fail on the recorded end state | rows that did not actually fail |
|
|
22
|
+
| 3 | the no-fix control fails every rollout | rows the continuation policy rescues with no intervention at all |
|
|
23
|
+
| 4 | the no-op control fails every rollout | rows an inert action plus continuation rescues, which is a flaky task or a lucky policy |
|
|
24
|
+
|
|
25
|
+
Condition 0 runs first and costs nothing at run time: it reads the task's certification, which was measured once by [`certify-task-oracle.sh`](../benchmarks/trace-repair/tools/README.md) and checked in as [`task-oracles.json`](../benchmarks/trace-repair/task-oracles.json).
|
|
26
|
+
Conditions 1 to 4 all read the same suite, so a suite that answers differently about identical bytes makes each of them a coin flip rather than a measurement.
|
|
27
|
+
A task with no certification is excluded as `task-oracle-uncertified`, which is a different fact from `task-oracle-nondeterministic`: one says the check has not run, the other says it ran and the task failed it.
|
|
28
|
+
|
|
29
|
+
Conditions 3 and 4 are the ones that keep `Delta-repair` honest.
|
|
30
|
+
Without them a row that any continuation would have passed counts as a repair, and the number measures the continuation policy rather than the analyst.
|
|
31
|
+
|
|
32
|
+
## The control has to be able to rescue
|
|
33
|
+
|
|
34
|
+
Conditions 3 and 4 ask whether a row is rescued by continuing from the recorded end state.
|
|
35
|
+
That question has an answer only under a control that can act.
|
|
36
|
+
|
|
37
|
+
A control rollout changes the graded state by executing commands, and it executes only what a model call asks for.
|
|
38
|
+
At a step budget of zero it executes nothing, so the container it grades holds the same bytes condition 2 already graded as failing.
|
|
39
|
+
A control pass under such a policy is not a rescue — it is the task's own grader answering differently about one state.
|
|
40
|
+
Condition 3 then cannot fire for the reason it exists, and every row walks through it.
|
|
41
|
+
|
|
42
|
+
So the control is a declared, hashed parameter rather than a default, and the criteria say which reading applies:
|
|
43
|
+
|
|
44
|
+
| `controlScreening` | requires | a control pass means |
|
|
45
|
+
| --- | --- | --- |
|
|
46
|
+
| `enforced` (default) | `stepBudget >= 1` | `no-fix-control-passed` — the row is repairable by continuing alone |
|
|
47
|
+
| `declared-inert` | `stepBudget == 0` | `control-passed-on-identical-state` — the task's grader disagreed with itself |
|
|
48
|
+
|
|
49
|
+
Either pairing the other way round raises `UncalibratedControlError` at the call, before a verdict is produced for any row.
|
|
50
|
+
A screening control that cannot act, and a control declared inert that can, are both configuration faults rather than properties of a row.
|
|
51
|
+
|
|
52
|
+
`defineControlPolicy` hashes the whole declaration — id, step budget, scaffold, model, command timeout — so a step budget cannot move under an unchanged label, and a policy that calls no model must record `model: null`.
|
|
53
|
+
|
|
54
|
+
Every admission decision carries an `AdmissionScreeningRecord`: the control policy and its digest, the screening mode, the task name, and the task's measured oracle flip rate.
|
|
55
|
+
It is on the rejected decisions too, so a reader of an artifact can tell which control screened a row without opening the runner's source.
|
|
56
|
+
`AdmissionRowVerdict` carries the same three fields for rows the executing pre-pass excluded before any control ran.
|
|
57
|
+
|
|
58
|
+
Divergence is `divergences / prefixExecuted`, and the threshold admits a row sitting exactly on it.
|
|
59
|
+
A replay that executed fewer steps than the recording holds is excluded as `prefix-replay-truncated` before that ratio is read, because a truncated run computes divergence over the steps it did reach and a short replay would look perfect.
|
|
60
|
+
|
|
61
|
+
## The population split
|
|
62
|
+
|
|
63
|
+
The corpus assay measured the final recorded return code of every admitted row.
|
|
64
|
+
|
|
65
|
+
| stratum | final return code | assay share | can one command repair it |
|
|
66
|
+
| --- | --- | --- | --- |
|
|
67
|
+
| `clean-exit` | `0` | 87.13% | yes — the agent stopped believing it was done and the tests disagree |
|
|
68
|
+
| `command-error` | `> 0` | 12.65% | yes |
|
|
69
|
+
| `signal-kill` | `< 0` | 0.22% | no — the command was killed at a timeout |
|
|
70
|
+
|
|
71
|
+
Rows are stratified before anything samples them, and every row carries its stratum, admitted or not.
|
|
72
|
+
`admitStrata` defaults to `clean-exit` and `command-error`; the excluded signal kills appear in the chain as `stratum-not-admitted` rather than disappearing.
|
|
73
|
+
|
|
74
|
+
`AdmissionReport.strata` holds admitted ids per stratum and there is no pooled list.
|
|
75
|
+
Sampling reads one stratum at a time; pooling an addressable population with an unaddressable one takes an explicit concatenation that a reader can see.
|
|
76
|
+
|
|
77
|
+
## The denominator chain
|
|
78
|
+
|
|
79
|
+
A benchmark whose denominator is not auditable is not a benchmark.
|
|
80
|
+
Every run emits the funnel as JSON and renders it as a table: the reason, the rows that reached that stage, the rows it removed, and the rows that survived.
|
|
81
|
+
|
|
82
|
+
| stage | exclusion reason | entering | excluded | remaining |
|
|
83
|
+
| --- | --- | --- | --- | --- |
|
|
84
|
+
| 1 | `no-recorded-commands` | 5 | 1 | 4 |
|
|
85
|
+
| 2 | `unparseable-final-returncode` | 4 | 0 | 4 |
|
|
86
|
+
| 3 | `stratum-not-admitted` | 4 | 1 | 3 |
|
|
87
|
+
| 4 | `task-oracle-uncertified` | 3 | 0 | 3 |
|
|
88
|
+
| 5 | `task-oracle-nondeterministic` | 3 | 0 | 3 |
|
|
89
|
+
| 6 | `prefix-replay-error` | 3 | 0 | 3 |
|
|
90
|
+
| 7 | `prefix-replay-empty` | 3 | 0 | 3 |
|
|
91
|
+
| 8 | `prefix-replay-truncated` | 3 | 0 | 3 |
|
|
92
|
+
| 9 | `prefix-divergence-above-threshold` | 3 | 1 | 2 |
|
|
93
|
+
| … | … | … | … | … |
|
|
94
|
+
|
|
95
|
+
`assertChainReconciles` runs on every build of the artifact and throws unless `input = admitted + sum(excluded)`, the last stage's `remaining` equals `admitted`, and the per-stratum inputs plus the unstratified rows cover the overall input.
|
|
96
|
+
|
|
97
|
+
The stage order is also the order of the checks, and it is cost-ordered.
|
|
98
|
+
Everything decidable from the recording runs before the first container, and the six control rollouts run last.
|
|
99
|
+
A stratum a campaign does not admit costs nothing at all.
|
|
100
|
+
|
|
101
|
+
## A boundary failure is never a verdict
|
|
102
|
+
|
|
103
|
+
Four exclusion reasons exist only because an external call failed: `prefix-replay-error`, `end-state-oracle-error`, `no-fix-control-error`, and `no-op-control-error`.
|
|
104
|
+
|
|
105
|
+
An errored control rollout is not counted as a failed one.
|
|
106
|
+
Counting it as a failure would admit a row nobody verified, which is the same corruption as a silent zero.
|
|
107
|
+
The row leaves with the error reason and the message from the boundary, and it stays visible in the chain.
|
|
108
|
+
|
|
109
|
+
## Analyst independence, in code
|
|
110
|
+
|
|
111
|
+
Two guards, both mechanical.
|
|
112
|
+
|
|
113
|
+
`assertAnalystIndependent(rows)` rejects any row carrying a key outside the closed list `rowId, taskName, recordedModel, recordedCommands, finalReturncode`.
|
|
114
|
+
The check is a closed key list rather than a list of known analyst field names, because the failure to catch is "some new analyst output leaked into the gate".
|
|
115
|
+
|
|
116
|
+
`assertDenominatorIntact({ report, strata, sampled, scored })` rejects the three ways a denominator moves after the fact:
|
|
117
|
+
|
|
118
|
+
- a sampled row that was never admitted,
|
|
119
|
+
- a scored row that was never sampled,
|
|
120
|
+
- a sampled row that was never scored.
|
|
121
|
+
|
|
122
|
+
The third is the one an analyst can cause alone.
|
|
123
|
+
A row it declines still needs an outcome — `no-decisive-failure` is an answer — so a missing outcome raises `denominator shrank by N row(s)` instead of quietly reducing `n`.
|
|
124
|
+
|
|
125
|
+
## What this changes about numbers already produced
|
|
126
|
+
|
|
127
|
+
Runs made before the control was declared and the task oracles were certified are not rewritten here.
|
|
128
|
+
What they measured is stated instead, so a reader can price them.
|
|
129
|
+
|
|
130
|
+
Two facts are measured, not inferred.
|
|
131
|
+
|
|
132
|
+
`largest-eigenval` is graded in part by a wall-clock assertion (`assert dt < ref_dt`, `tests/test_outputs.py:111`), the only one across the four tasks the milestones sampled.
|
|
133
|
+
Certification re-graded each task's published image 16 times with nothing written between the runs — 8 on the untouched image and 8 after the reference solution, half of each under CPU contention — at the same image digests the milestones ran against.
|
|
134
|
+
|
|
135
|
+
| task | replicates | units that flipped | worst per-unit flip | verdict |
|
|
136
|
+
| --- | --- | --- | --- | --- |
|
|
137
|
+
| `password-recovery` | 16 | 0 | 0 % | `CERTIFIED` |
|
|
138
|
+
| `sanitize-git-repo` | 16 | 0 | 0 % | `CERTIFIED` |
|
|
139
|
+
| `count-dataset-tokens` | 16 | 0 | 0 % | `CERTIFIED` |
|
|
140
|
+
| `largest-eigenval` | 16 | 8 | 37.5 % | `NONDETERMINISTIC_ORACLE` |
|
|
141
|
+
|
|
142
|
+
All eight flipped units are `test_speedup[size]` parameters on the untouched image, where `/app/eigen.py` holds the reference implementation the assertion compares against.
|
|
143
|
+
`test_speedup[10]`, `[3]` and `[6]` each returned 5 passes and 3 fails over 8 gradings of identical bytes.
|
|
144
|
+
The whole-suite reward never moved on that state — 0 of 8 — which is why the conjunction has to be counted term by term to see any of this.
|
|
145
|
+
An earlier run of the same certification put the worst unit at 50 %; a flip rate that is itself unstable is what a coin flip looks like.
|
|
146
|
+
|
|
147
|
+
No row from `largest-eigenval` can be admitted while that certification stands.
|
|
148
|
+
|
|
149
|
+
Both milestone runs screened their controls with a policy pinned to zero model calls, so conditions 3 and 4 could never fire for the reason they exist.
|
|
150
|
+
Every control pass those runs recorded fell on `largest-eigenval`: 3 of 3 across 68 row-evaluations, and 0 of 61 on the three tasks whose graders certify stable.
|
|
151
|
+
One row, `largest-eigenval__4GTN8MQ`, was excluded by milestone 1 on a control pass of 1/3 and admitted by milestone 2 on a control pass of 0/3 — the same row and the same control, with opposite verdicts.
|
|
152
|
+
|
|
153
|
+
The consequence differs by run.
|
|
154
|
+
|
|
155
|
+
| run | rows | rows from the timing-graded task | what moves |
|
|
156
|
+
| --- | --- | --- | --- |
|
|
157
|
+
| milestone 1 | 20 evaluated, 17 admitted | 2 evaluated, **0 admitted** | the two exclusions were labelled `no-fix-control-passed`, which claimed the row was repairable by continuing alone; nothing continued. The headline is computed on 17 rows that contain none of them |
|
|
158
|
+
| milestone 2 | 48 evaluated, 43 admitted | 5 evaluated, **4 admitted** | four admitted rows come from a task whose grader is not a function of the state; its admitted set is contaminated and its numbers are conditional on that |
|
|
159
|
+
|
|
160
|
+
Milestone 1's headline separation survives excluding the timing-graded task, because it never included it: oracle-fix `+0.353` over 17 rows and inert-probe `0.000` over the same 17 rows are unchanged when `largest-eigenval` is dropped.
|
|
161
|
+
What does not survive is the reason recorded for its two exclusions, and any reading of that run as evidence that the no-fix control screened anything.
|
|
162
|
+
|
|
163
|
+
## Running it
|
|
164
|
+
|
|
165
|
+
```ts
|
|
166
|
+
import {
|
|
167
|
+
admissionArtifact,
|
|
168
|
+
definePinnedContinuationPolicy,
|
|
169
|
+
renderAdmissionReport,
|
|
170
|
+
runAdmission,
|
|
171
|
+
} from '@tangle-network/agent-eval/../src/trace-repair'
|
|
172
|
+
|
|
173
|
+
const policy = definePinnedContinuationPolicy({ model: 'pinned/model-id', seed: 20260808 })
|
|
174
|
+
|
|
175
|
+
const report = await runAdmission({
|
|
176
|
+
rows, // corpus rows, recording fields only
|
|
177
|
+
policy,
|
|
178
|
+
replayer, // replays the prefix in the task's pinned image
|
|
179
|
+
oracle, // runs the task's held-out tests on the recorded end state
|
|
180
|
+
controls, // restores the state, applies the arm, continues, then grades
|
|
181
|
+
taskOracles: parseTaskOracleRegistry(JSON.parse(readFileSync(taskOraclesPath, 'utf8'))),
|
|
182
|
+
config: { concurrency: 8 },
|
|
183
|
+
})
|
|
184
|
+
|
|
185
|
+
const artifact = admissionArtifact(report)
|
|
186
|
+
writeFileSync('admission.json', JSON.stringify(artifact, null, 2))
|
|
187
|
+
writeFileSync('ADMISSION.md', renderAdmissionReport(artifact, { rowLimit: 50 }))
|
|
188
|
+
```
|
|
189
|
+
|
|
190
|
+
The three boundaries are injected, and each carries an `id` that lands in the artifact.
|
|
191
|
+
A report produced against fakes names those fakes, so it cannot be read as one produced against real containers.
|
|
192
|
+
|
|
193
|
+
The no-op injection step is drawn from the policy seed, the row id, and the rollout index, so it is reproducible from the artifact.
|
|
194
|
+
Every rollout the controls return is checked against the arm, row, and index that was requested, and both arms are checked for a shared policy digest and paired seeds before a row is admitted.
|
|
@@ -0,0 +1,121 @@
|
|
|
1
|
+
# The repair analyst arms
|
|
2
|
+
|
|
3
|
+
The [repair grader](./trace-repair-grader.md) executes an answer.
|
|
4
|
+
This page is about who produces the answer, and about what has to be equal between them before a difference between two arms means anything.
|
|
5
|
+
|
|
6
|
+
An **arm** is one way of executing the repair question: a single chat completion, an agent harness, a DSPy program.
|
|
7
|
+
The question, the reply grammar, the action budget and the bounded repair turn belong to the comparison and not to the arm.
|
|
8
|
+
|
|
9
|
+
## Nothing is certified on this task
|
|
10
|
+
|
|
11
|
+
No arm here runs a certified prompt, and each one says so in its own declaration.
|
|
12
|
+
|
|
13
|
+
The one GEPA-certified analyst artifact this repository holds — `oht2-coverage-instructions.txt` — was earned on the CodeTraceBench incorrect-step task.
|
|
14
|
+
That contract asks for blocks of recorded steps that were wrong, carrying `first_step`, `last_step`, `consequence_step` and `escape_status`.
|
|
15
|
+
The repair contract asks a different question: one step `k`, and one executable action of the same type and budget the recorded scaffold itself took, which must make the task's held-out suite pass.
|
|
16
|
+
The certified text cannot answer that, and re-authoring it for the repair contract would void the certification, because the benchmark it was earned on has been retired.
|
|
17
|
+
|
|
18
|
+
So every arm is authored fresh for the repair task, and every arm is uncertified.
|
|
19
|
+
`repairArmAsymmetries` refuses a set where some arms carry a certification and others do not: an optimisation applies to every arm or to none.
|
|
20
|
+
|
|
21
|
+
## What is equal by construction
|
|
22
|
+
|
|
23
|
+
| Property | Where it is enforced |
|
|
24
|
+
|---|---|
|
|
25
|
+
| One question, one task policy | `repair-prompt.ts` holds them; each arm composes its prompt from those shared constants and declares its own reply grammar as `promptContract` |
|
|
26
|
+
| One action budget (4096 bytes, one top-level statement, one heredoc) | `askRepairArm` measures it; `gradeRepairRow` refuses a violation |
|
|
27
|
+
| One bounded repair turn on a malformed reply | declared per arm as `repairTurns`; `repairArmAsymmetries` refuses a set that disagrees |
|
|
28
|
+
| No arm sees a grading field | `RepairArmRequest` carries only a `BlindedTrajectoryPrefix`; the admitted row is not in the type, so a grading field is unreachable rather than merely unread |
|
|
29
|
+
| A row an arm could not answer is a typed failure | `RepairArmReply` has no shape for a silent null |
|
|
30
|
+
| A k the recording does not hold is a declined reply | both arms drop the row with its reason; the same model mistake lands in the same funnel cell on every execution path |
|
|
31
|
+
|
|
32
|
+
The budget is **recorded** at answer time and **enforced** at grade time.
|
|
33
|
+
One authority refuses; measuring early is what lets a report say what an arm spent its bytes on without paying for a rollout to find out.
|
|
34
|
+
|
|
35
|
+
Prompt identity is recorded at two grains.
|
|
36
|
+
`repairQuestionSha256` digests what every arm shares: the question, the task policy, the budget the caps are read from.
|
|
37
|
+
Each answer additionally stamps `repairArmPromptSha256`, which folds in the arm's own declared contract text — so the chat arms, which ask the identical composed question, share one digest, and the DSPy arm, whose typed SUBMIT grammar is a materially different question, stamps another.
|
|
38
|
+
A contract change, including a `DSPY_REPAIR_SIGNATURE` version bump, changes the per-arm digest.
|
|
39
|
+
|
|
40
|
+
## What is allowed to differ, and is recorded
|
|
41
|
+
|
|
42
|
+
`repairArmAsymmetries` renders the differences beside the result rather than leaving a reader to infer them from two runners' source.
|
|
43
|
+
|
|
44
|
+
| Arm | Executes | Affordances |
|
|
45
|
+
|---|---|---|
|
|
46
|
+
| `bare-framing` | one chat completion, no harness | inline trajectory |
|
|
47
|
+
| `prime` | the prime agent harness through a bridge | inline trajectory, agent loop |
|
|
48
|
+
| `dspy-rlm` | a DSPy RLM program with a typed repair signature | inline trajectory, code interpreter, agent loop |
|
|
49
|
+
|
|
50
|
+
The DSPy arm reads the trajectory with code inside its own environment; the completion arms read it as prompt text.
|
|
51
|
+
That is a real advantage and it is declared, not hidden.
|
|
52
|
+
|
|
53
|
+
## The DSPy arm
|
|
54
|
+
|
|
55
|
+
`createDspyRepairArm` binds the DSPy RLM engine to the repair contract.
|
|
56
|
+
The engine is the incumbent one, unchanged. What is new is a typed signature authored for this task:
|
|
57
|
+
|
|
58
|
+
```
|
|
59
|
+
question, analyst_instructions, trajectory, taskStatement → answer, repairs: list[RepairProposal]
|
|
60
|
+
```
|
|
61
|
+
|
|
62
|
+
`RepairProposal` is a pydantic model in `dspy_rlm_bridge.py`: `k`, `failure_claim`, `intervention_kind`, `action`.
|
|
63
|
+
Its action cap mirrors the scaffold budget the grader enforces, so the typed field cannot refuse an action the grader would have accepted.
|
|
64
|
+
The instructions name the task token `tb-repair-typed-`, which is how the bridge selects this signature; the same mechanism selects the CodeTraceBench signature.
|
|
65
|
+
|
|
66
|
+
The trajectory arrives as `taskInputs` on the engine request and is bound as a variable in the program's environment.
|
|
67
|
+
It is not fetched from a trace store: a Terminal-Bench trajectory is not in one, and a bridge that silently answered without it would be answering about material the caller never delivered — so a repair analysis with no `taskInputs.trajectory` stops the run.
|
|
68
|
+
|
|
69
|
+
Typed rows return under `runtime.repair`, not under `findings`.
|
|
70
|
+
The engine-neutral finding schema caps `recommended_action` at 2000 characters while the scaffold action budget is 4096 bytes, so routing a repair through it would hand this arm a smaller action than every other arm gets — an affordance asymmetry hidden inside a schema.
|
|
71
|
+
The measured comparison run makes that concrete: chat-completion answers ran to 2705, 3413, 3987 and 4733 bytes.
|
|
72
|
+
|
|
73
|
+
## Omni is still blocked, for a different reason than before
|
|
74
|
+
|
|
75
|
+
GEPA's Omni recipe was deferred on 2026-08-03 because no selection metric existed that a candidate could not game.
|
|
76
|
+
The repair grader discharges that: the metric is whether the held-out suite passes after the proposed action executes, which no prose can satisfy.
|
|
77
|
+
|
|
78
|
+
It stays blocked, on two grounds that the grader does not touch.
|
|
79
|
+
|
|
80
|
+
**It is an optimiser, not an arm.**
|
|
81
|
+
Running it produces certified instruction text for one arm.
|
|
82
|
+
The comparison rule above — an optimisation applies to every arm or to none — is now enforced in code, so an Omni-tuned arm beside three untuned ones is a set `repairArmAsymmetries` refuses.
|
|
83
|
+
|
|
84
|
+
**There is no leakage-free split to train on.**
|
|
85
|
+
The corpus records 48 rows from exactly four Terminal-Bench-2 tasks, and admission passes 43: `sanitize-git-repo` 16, `count-dataset-tokens` 12, `password-recovery` 11, `largest-eigenval` 4.
|
|
86
|
+
The 20 pre-registered measurement rows draw from all four.
|
|
87
|
+
The binding constraint on the existing result is the four task clusters, not the 20 rows, so any training split shares clusters with the measurement set at exactly the level that binds.
|
|
88
|
+
|
|
89
|
+
The usable set is smaller still.
|
|
90
|
+
`largest-eigenval` grades speedup with a wall-clock assertion and its suite is certified nondeterministic — 8 of its 27 assertion units flip on byte-identical state, worst-unit flip rate 0.375 — so the oracle-determinism gate refuses every one of its rows, and 16 of the 20 pre-registered rows survive, in three clusters.
|
|
91
|
+
All five `password-recovery` rows among those 16 score zero on the oracle-fix ceiling arm because the reference solution is bash and the scaffold runs dash, which makes their ceiling unmeasured rather than zero.
|
|
92
|
+
That leaves 11 measurement rows in two clusters with a measured ceiling.
|
|
93
|
+
|
|
94
|
+
Omni unblocks when the corpus carries enough independent task clusters to hold out a measurement set that shares none with the training split — not when the grader improves.
|
|
95
|
+
|
|
96
|
+
## Wiring
|
|
97
|
+
|
|
98
|
+
```ts
|
|
99
|
+
import {
|
|
100
|
+
askRepairArm,
|
|
101
|
+
createCompletionRepairArm,
|
|
102
|
+
createDspyRepairArm,
|
|
103
|
+
repairArmAsymmetries,
|
|
104
|
+
repairArmResponse,
|
|
105
|
+
gradeRepairRow,
|
|
106
|
+
} from '@tangle-network/agent-eval/trace-repair'
|
|
107
|
+
|
|
108
|
+
const arms = [bareFraming, prime, dspy]
|
|
109
|
+
// Refuses unequal budgets, unequal repair turns, or a partly certified set.
|
|
110
|
+
const asymmetries = repairArmAsymmetries(arms)
|
|
111
|
+
|
|
112
|
+
for (const arm of arms) {
|
|
113
|
+
const answer = await askRepairArm({ arm, row })
|
|
114
|
+
const response = repairArmResponse(answer)
|
|
115
|
+
if (response === null) continue // the arm failed; the row still counts in the denominator
|
|
116
|
+
const result = await gradeRepairRow({ row, response, sessions, oracle, continuation })
|
|
117
|
+
}
|
|
118
|
+
```
|
|
119
|
+
|
|
120
|
+
Publish `asymmetries` with the result.
|
|
121
|
+
A comparison that reports a difference without reporting what still differs between the arms is not reporting the difference it measured.
|
|
@@ -0,0 +1,107 @@
|
|
|
1
|
+
# TB-Repair continuation policy
|
|
2
|
+
|
|
3
|
+
The continuation policy answers one question: after an analyst names a failing step `k` and proposes a single action, what happens if the agent keeps working from there?
|
|
4
|
+
|
|
5
|
+
`Delta-repair = P(tests pass | intervention) - P(tests pass | no-fix control)` is the headline number of TB-Repair.
|
|
6
|
+
It only measures the intervention when all three arms run forward under the same policy.
|
|
7
|
+
This module makes that symmetry structural rather than promised.
|
|
8
|
+
|
|
9
|
+
Source: [`src/trace-repair/`](../src/trace-repair/).
|
|
10
|
+
The pre-pass that decides which rows the arms run on is in [trace-repair-admission.md](./trace-repair-admission.md).
|
|
11
|
+
|
|
12
|
+
## What runs
|
|
13
|
+
|
|
14
|
+
| element | value | why it is fixed |
|
|
15
|
+
| --- | --- | --- |
|
|
16
|
+
| scaffold | mini-swe-agent | The corpus recorded it, so a continuation stays in the same distribution as the prefix. |
|
|
17
|
+
| step budget | 20 model calls | Bounds a rollout without a wall-clock limit, which would end rollouts at different points. |
|
|
18
|
+
| temperature | 0 | With a fixed seed, the same prefix draws the same continuation. |
|
|
19
|
+
| command timeout | 30 s | The recorded runs used the scaffold's own 30-second limit. A longer limit lets the continuation finish commands the recorded agent could not. |
|
|
20
|
+
| network | `none` | A container with a network can install what the recorded run could not. |
|
|
21
|
+
| format-error cap | 3 consecutive turns | Ends a rollout that has stopped producing actions. |
|
|
22
|
+
| model and seed | chosen per campaign | No default stands in for either. |
|
|
23
|
+
|
|
24
|
+
`definePinnedContinuationPolicy({ model, seed })` freezes a policy and validates it.
|
|
25
|
+
`continuationPolicyDigest(policy)` hashes the policy together with the scaffold text it renders, so an edited template produces a different digest and cannot be pooled with earlier rollouts.
|
|
26
|
+
|
|
27
|
+
The step budget is what lets this policy serve as a control at all.
|
|
28
|
+
A rollout changes the graded state by executing commands, and it executes only what a model call asks for, so at a budget of zero a control arm grades the bytes the end-state check already graded as failing.
|
|
29
|
+
`runAdmission` checks that with `assertControlCalibrated` before it opens a container; the reading of a control pass is in [trace-repair-admission.md](./trace-repair-admission.md).
|
|
30
|
+
|
|
31
|
+
## Why the arms cannot drift apart
|
|
32
|
+
|
|
33
|
+
Three arms share one code path, and the arm is a label on the record:
|
|
34
|
+
|
|
35
|
+
- **intervention** — the analyst's action substituted at step `k`.
|
|
36
|
+
- **no-fix control** — step `k` replayed unchanged.
|
|
37
|
+
- **no-op control** — step `k` replaced by an action that changes nothing.
|
|
38
|
+
|
|
39
|
+
The arms differ only in the container state handed to the runner, which is the treatment under test.
|
|
40
|
+
Three guards keep the policy itself identical:
|
|
41
|
+
|
|
42
|
+
1. `continuationSeed(policySeed, rowId, rolloutIndex)` cannot read the arm, so paired rollouts across arms draw the same seed.
|
|
43
|
+
2. Every rollout records the `policyDigest` it ran under.
|
|
44
|
+
3. `assertArmSymmetry(rollouts)` rejects a set whose rollouts disagree on that digest or on a paired seed.
|
|
45
|
+
|
|
46
|
+
## What a rollout records
|
|
47
|
+
|
|
48
|
+
Every rollout carries the assistant message, the parsed action, the observation, the return code, the timeout flag, per-call latency, the served model id, token usage, and cost.
|
|
49
|
+
|
|
50
|
+
Two record rules keep a partial measurement from reading as a complete one:
|
|
51
|
+
|
|
52
|
+
- `usage.callsWithUsage` below `usage.calls` clears `usage.captured`. Calls that reported nothing are never filled with zeros.
|
|
53
|
+
- One unpriced call makes the whole rollout `{ kind: 'uncaptured', usd: null }`. Summing the priced calls would report less than what was spent.
|
|
54
|
+
|
|
55
|
+
`rolloutRecordedSteps(rollout)` projects the continuation into the `{ src, msg, tools, obs }` steps the trajectory corpus stores, so the replay layer reads a continuation with the reader it already has.
|
|
56
|
+
`rolloutDigest(rollout)` hashes the deterministic content — actions, observations, seeds, exit status, usage — and excludes wall-clock fields, which vary between identical runs.
|
|
57
|
+
|
|
58
|
+
## Exit statuses
|
|
59
|
+
|
|
60
|
+
| status | meaning |
|
|
61
|
+
| --- | --- |
|
|
62
|
+
| `submitted` | A command echoed the sentinel and exited 0. The submission text is recorded. |
|
|
63
|
+
| `step-budget-exhausted` | The rollout used all 20 calls without submitting. |
|
|
64
|
+
| `repeated-format-error` | Three consecutive turns held no single bash block. |
|
|
65
|
+
| `model-error` | The provider call failed. The rollout is recorded with `terminalError`, not dropped. |
|
|
66
|
+
| `environment-error` | A container call failed. The step keeps the action that hit it. |
|
|
67
|
+
|
|
68
|
+
## Running one
|
|
69
|
+
|
|
70
|
+
```ts
|
|
71
|
+
import {
|
|
72
|
+
createDockerContinuationEnvironment,
|
|
73
|
+
definePinnedContinuationPolicy,
|
|
74
|
+
nodeProcessRunner,
|
|
75
|
+
runContinuation,
|
|
76
|
+
} from '@tangle-network/agent-eval/../src/trace-repair'
|
|
77
|
+
|
|
78
|
+
const policy = definePinnedContinuationPolicy({ model: 'pinned/model-id', seed: 20260808 })
|
|
79
|
+
|
|
80
|
+
const rollouts = await runContinuation({
|
|
81
|
+
policy,
|
|
82
|
+
arm: 'intervention',
|
|
83
|
+
rowId: 'break-filter-js-from-html:7',
|
|
84
|
+
prefix, // messages through step k, rebuilt by the replay layer
|
|
85
|
+
rollouts: 3,
|
|
86
|
+
model,
|
|
87
|
+
environments: {
|
|
88
|
+
id: 'docker',
|
|
89
|
+
async create({ arm, rolloutIndex }) {
|
|
90
|
+
return createDockerContinuationEnvironment({
|
|
91
|
+
containerRef: await restoreState({ arm, rolloutIndex }),
|
|
92
|
+
cwd: '/app',
|
|
93
|
+
runProcess: nodeProcessRunner,
|
|
94
|
+
removeOnDispose: true,
|
|
95
|
+
})
|
|
96
|
+
},
|
|
97
|
+
},
|
|
98
|
+
})
|
|
99
|
+
```
|
|
100
|
+
|
|
101
|
+
The runner calls `describe()` before the first model call and refuses any container whose network mode is not `none`.
|
|
102
|
+
The environment reports the mode the daemon holds, not the mode the caller asked for.
|
|
103
|
+
|
|
104
|
+
The prefix must start with a system message then a user message, and must end on a user message.
|
|
105
|
+
A prefix ending on an assistant turn means the replay left an action unanswered, and the runner rejects it.
|
|
106
|
+
|
|
107
|
+
Each rollout gets its own environment, because a rollout mutates the container it runs in.
|