@tangle-network/agent-eval 0.144.6 → 0.144.7
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +24 -0
- package/README.md +2 -0
- package/dist/{benchmark-J9Qe6j2_.d.ts → agent-profile-CgDTo40f.d.ts} +91 -4
- package/dist/agent-profile-CgDTo40f.d.ts.map +1 -0
- package/dist/analyst/index.d.ts +470 -88
- package/dist/analyst/index.d.ts.map +1 -1
- package/dist/analyst/index.js +24 -5
- package/dist/analyst/index.js.map +1 -1
- package/dist/{analyze-runs-DWIvOAGk.js → analyze-runs-BNNK7irB.js} +3 -3
- package/dist/{analyze-runs-DWIvOAGk.js.map → analyze-runs-BNNK7irB.js.map} +1 -1
- package/dist/{baseline-D_fT6277.d.ts → baseline-CavEbRyH.d.ts} +2 -23
- package/dist/baseline-CavEbRyH.d.ts.map +1 -0
- package/dist/{benchmark-command-CQd78YHt.js → benchmark-command-BCafwNrf.js} +662 -605
- package/dist/benchmark-command-BCafwNrf.js.map +1 -0
- package/dist/benchmarks/index.d.ts +1 -1
- package/dist/benchmarks/index.js +1 -1
- package/dist/{benchmarks-BEOkuvIg.js → benchmarks-CDSolHq7.js} +4 -4
- package/dist/{benchmarks-BEOkuvIg.js.map → benchmarks-CDSolHq7.js.map} +1 -1
- package/dist/campaign/index.d.ts +7 -5
- package/dist/campaign/index.js +5 -3
- package/dist/{campaign-CXsdyym7.js → campaign-Tdy3h62h.js} +17 -301
- package/dist/campaign-Tdy3h62h.js.map +1 -0
- package/dist/cli.js +2 -2
- package/dist/{client-0JI64ovJ.d.ts → client-DjXROWpx.d.ts} +3 -3
- package/dist/{client-0JI64ovJ.d.ts.map → client-DjXROWpx.d.ts.map} +1 -1
- package/dist/{completion-verifier-CBiee74w.d.ts → completion-verifier-foUCLif_.d.ts} +5 -5
- package/dist/{completion-verifier-CBiee74w.d.ts.map → completion-verifier-foUCLif_.d.ts.map} +1 -1
- package/dist/contract/index.d.ts +9 -8
- package/dist/contract/index.d.ts.map +1 -1
- package/dist/contract/index.js +7 -6
- package/dist/contract/index.js.map +1 -1
- package/dist/control.d.ts +2 -2
- package/dist/control.js +1 -1
- package/dist/counterfactual-CWPTrMH7.js +126 -0
- package/dist/counterfactual-CWPTrMH7.js.map +1 -0
- package/dist/counterfactual-CxmxAONP.d.ts +72 -0
- package/dist/counterfactual-CxmxAONP.d.ts.map +1 -0
- package/dist/{default-registry-J9m-_tya.d.ts → default-registry-BZhStdjl.d.ts} +6 -5
- package/dist/default-registry-BZhStdjl.d.ts.map +1 -0
- package/dist/{default-registry-Dta70shL.js → default-registry-BaQXW1Ow.js} +2 -2
- package/dist/{default-registry-Dta70shL.js.map → default-registry-BaQXW1Ow.js.map} +1 -1
- package/dist/{dspy-rlm-engine-19FQEMBK.js → dspy-rlm-engine-BiN49gK6.js} +4 -3
- package/dist/{dspy-rlm-engine-19FQEMBK.js.map → dspy-rlm-engine-BiN49gK6.js.map} +1 -1
- package/dist/{tool-groups-CK0JCkqO.d.ts → engine-nB64f48I.d.ts} +18 -31
- package/dist/engine-nB64f48I.d.ts.map +1 -0
- package/dist/{eval-campaign-CfLQQs9B.js → eval-campaign-DNjCvAm-.js} +7 -6
- package/dist/{eval-campaign-CfLQQs9B.js.map → eval-campaign-DNjCvAm-.js.map} +1 -1
- package/dist/{exact-types-CBYF5MGd.d.ts → exact-types-Djvzosly.d.ts} +2 -2
- package/dist/{exact-types-CBYF5MGd.d.ts.map → exact-types-Djvzosly.d.ts.map} +1 -1
- package/dist/exec-BLtYZdWo.js +49 -0
- package/dist/exec-BLtYZdWo.js.map +1 -0
- package/dist/experiment/index.d.ts +802 -0
- package/dist/experiment/index.d.ts.map +1 -0
- package/dist/experiment/index.js +1108 -0
- package/dist/experiment/index.js.map +1 -0
- package/dist/experiment-tracker-CnRICnMl.js +500 -0
- package/dist/experiment-tracker-CnRICnMl.js.map +1 -0
- package/dist/experiment-tracker-IMntXr6J.d.ts +299 -0
- package/dist/experiment-tracker-IMntXr6J.d.ts.map +1 -0
- package/dist/{external-optimizer-contracts-Q9c0L3eY.d.ts → external-optimizer-contracts-CZuJNcT5.d.ts} +2 -2
- package/dist/{external-optimizer-contracts-Q9c0L3eY.d.ts.map → external-optimizer-contracts-CZuJNcT5.d.ts.map} +1 -1
- package/dist/{feedback-trajectory-GgoS0-MK.d.ts → feedback-trajectory-Rh280oXo.d.ts} +3 -3
- package/dist/{feedback-trajectory-GgoS0-MK.d.ts.map → feedback-trajectory-Rh280oXo.d.ts.map} +1 -1
- package/dist/fuzz.d.ts +1 -1
- package/dist/hosted/index.d.ts +3 -3
- package/dist/{index-BZUe-ODI.d.ts → index-BZ3-y4YL.d.ts} +3 -3
- package/dist/{index-BZUe-ODI.d.ts.map → index-BZ3-y4YL.d.ts.map} +1 -1
- package/dist/{index-DSC51roc.d.ts → index-C3ssXVLv.d.ts} +2 -2
- package/dist/{index-DSC51roc.d.ts.map → index-C3ssXVLv.d.ts.map} +1 -1
- package/dist/{index-4XwggC10.d.ts → index-C5HOo4ZF2.d.ts} +4 -4
- package/dist/index-C5HOo4ZF2.d.ts.map +1 -0
- package/dist/{index-B6-B0zTB.d.ts → index-CvXXlyz7.d.ts} +2 -2
- package/dist/{index-B6-B0zTB.d.ts.map → index-CvXXlyz7.d.ts.map} +1 -1
- package/dist/{index-Dx1kF3Ez.d.ts → index-CwDrUMe0.d.ts} +2 -2
- package/dist/{index-Dx1kF3Ez.d.ts.map → index-CwDrUMe0.d.ts.map} +1 -1
- package/dist/{index-BIL5vxxt.d.ts → index-Sh2I0DRc.d.ts} +11 -645
- package/dist/index-Sh2I0DRc.d.ts.map +1 -0
- package/dist/index.d.ts +214 -404
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +233 -649
- package/dist/index.js.map +1 -1
- package/dist/{insight-report-DqEsugpr.d.ts → insight-report-C6h6F_4L.d.ts} +3 -3
- package/dist/{insight-report-DqEsugpr.d.ts.map → insight-report-C6h6F_4L.d.ts.map} +1 -1
- package/dist/{integrity-CNGUaGBY.d.ts → integrity-BuqEKu-x.d.ts} +2 -2
- package/dist/{integrity-CNGUaGBY.d.ts.map → integrity-BuqEKu-x.d.ts.map} +1 -1
- package/dist/integrity-MLzHOfV9.js +141 -0
- package/dist/integrity-MLzHOfV9.js.map +1 -0
- package/dist/kind-factory-BHIgPmzS.js.map +1 -1
- package/dist/{llm-client-Dv5BiKLE.js → llm-client-DzvMUsS_.js} +24 -6
- package/dist/llm-client-DzvMUsS_.js.map +1 -0
- package/dist/matrix/index.d.ts +2 -2
- package/dist/meta-eval/index.d.ts +1 -1
- package/dist/{mint-DD-0oQTA.js → mint-CGEkzPLf.js} +2 -2
- package/dist/{mint-DD-0oQTA.js.map → mint-CGEkzPLf.js.map} +1 -1
- package/dist/{multi-layer-verifier-BHY1gWAc.d.ts → multi-layer-verifier-DnAqwl0h.d.ts} +2 -2
- package/dist/{multi-layer-verifier-BHY1gWAc.d.ts.map → multi-layer-verifier-DnAqwl0h.d.ts.map} +1 -1
- package/dist/multishot/index.d.ts +2 -2
- package/dist/openapi.json +1 -1
- package/dist/paired-promotion-decision-B6zJ3gYM.d.ts +114 -0
- package/dist/paired-promotion-decision-B6zJ3gYM.d.ts.map +1 -0
- package/dist/pipelines/index.d.ts +2 -1
- package/dist/pipelines/index.d.ts.map +1 -1
- package/dist/pre-registration-DakwTRXk.js +96 -0
- package/dist/pre-registration-DakwTRXk.js.map +1 -0
- package/dist/prime-bridge-transport-6feEglLf.d.ts +49 -0
- package/dist/prime-bridge-transport-6feEglLf.d.ts.map +1 -0
- package/dist/prime-protocol-BfSalTfR.js +453 -0
- package/dist/prime-protocol-BfSalTfR.js.map +1 -0
- package/dist/profile-cell.js +242 -1
- package/dist/profile-cell.js.map +1 -0
- package/dist/promotion-policy-ChWhTDBH.d.ts +289 -0
- package/dist/promotion-policy-ChWhTDBH.d.ts.map +1 -0
- package/dist/promotion-policy-CrLrmys8.js +682 -0
- package/dist/promotion-policy-CrLrmys8.js.map +1 -0
- package/dist/{propose-review-control-BciUCZoh.js → propose-review-control-qgWLA6E9.js} +2 -2
- package/dist/{propose-review-control-BciUCZoh.js.map → propose-review-control-qgWLA6E9.js.map} +1 -1
- package/dist/{release-report-ChOgpIoQ.d.ts → release-report-CI8uisI1.d.ts} +2 -2
- package/dist/{release-report-ChOgpIoQ.d.ts.map → release-report-CI8uisI1.d.ts.map} +1 -1
- package/dist/{release-report-Sl0xfkFv.js → release-report-Dy39cbFF.js} +3 -3
- package/dist/{release-report-Sl0xfkFv.js.map → release-report-Dy39cbFF.js.map} +1 -1
- package/dist/{replay-Krvb114g.d.ts → replay-DFf-teiC.d.ts} +5 -4
- package/dist/replay-DFf-teiC.d.ts.map +1 -0
- package/dist/{replay-GW61ezMW.js → replay-DyBLaKFc.js} +3 -3
- package/dist/{replay-GW61ezMW.js.map → replay-DyBLaKFc.js.map} +1 -1
- package/dist/reporting.d.ts +3 -3
- package/dist/reporting.js +2 -2
- package/dist/{researcher-xLeNcpKX.d.ts → researcher-BoaxeCzP.d.ts} +4 -4
- package/dist/{researcher-xLeNcpKX.d.ts.map → researcher-BoaxeCzP.d.ts.map} +1 -1
- package/dist/{reward-hacking-CyuzxKly.js → reward-hacking-B2tPL9a3.js} +12 -3
- package/dist/reward-hacking-B2tPL9a3.js.map +1 -0
- package/dist/{reward-hacking-RZgnGWlx.d.ts → reward-hacking-Cf1PtEOz.d.ts} +33 -3
- package/dist/reward-hacking-Cf1PtEOz.d.ts.map +1 -0
- package/dist/rl.d.ts +17 -7
- package/dist/rl.d.ts.map +1 -1
- package/dist/rl.js +16 -7
- package/dist/rl.js.map +1 -1
- package/dist/rollout/index.d.ts +2 -2
- package/dist/rollout/index.js +2 -2
- package/dist/{rollout-C2fD1cf4.js → rollout-2ECTXb2N.js} +2 -2
- package/dist/{rollout-C2fD1cf4.js.map → rollout-2ECTXb2N.js.map} +1 -1
- package/dist/{run-evidence-C6G41MSI.d.ts → run-evidence-BDFFai9R.d.ts} +2 -2
- package/dist/{run-evidence-C6G41MSI.d.ts.map → run-evidence-BDFFai9R.d.ts.map} +1 -1
- package/dist/{run-record-CWN8-VsV.js → run-record-DqOw5X6_.js} +2 -2
- package/dist/{run-record-CWN8-VsV.js.map → run-record-DqOw5X6_.js.map} +1 -1
- package/dist/{schema-Cef2cFmb.d.ts → schema-Cef2cFmb2.d.ts} +1 -1
- package/dist/schema-Cef2cFmb2.d.ts.map +1 -0
- package/dist/{semantic-concept-judge-DwF6n05O.js → semantic-concept-judge-Bmrq6yqU.js} +2 -2
- package/dist/{semantic-concept-judge-DwF6n05O.js.map → semantic-concept-judge-Bmrq6yqU.js.map} +1 -1
- package/dist/sequential-D-BLJBKU.js +299 -0
- package/dist/sequential-D-BLJBKU.js.map +1 -0
- package/dist/{server-D6XJQHw7.js → server-iu0ede49.js} +2 -2
- package/dist/{server-D6XJQHw7.js.map → server-iu0ede49.js.map} +1 -1
- package/dist/{single-run-lock-D5iN0Xzb.js → single-run-lock-BMQEv1wG.js} +7 -161
- package/dist/single-run-lock-BMQEv1wG.js.map +1 -0
- package/dist/{skill-usage-GlOphAhX.d.ts → skill-usage-CJlWEUFt.d.ts} +10 -10
- package/dist/{skill-usage-GlOphAhX.d.ts.map → skill-usage-CJlWEUFt.d.ts.map} +1 -1
- package/dist/{skillopt-optimization-method-7S43rbDB.d.ts → skillopt-optimization-method-BO7NAl3b.d.ts} +8 -294
- package/dist/skillopt-optimization-method-BO7NAl3b.d.ts.map +1 -0
- package/dist/{skillopt-optimization-method-Bfb-vBKe.js → skillopt-optimization-method-CQwZ-ZX8.js} +8 -652
- package/dist/skillopt-optimization-method-CQwZ-ZX8.js.map +1 -0
- package/dist/statistical-heldout-Dn9ruizm.d.ts +555 -0
- package/dist/statistical-heldout-Dn9ruizm.d.ts.map +1 -0
- package/dist/{statistics-C-dm-J6H.d.ts → statistics-D6Uebe_4.d.ts} +2 -2
- package/dist/{statistics-C-dm-J6H.d.ts.map → statistics-D6Uebe_4.d.ts.map} +1 -1
- package/dist/steps-BArUxhna.d.ts +51 -0
- package/dist/steps-BArUxhna.d.ts.map +1 -0
- package/dist/{integrity-fdt8XPAv.js → store-DNe_Uv1Q.js} +2 -140
- package/dist/store-DNe_Uv1Q.js.map +1 -0
- package/dist/{summary-report-B0cAyA7N.d.ts → summary-report-DuUS_i7W.d.ts} +3 -114
- package/dist/summary-report-DuUS_i7W.d.ts.map +1 -0
- package/dist/{summary-report-9A5y7EsK.js → summary-report-Lf-5I7xh.js} +2 -2
- package/dist/{summary-report-9A5y7EsK.js.map → summary-report-Lf-5I7xh.js.map} +1 -1
- package/dist/supervisor-run/index.d.ts +2 -2
- package/dist/supervisor-run/index.js +1 -1
- package/dist/{supervisor-run-DiyQVczd.js → supervisor-run-D_sokXcO.js} +32 -8
- package/dist/{supervisor-run-DiyQVczd.js.map → supervisor-run-D_sokXcO.js.map} +1 -1
- package/dist/tool-groups-DIVBnJyl.d.ts +28 -0
- package/dist/tool-groups-DIVBnJyl.d.ts.map +1 -0
- package/dist/trace-repair/index.d.ts +2102 -0
- package/dist/trace-repair/index.d.ts.map +1 -0
- package/dist/trace-repair/index.js +3878 -0
- package/dist/trace-repair/index.js.map +1 -0
- package/dist/traces.d.ts +6 -6
- package/dist/traces.js +3 -2
- package/dist/trajectory-YC15QDYQ.d.ts +24 -0
- package/dist/trajectory-YC15QDYQ.d.ts.map +1 -0
- package/dist/trajectory-replay/index.d.ts +781 -0
- package/dist/trajectory-replay/index.d.ts.map +1 -0
- package/dist/trajectory-replay/index.js +2103 -0
- package/dist/trajectory-replay/index.js.map +1 -0
- package/dist/{types-XMVEdrE_.d.ts → types-D216SgwM.d.ts} +24 -6
- package/dist/types-D216SgwM.d.ts.map +1 -0
- package/dist/{types-yLK8gXE9.d.ts → types-D4mog56g.d.ts} +2 -2
- package/dist/{types-yLK8gXE9.d.ts.map → types-D4mog56g.d.ts.map} +1 -1
- package/dist/{types-BhP9q0Fq.d.ts → types-DF_Udrp-.d.ts} +52 -3
- package/dist/{types-BhP9q0Fq.d.ts.map → types-DF_Udrp-.d.ts.map} +1 -1
- package/dist/{types-DOZyvsFU.d.ts → types-DYuNHo9R.d.ts} +3 -3
- package/dist/{types-DOZyvsFU.d.ts.map → types-DYuNHo9R.d.ts.map} +1 -1
- package/dist/usage-receipt-EVI8B8Xu.js.map +1 -1
- package/dist/verdict-DExhxfgR.d.ts +201 -0
- package/dist/verdict-DExhxfgR.d.ts.map +1 -0
- package/dist/verdict-cache-BCcOh0kF.js +159 -0
- package/dist/verdict-cache-BCcOh0kF.js.map +1 -0
- package/dist/wire/index.d.ts +2 -2
- package/dist/wire/index.js +1 -1
- package/docs/charter.md +112 -0
- package/docs/experiment.md +104 -0
- package/docs/prime-analyst.md +1 -0
- package/docs/trace-analysis.md +26 -0
- package/docs/trace-repair-admission.md +194 -0
- package/docs/trace-repair-analyst-arms.md +121 -0
- package/docs/trace-repair-continuation.md +107 -0
- package/docs/trace-repair-grader.md +163 -0
- package/docs/trajectory-replay.md +110 -0
- package/docs/verification-strategies.md +103 -0
- package/package.json +19 -2
- package/dist/agent-profile-cell-CbfBm2g6.js +0 -335
- package/dist/agent-profile-cell-CbfBm2g6.js.map +0 -1
- package/dist/baseline-D_fT6277.d.ts.map +0 -1
- package/dist/benchmark-J9Qe6j2_.d.ts.map +0 -1
- package/dist/benchmark-command-CQd78YHt.js.map +0 -1
- package/dist/campaign-CXsdyym7.js.map +0 -1
- package/dist/default-registry-J9m-_tya.d.ts.map +0 -1
- package/dist/index-4XwggC10.d.ts.map +0 -1
- package/dist/index-BIL5vxxt.d.ts.map +0 -1
- package/dist/integrity-fdt8XPAv.js.map +0 -1
- package/dist/llm-client-Dv5BiKLE.js.map +0 -1
- package/dist/replay-Krvb114g.d.ts.map +0 -1
- package/dist/reward-hacking-CyuzxKly.js.map +0 -1
- package/dist/reward-hacking-RZgnGWlx.d.ts.map +0 -1
- package/dist/schema-Cef2cFmb.d.ts.map +0 -1
- package/dist/single-run-lock-D5iN0Xzb.js.map +0 -1
- package/dist/skillopt-optimization-method-7S43rbDB.d.ts.map +0 -1
- package/dist/skillopt-optimization-method-Bfb-vBKe.js.map +0 -1
- package/dist/summary-report-B0cAyA7N.d.ts.map +0 -1
- package/dist/tool-groups-CK0JCkqO.d.ts.map +0 -1
- package/dist/types-XMVEdrE_.d.ts.map +0 -1
- package/dist/verdict-Dps8_okt.d.ts +0 -37
- package/dist/verdict-Dps8_okt.d.ts.map +0 -1
|
@@ -0,0 +1,107 @@
|
|
|
1
|
+
# TB-Repair continuation policy
|
|
2
|
+
|
|
3
|
+
The continuation policy answers one question: after an analyst names a failing step `k` and proposes a single action, what happens if the agent keeps working from there?
|
|
4
|
+
|
|
5
|
+
`Delta-repair = P(tests pass | intervention) - P(tests pass | no-fix control)` is the headline number of TB-Repair.
|
|
6
|
+
It only measures the intervention when all three arms run forward under the same policy.
|
|
7
|
+
This module makes that symmetry structural rather than promised.
|
|
8
|
+
|
|
9
|
+
Source: [`src/trace-repair/`](../src/trace-repair/).
|
|
10
|
+
The pre-pass that decides which rows the arms run on is in [trace-repair-admission.md](./trace-repair-admission.md).
|
|
11
|
+
|
|
12
|
+
## What runs
|
|
13
|
+
|
|
14
|
+
| element | value | why it is fixed |
|
|
15
|
+
| --- | --- | --- |
|
|
16
|
+
| scaffold | mini-swe-agent | The corpus recorded it, so a continuation stays in the same distribution as the prefix. |
|
|
17
|
+
| step budget | 20 model calls | Bounds a rollout without a wall-clock limit, which would end rollouts at different points. |
|
|
18
|
+
| temperature | 0 | With a fixed seed, the same prefix draws the same continuation. |
|
|
19
|
+
| command timeout | 30 s | The recorded runs used the scaffold's own 30-second limit. A longer limit lets the continuation finish commands the recorded agent could not. |
|
|
20
|
+
| network | `none` | A container with a network can install what the recorded run could not. |
|
|
21
|
+
| format-error cap | 3 consecutive turns | Ends a rollout that has stopped producing actions. |
|
|
22
|
+
| model and seed | chosen per campaign | No default stands in for either. |
|
|
23
|
+
|
|
24
|
+
`definePinnedContinuationPolicy({ model, seed })` freezes a policy and validates it.
|
|
25
|
+
`continuationPolicyDigest(policy)` hashes the policy together with the scaffold text it renders, so an edited template produces a different digest and cannot be pooled with earlier rollouts.
|
|
26
|
+
|
|
27
|
+
The step budget is what lets this policy serve as a control at all.
|
|
28
|
+
A rollout changes the graded state by executing commands, and it executes only what a model call asks for, so at a budget of zero a control arm grades the bytes the end-state check already graded as failing.
|
|
29
|
+
`runAdmission` checks that with `assertControlCalibrated` before it opens a container; the reading of a control pass is in [trace-repair-admission.md](./trace-repair-admission.md).
|
|
30
|
+
|
|
31
|
+
## Why the arms cannot drift apart
|
|
32
|
+
|
|
33
|
+
Three arms share one code path, and the arm is a label on the record:
|
|
34
|
+
|
|
35
|
+
- **intervention** — the analyst's action substituted at step `k`.
|
|
36
|
+
- **no-fix control** — step `k` replayed unchanged.
|
|
37
|
+
- **no-op control** — step `k` replaced by an action that changes nothing.
|
|
38
|
+
|
|
39
|
+
The arms differ only in the container state handed to the runner, which is the treatment under test.
|
|
40
|
+
Three guards keep the policy itself identical:
|
|
41
|
+
|
|
42
|
+
1. `continuationSeed(policySeed, rowId, rolloutIndex)` cannot read the arm, so paired rollouts across arms draw the same seed.
|
|
43
|
+
2. Every rollout records the `policyDigest` it ran under.
|
|
44
|
+
3. `assertArmSymmetry(rollouts)` rejects a set whose rollouts disagree on that digest or on a paired seed.
|
|
45
|
+
|
|
46
|
+
## What a rollout records
|
|
47
|
+
|
|
48
|
+
Every rollout carries the assistant message, the parsed action, the observation, the return code, the timeout flag, per-call latency, the served model id, token usage, and cost.
|
|
49
|
+
|
|
50
|
+
Two record rules keep a partial measurement from reading as a complete one:
|
|
51
|
+
|
|
52
|
+
- `usage.callsWithUsage` below `usage.calls` clears `usage.captured`. Calls that reported nothing are never filled with zeros.
|
|
53
|
+
- One unpriced call makes the whole rollout `{ kind: 'uncaptured', usd: null }`. Summing the priced calls would report less than what was spent.
|
|
54
|
+
|
|
55
|
+
`rolloutRecordedSteps(rollout)` projects the continuation into the `{ src, msg, tools, obs }` steps the trajectory corpus stores, so the replay layer reads a continuation with the reader it already has.
|
|
56
|
+
`rolloutDigest(rollout)` hashes the deterministic content — actions, observations, seeds, exit status, usage — and excludes wall-clock fields, which vary between identical runs.
|
|
57
|
+
|
|
58
|
+
## Exit statuses
|
|
59
|
+
|
|
60
|
+
| status | meaning |
|
|
61
|
+
| --- | --- |
|
|
62
|
+
| `submitted` | A command echoed the sentinel and exited 0. The submission text is recorded. |
|
|
63
|
+
| `step-budget-exhausted` | The rollout used all 20 calls without submitting. |
|
|
64
|
+
| `repeated-format-error` | Three consecutive turns held no single bash block. |
|
|
65
|
+
| `model-error` | The provider call failed. The rollout is recorded with `terminalError`, not dropped. |
|
|
66
|
+
| `environment-error` | A container call failed. The step keeps the action that hit it. |
|
|
67
|
+
|
|
68
|
+
## Running one
|
|
69
|
+
|
|
70
|
+
```ts
|
|
71
|
+
import {
|
|
72
|
+
createDockerContinuationEnvironment,
|
|
73
|
+
definePinnedContinuationPolicy,
|
|
74
|
+
nodeProcessRunner,
|
|
75
|
+
runContinuation,
|
|
76
|
+
} from '@tangle-network/agent-eval/../src/trace-repair'
|
|
77
|
+
|
|
78
|
+
const policy = definePinnedContinuationPolicy({ model: 'pinned/model-id', seed: 20260808 })
|
|
79
|
+
|
|
80
|
+
const rollouts = await runContinuation({
|
|
81
|
+
policy,
|
|
82
|
+
arm: 'intervention',
|
|
83
|
+
rowId: 'break-filter-js-from-html:7',
|
|
84
|
+
prefix, // messages through step k, rebuilt by the replay layer
|
|
85
|
+
rollouts: 3,
|
|
86
|
+
model,
|
|
87
|
+
environments: {
|
|
88
|
+
id: 'docker',
|
|
89
|
+
async create({ arm, rolloutIndex }) {
|
|
90
|
+
return createDockerContinuationEnvironment({
|
|
91
|
+
containerRef: await restoreState({ arm, rolloutIndex }),
|
|
92
|
+
cwd: '/app',
|
|
93
|
+
runProcess: nodeProcessRunner,
|
|
94
|
+
removeOnDispose: true,
|
|
95
|
+
})
|
|
96
|
+
},
|
|
97
|
+
},
|
|
98
|
+
})
|
|
99
|
+
```
|
|
100
|
+
|
|
101
|
+
The runner calls `describe()` before the first model call and refuses any container whose network mode is not `none`.
|
|
102
|
+
The environment reports the mode the daemon holds, not the mode the caller asked for.
|
|
103
|
+
|
|
104
|
+
The prefix must start with a system message then a user message, and must end on a user message.
|
|
105
|
+
A prefix ending on an assistant turn means the replay left an action unanswered, and the runner rejects it.
|
|
106
|
+
|
|
107
|
+
Each rollout gets its own environment, because a rollout mutates the container it runs in.
|
|
@@ -0,0 +1,163 @@
|
|
|
1
|
+
# The repair grader
|
|
2
|
+
|
|
3
|
+
`@tangle-network/agent-eval/trace-repair` grades one claim about a recorded failure by executing the repair it proposes.
|
|
4
|
+
|
|
5
|
+
An analyst reads a blinded trajectory prefix and answers with exactly one finding — a step `k`, what went wrong there, and one action to run instead — or with the literal `no-decisive-failure`.
|
|
6
|
+
The grader turns that answer into a measured difference.
|
|
7
|
+
|
|
8
|
+
[The analyst arms](./trace-repair-analyst-arms.md) page covers who produces the answer, and what has to be equal between two analysts before their difference means anything.
|
|
9
|
+
|
|
10
|
+
## The headline
|
|
11
|
+
|
|
12
|
+
```
|
|
13
|
+
Delta-repair = P(tests pass | intervention) − P(tests pass | no-fix control)
|
|
14
|
+
```
|
|
15
|
+
|
|
16
|
+
Paired per row, then bootstrapped over rows.
|
|
17
|
+
Every admitted row stays in the denominator, including the rows an analyst declined and the answers the grader rejected.
|
|
18
|
+
Those rows contribute a paired difference of exactly zero, because with no intervention to run their arm is their control arm.
|
|
19
|
+
|
|
20
|
+
## The five tiers
|
|
21
|
+
|
|
22
|
+
| Tier | Question | Pays |
|
|
23
|
+
|---|---|---|
|
|
24
|
+
| t0 parsed | Is this one answer, inside the action budget? | — |
|
|
25
|
+
| t1 reproduced | Does the recorded state at `k` come back under replay? | **nothing** |
|
|
26
|
+
| t2 executes | Does the intervention run at `k`? | `executes` |
|
|
27
|
+
| t3 local flip | Does the held-out suite pass immediately after it? | `localFlip` |
|
|
28
|
+
| t4 repair flip | Does the suite pass after the pinned policy continues? | `repairRate` |
|
|
29
|
+
|
|
30
|
+
t0, t1 and t2 are nested gates.
|
|
31
|
+
t3 and t4 are two measurements of the same intervention, and neither contains the other: an intervention that fixes the task outright flips locally, and one that unblocks an agent with work still to do flips only after the continuation.
|
|
32
|
+
|
|
33
|
+
**t1 pays nothing, structurally.**
|
|
34
|
+
`RepairCredit` has three numeric terms and no reproduction term, so there is no field a reproduced step could pay into.
|
|
35
|
+
Naming the first step with a nonzero exit code earns exactly what naming a step that does not reproduce earns.
|
|
36
|
+
|
|
37
|
+
Reproduction is not "a nonzero exit came back".
|
|
38
|
+
Most admitted rows end on a clean exit, so the gate asks whether the recorded state at `k` reproduces, whatever that state was, including a returncode of zero.
|
|
39
|
+
A step that recorded no observation has nothing to reproduce; the gate passes vacuously, opens no container, and records `basis: 'no-recorded-observation'`.
|
|
40
|
+
|
|
41
|
+
## The action budget
|
|
42
|
+
|
|
43
|
+
The intervention is one action from the action space the scaffold had.
|
|
44
|
+
|
|
45
|
+
| Rule | Default | Why |
|
|
46
|
+
|---|---|---|
|
|
47
|
+
| One top-level statement | `maxStatements: 1` | A command list joined by `&&`, `\|\|` or a pipe is one statement. Two statements separated by a newline or `;` are two actions. |
|
|
48
|
+
| One authored file | `maxHeredocs: 1` | An `edit` writes file content inline through one heredoc. |
|
|
49
|
+
| 4 KB | `maxBytes: 4096` | The scaffold's per-action budget. |
|
|
50
|
+
|
|
51
|
+
Heredoc bodies, comments and compound blocks (`if`, `for`, `while`, `until`, `case`, `{ … }`) are inside a statement, never separators.
|
|
52
|
+
An action the scanner cannot resolve stays inside one oversized statement and is rejected on bytes, so an unparseable answer fails closed.
|
|
53
|
+
|
|
54
|
+
The declared kind must match the action: an answer that calls a file rewrite a shell command is rejected rather than silently corrected.
|
|
55
|
+
|
|
56
|
+
## Admission
|
|
57
|
+
|
|
58
|
+
Five checks, none of which reads a finding, a `k`, or a label.
|
|
59
|
+
Four are anchored at the recorded end state, so the same evidence admits a row whatever an analyst later blames; the fifth is about the task rather than the row.
|
|
60
|
+
|
|
61
|
+
| Check | Criterion |
|
|
62
|
+
|---|---|
|
|
63
|
+
| oracle determinism | the task's own suite returned one verdict per assertion on byte-identical state |
|
|
64
|
+
| prefix fidelity | at most 10 % of replayed steps diverge from their recorded returncode |
|
|
65
|
+
| end state fails | the held-out suite fails on the recorded end state |
|
|
66
|
+
| no-fix control | 3 of 3 continuations from the end state fail |
|
|
67
|
+
| no-op control | 3 of 3 continuations from the end state, after an action that changes nothing, fail |
|
|
68
|
+
|
|
69
|
+
The two controls only screen under a control that can act.
|
|
70
|
+
A control with a step budget of zero executes no command, so it grades the bytes the end-state check already read as failing, and a pass there is the task's grader disagreeing with itself.
|
|
71
|
+
`admitRow` refuses that pairing at the call rather than admitting every row through a check that cannot fire; see [trace-repair-admission.md](./trace-repair-admission.md).
|
|
72
|
+
|
|
73
|
+
`admitRow` is pure: a campaign runner executes the checks against real containers and hands it the measured evidence.
|
|
74
|
+
Splitting it that way keeps every admission re-derivable from the recorded numbers without re-running a container.
|
|
75
|
+
|
|
76
|
+
An admitted row carries a phantom brand that only `admitRow` can produce.
|
|
77
|
+
`blindTrajectory` and `gradeRepairRow` both require one, so there is no signature anywhere that shows an unadmitted row to an analyst or grades an answer about one.
|
|
78
|
+
|
|
79
|
+
## What defeats the degenerate answers
|
|
80
|
+
|
|
81
|
+
`DEGENERATE_STRATEGIES` names each one in code, with the mechanism that removes its reward and where that mechanism lives.
|
|
82
|
+
A gate refuses the answer before a rollout is spent; a measurement lets it run and measures it at its control.
|
|
83
|
+
|
|
84
|
+
| Strategy | Defeated by |
|
|
85
|
+
|---|---|
|
|
86
|
+
| point at any nonzero-exit step | gate — reproduction pays nothing |
|
|
87
|
+
| propose the recorded command again | gate — compared against the recorded action at `k` |
|
|
88
|
+
| propose a no-op | gate for the literal ones, measurement for a semantic one |
|
|
89
|
+
| submit instead of repair | gate — the sentinel is rejected |
|
|
90
|
+
| touch the test suite | gate — the oracle injects the suite from outside and verifies the bytes it reads back |
|
|
91
|
+
| buy a bigger action | gate — statements, heredocs and bytes |
|
|
92
|
+
| decline every hard row | measurement — a declined row keeps its cell and its place in the denominator |
|
|
93
|
+
| repair somewhere other than `k` | measurement — the intervention runs at the named `k` only |
|
|
94
|
+
|
|
95
|
+
The suite is the load-bearing one.
|
|
96
|
+
`injectedTestOracle` purges the suite root, uploads the held-out suite from outside the session, reads the bytes back from inside and hashes them, and refuses to grade when the read-back digest is not the uploaded digest.
|
|
97
|
+
A container that silently drops the upload raises `TestSuiteTamperedError` rather than returning a result.
|
|
98
|
+
|
|
99
|
+
Ordering carries the same weight as the upload.
|
|
100
|
+
A repair rollout replays the prefix, runs the intervention, hands the container to the continuation policy, and only then calls the oracle.
|
|
101
|
+
The continuing agent therefore works in a container where the suite does not exist yet, and cannot read the thing that will grade it.
|
|
102
|
+
|
|
103
|
+
Every arm is graded against the suite the row was admitted against.
|
|
104
|
+
A grade produced from a different suite digest raises rather than being compared, because two arms that answered different questions are not a difference.
|
|
105
|
+
|
|
106
|
+
## Wiring
|
|
107
|
+
|
|
108
|
+
Three injected ports, one job each.
|
|
109
|
+
|
|
110
|
+
```ts
|
|
111
|
+
import {
|
|
112
|
+
admitRow,
|
|
113
|
+
blindTrajectory,
|
|
114
|
+
deltaRepair,
|
|
115
|
+
gradeRepairRow,
|
|
116
|
+
injectedTestOracle,
|
|
117
|
+
parseAnalystResponse,
|
|
118
|
+
renderDeltaRepairReport,
|
|
119
|
+
} from '@tangle-network/agent-eval/trace-repair'
|
|
120
|
+
|
|
121
|
+
const outcome = admitRow(evidence)
|
|
122
|
+
if (!outcome.admitted) return
|
|
123
|
+
|
|
124
|
+
const prompt = blindTrajectory(outcome.row)
|
|
125
|
+
const answer = parseAnalystResponse(await analyst(prompt))
|
|
126
|
+
if (!answer.succeeded) return
|
|
127
|
+
|
|
128
|
+
const graded = await gradeRepairRow({
|
|
129
|
+
row: outcome.row,
|
|
130
|
+
response: answer.value,
|
|
131
|
+
sessions, // a fresh container at the trajectory's own image
|
|
132
|
+
oracle, // injectedTestOracle({ files, command, purge })
|
|
133
|
+
continuation, // the pinned policy, run forward from a prepared session
|
|
134
|
+
})
|
|
135
|
+
|
|
136
|
+
const report = deltaRepair(rows)
|
|
137
|
+
console.log(renderDeltaRepairReport(report))
|
|
138
|
+
```
|
|
139
|
+
|
|
140
|
+
The image must be the published one the trajectory was recorded against.
|
|
141
|
+
A locally rebuilt image drifts from the recording through unpinned apt and pip installs, and every number measured on it is about a different environment.
|
|
142
|
+
|
|
143
|
+
## Honest limits
|
|
144
|
+
|
|
145
|
+
The arms are matched on policy, budget and suite, and not on position.
|
|
146
|
+
The no-fix control continues from the recorded end state; the intervention arm continues from `k` and has to redo the work the recording did after `k` inside the same step budget.
|
|
147
|
+
That asymmetry biases against the intervention, and it travels with the number as the `control-position-asymmetry` threat.
|
|
148
|
+
|
|
149
|
+
Every admitted row has a control rate of zero by admission, so Delta-repair equals the intervention rate on an admitted corpus.
|
|
150
|
+
The estimate is conditional on that admission and says nothing about rows the control can already repair.
|
|
151
|
+
|
|
152
|
+
Under `controlScreening: 'declared-inert'` that zero is weaker still: the control made no model call, so a control rate of zero restates the end-state check instead of measuring what continuing alone can repair.
|
|
153
|
+
Rows screened that way carry the `control-cannot-rescue` threat into the report, so the caveat travels with the number.
|
|
154
|
+
|
|
155
|
+
Oracle determinism is certified per task, not per row, and it is certified at the two states certification can construct: the published image and that image after the reference solution ran.
|
|
156
|
+
Those are anchors, and a suite can be steady at an anchor while flipping near its threshold.
|
|
157
|
+
Per-assertion counting is what makes the anchor informative — a suite whose per-parameter timing assertions flip shows it there even when the whole-suite reward does not — but the certification is still a measurement at two states and not a proof about every state a campaign will grade.
|
|
158
|
+
|
|
159
|
+
A command-level repair cannot address a run the harness killed at a timeout.
|
|
160
|
+
Split that class out before sampling; the grader measures actions, not wall clock.
|
|
161
|
+
|
|
162
|
+
The interval carries `gateEligible`, which is false below the pair count where a percentile bootstrap holds its nominal error rate.
|
|
163
|
+
Below it the interval is descriptive spread and a promotion must not turn on it.
|
|
@@ -0,0 +1,110 @@
|
|
|
1
|
+
# Trajectory replay
|
|
2
|
+
|
|
3
|
+
`@tangle-network/agent-eval/trajectory-replay` re-executes a recorded shell trajectory and produces a verdict about what really happened.
|
|
4
|
+
|
|
5
|
+
The primitive is one question: replay steps `1..k-1` of a recording inside the image it ran in, execute step `k`, and does the recorded failure come back?
|
|
6
|
+
That question needs a recording, an image, and a way to run a command.
|
|
7
|
+
It does not need a running agent loop, so it is substrate, not runtime.
|
|
8
|
+
|
|
9
|
+
## The one thing you must supply
|
|
10
|
+
|
|
11
|
+
The package never depends on a sandbox client, a container runtime, or a model provider.
|
|
12
|
+
Every entry point takes the execution boundary as an argument.
|
|
13
|
+
|
|
14
|
+
```ts
|
|
15
|
+
import {
|
|
16
|
+
type ReplayExecBackend,
|
|
17
|
+
replayVerify,
|
|
18
|
+
} from '@tangle-network/agent-eval/trajectory-replay'
|
|
19
|
+
|
|
20
|
+
const backend: ReplayExecBackend = {
|
|
21
|
+
async open() {
|
|
22
|
+
const box = await myPlatform.create({ image })
|
|
23
|
+
return {
|
|
24
|
+
async exec(command, timeoutMs) {
|
|
25
|
+
const r = await box.exec(command, { timeoutMs })
|
|
26
|
+
return { exitCode: r.exitCode, stdout: r.stdout, stderr: r.stderr }
|
|
27
|
+
},
|
|
28
|
+
async close() {
|
|
29
|
+
await box.delete()
|
|
30
|
+
},
|
|
31
|
+
}
|
|
32
|
+
},
|
|
33
|
+
}
|
|
34
|
+
|
|
35
|
+
const verdict = await replayVerify({
|
|
36
|
+
stepsPath: 'normalized/<traj>/steps.json',
|
|
37
|
+
image: 'example/img:tag',
|
|
38
|
+
at: 37,
|
|
39
|
+
cwd: '/repo',
|
|
40
|
+
out: 'out/proof',
|
|
41
|
+
backend,
|
|
42
|
+
})
|
|
43
|
+
```
|
|
44
|
+
|
|
45
|
+
`open()` must return a FRESH environment every call.
|
|
46
|
+
Each arm replays the prefix in its own session, so a reused session would prove the corrected step against debris from the previous arm.
|
|
47
|
+
|
|
48
|
+
Entry points that resolve images themselves take a `ReplayExecBackendFactory` instead: `(image) => ReplayExecBackend`.
|
|
49
|
+
Those are `runReplayBatch`, `replayVerifyFinding`, and `verifyFindings`.
|
|
50
|
+
|
|
51
|
+
## Verdict shape
|
|
52
|
+
|
|
53
|
+
`replayVerify` runs up to two arms and writes `replay-verdict.json` plus `report.md` into `out`.
|
|
54
|
+
|
|
55
|
+
| Field | Meaning |
|
|
56
|
+
|---|---|
|
|
57
|
+
| `armA.failureSignatureMatch` | The recorded returncode came back, and the recorded error substring appeared. |
|
|
58
|
+
| `armB.failureVanished` | A corrected command exited 0 and the error substring was gone. |
|
|
59
|
+
| `prefixDivergences` | Prefix steps that did not confirm the recording, each with its `kind`. |
|
|
60
|
+
| `prefixDivergencePct` | Divergent steps over executed steps. This is the number an admission pre-pass gates on. |
|
|
61
|
+
| `prefixWithinTolerance` | `prefixDivergencePct` is at most `PREFIX_DIVERGENCE_TOLERANCE_PCT` (10). |
|
|
62
|
+
| `signatureBasis` | `returncode+output-substring`, or `returncode-only` when the recording carries no error line. |
|
|
63
|
+
|
|
64
|
+
## Prefix fidelity
|
|
65
|
+
|
|
66
|
+
A verdict is only as good as the state the prefix rebuilt.
|
|
67
|
+
Agreement therefore requires positive evidence: a prefix step counts as confirmed only when the recording carries a returncode AND the replayed exit equals it.
|
|
68
|
+
|
|
69
|
+
Every other step is a divergence of one of two kinds.
|
|
70
|
+
|
|
71
|
+
| Kind | Meaning |
|
|
72
|
+
|---|---|
|
|
73
|
+
| `returncode-mismatch` | The recording carries a returncode and the replayed exit differs. |
|
|
74
|
+
| `unknown-expectation` | The recording carries no returncode, so the replay was never checked. |
|
|
75
|
+
|
|
76
|
+
An `unknown-expectation` step is never agreement.
|
|
77
|
+
Counting it as agreement lets a replay that fails on every step report a perfect prefix, which makes every verdict built on that prefix meaningless.
|
|
78
|
+
|
|
79
|
+
`runReplayBatch` reports the same split per case and across the corpus, under `headline.prefixFidelity`.
|
|
80
|
+
Its `replayed` predicate requires `prefixWithinTolerance`, so an unconfirmed prefix is never admitted.
|
|
81
|
+
`verifyFindings` applies the same rule: a proof whose prefix fell outside the tolerance is `divergent`, whatever its arms did.
|
|
82
|
+
|
|
83
|
+
Prefix divergence is reported, never hidden.
|
|
84
|
+
A high divergence rate is a finding about replay fidelity, not a harness error.
|
|
85
|
+
|
|
86
|
+
## Layers
|
|
87
|
+
|
|
88
|
+
| Module | Role |
|
|
89
|
+
|---|---|
|
|
90
|
+
| `steps` | The recorded step and its `<returncode>` / `<output>` grammar. |
|
|
91
|
+
| `exec` | The execution boundary and mini-SWE `/bin/sh` command wrapping. |
|
|
92
|
+
| `verify` | One case: prefix replay, arm A, optional arm B, `ReplayVerdict`. |
|
|
93
|
+
| `corpus` | Labeled corpora to replayable cases, with a reason for every exclusion. |
|
|
94
|
+
| `image-preparer` | Derive a replay-ready image from a recorded one. |
|
|
95
|
+
| `fix` / `fix-loop` | Generate and iterate arm-B corrections through an injected chat caller. |
|
|
96
|
+
| `batch` | Every replayable case to replayability and fix-flip rates. |
|
|
97
|
+
| `wire` / `findings` | One analyst finding to an executed, receipted proof. |
|
|
98
|
+
|
|
99
|
+
## Honest limits
|
|
100
|
+
|
|
101
|
+
Only trajectories that record their image are replayable.
|
|
102
|
+
A task whose environment needs external compose peers cannot be replayed this way.
|
|
103
|
+
|
|
104
|
+
A gold label on the submit step is never a replay target.
|
|
105
|
+
A submit decision has no executable failure to reproduce, so those cases are excluded and counted.
|
|
106
|
+
|
|
107
|
+
`dockerImagePreparer` shells out to `docker`.
|
|
108
|
+
Pass `preparer: null` on a corpus source when the images are already replay-ready.
|
|
109
|
+
|
|
110
|
+
Verified batch verdicts become RL rows through `src/rl/verified-findings-dataset.ts`; see [verified-labels-flywheel](./verified-labels-flywheel.md).
|
|
@@ -0,0 +1,103 @@
|
|
|
1
|
+
# Verification strategies: certifying without an answer key
|
|
2
|
+
|
|
3
|
+
This document covers the verification-strategy family (`src/verification-strategy.ts`), verdict epistemics (`src/verdict.ts`), and the blind statement-equivalence protocol (`src/equivalence-check.ts`).
|
|
4
|
+
It is the Wave 5 surface of the charter (`docs/charter.md`): an unsolved problem has no held-out test suite by definition, so the held-out suite must be one member of a strategy family, not the family itself.
|
|
5
|
+
|
|
6
|
+
The package ships the taxonomy, the record types, and the refusals.
|
|
7
|
+
It ships no checker implementation.
|
|
8
|
+
A checker is an injected executable boundary that returns a typed outcome, so a consumer binds its own Lean toolchain, invariant harness, replication runner, or judge.
|
|
9
|
+
Instruments, never methods.
|
|
10
|
+
|
|
11
|
+
## The family
|
|
12
|
+
|
|
13
|
+
Every member has a documented failure mode: the specific way it can certify a wrong result.
|
|
14
|
+
A consumer that reads a certification must weigh the member's failure mode, not treat "certified" as one bit.
|
|
15
|
+
The registry `VERIFICATION_STRATEGIES` carries each profile at runtime, so a reader can surface the failure mode without this document at hand.
|
|
16
|
+
|
|
17
|
+
| Member | Determinism | Certifies | Failure mode |
|
|
18
|
+
| --- | --- | --- | --- |
|
|
19
|
+
| `compile` | deterministic | typecheck / build / lint passed | code that compiles is not code that is correct |
|
|
20
|
+
| `test` | deterministic | suite pass-rate against an answer key | certifies nothing outside suite coverage; a stubbed integration reports green |
|
|
21
|
+
| `schema` | deterministic | structured output validates | shape is not meaning; a well-formed wrong answer passes |
|
|
22
|
+
| `sandbox` | deterministic | sandbox execution exit code | one bit compresses the run; a faked success exits 0 |
|
|
23
|
+
| `judge` | probabilistic | LLM judge score | drifts across model versions; Goodhart-gameable by the graded policy |
|
|
24
|
+
| `composite` | inherited | weighted blend of members | scalar collapse hides which member carried the score |
|
|
25
|
+
| `proof-kernel` | deterministic | a proof assistant's kernel accepted a formal proof | the formalization gap: the kernel never certifies that the formal statement matches the informal claim |
|
|
26
|
+
| `invariant` | deterministic | invariant / metamorphic properties held | weak invariants pass everything; a set uncalibrated by seeded bugs is a rubber stamp |
|
|
27
|
+
| `replication` | deterministic (given pins) | independent re-execution reproduced the result | re-runs the method, so it never catches an error the method itself carries |
|
|
28
|
+
| `agreement` | probabilistic | independently-derived results agree | the shared blind spot: derivers with common corpora or priors agree for the same wrong reason |
|
|
29
|
+
|
|
30
|
+
Three calibration rules follow from the failure modes.
|
|
31
|
+
|
|
32
|
+
1. A `proof-kernel` certificate is incomplete until the statement-equivalence obligation below is discharged; the kernel proves theorems about statements, never about intentions.
|
|
33
|
+
2. An `invariant` set earns weight only through seeded-bug calibration: demonstrate a mutation the set provably rejects before its pass carries any.
|
|
34
|
+
3. An `agreement` certificate is only as strong as its blindness provenance; the two-arm protocol below is the shape that makes the provenance checkable.
|
|
35
|
+
|
|
36
|
+
## The port shape
|
|
37
|
+
|
|
38
|
+
Execution binds through one port, `StrategyChecker<Input, Result>`:
|
|
39
|
+
|
|
40
|
+
- `strategy` — which family member the checker discharges obligations for;
|
|
41
|
+
- `identity` — exact checker identity (`CheckerIdentity`): name, version, and content pins, e.g. `{ name: 'lean4', version: '4.33.0', pins: { mathlib: 'db584cd6d46c' } }`;
|
|
42
|
+
- `determinism` — the class the checker claims for its outcomes;
|
|
43
|
+
- `check(input)` — returns `CheckerOutcome<Result>`: `{ succeeded: true, value }` or `{ succeeded: false, error }` with the full error text.
|
|
44
|
+
|
|
45
|
+
A checker that ran but could not decide returns `succeeded: false` with the reason.
|
|
46
|
+
`succeeded: true` is reserved for a discharged check.
|
|
47
|
+
|
|
48
|
+
## Verdict epistemics
|
|
49
|
+
|
|
50
|
+
`DefaultVerdict` carries an optional `certification: VerdictCertification`:
|
|
51
|
+
|
|
52
|
+
- `strategy` — the family member that certified the verdict;
|
|
53
|
+
- `checker` — the exact `CheckerIdentity`;
|
|
54
|
+
- `assumptions` — every step the certificate rests on that the checker did not verify; an empty array is an explicit, auditable claim of none;
|
|
55
|
+
- `evidenceDigest` — digest of the evidence artifact.
|
|
56
|
+
|
|
57
|
+
A kernel-checked verdict and a judge-scored verdict can carry the same `{ valid: true, score: 1 }`.
|
|
58
|
+
A consumer that cares reads `certification.strategy` and tells them apart.
|
|
59
|
+
A consumer that does not read the field sees the exact verdict it always saw; the field is additive and every pre-existing verdict shape remains valid.
|
|
60
|
+
|
|
61
|
+
The reward-source union `VerifiableRewardSource` (`src/rl/verifiable-reward.ts`) is the same family, so an RL consumer and a certification reader mean the same thing by `proof-kernel`.
|
|
62
|
+
|
|
63
|
+
## The statement-equivalence protocol
|
|
64
|
+
|
|
65
|
+
The proof-kernel failure mode gets its own discharge protocol, `defineEquivalenceCheck` (`src/equivalence-check.ts`): the blind two-arm design as a typed primitive.
|
|
66
|
+
|
|
67
|
+
Two arms derive the formal statement independently, blind to each other and to the outcome.
|
|
68
|
+
A bound checker then discharges the obligation that the two statements are equivalent:
|
|
69
|
+
|
|
70
|
+
- `proved` — the statements match; the formalization gap is closed for this claim;
|
|
71
|
+
- `refuted-with-separating-witness` — the statements provably differ, with the witness in hand; **a mismatch is a successful outcome** — it is the formalization gap made visible;
|
|
72
|
+
- `unresolved` — the obligation was not discharged; the record keeps the full reason.
|
|
73
|
+
|
|
74
|
+
The refusals are the design, and each one throws `EquivalenceProtocolError` with a machine-readable `code`:
|
|
75
|
+
|
|
76
|
+
- an arm that saw the other arm's statement (`arm-saw-other`) or the outcome (`arm-saw-outcome`) invalidates the check — nothing was independently derived;
|
|
77
|
+
- a design with any arm count other than 2 (`arm-count`) or with `blind: false` (`not-blind`) is a different protocol, not a parameter choice;
|
|
78
|
+
- a refutation without its separating witness (`witness-missing`), a proof carrying one (`witness-on-proved`), an unresolved obligation without its reason (`reason-missing`), and a verdict without its evidence digest (`evidence-missing`) are all refused;
|
|
79
|
+
- a checker whose declared strategy differs from the spec's (`checker-strategy-mismatch`) is refused before it runs.
|
|
80
|
+
|
|
81
|
+
Arm refusals fire before the checker executes: an invalid check must not spend.
|
|
82
|
+
|
|
83
|
+
## Worked example: the BCWW (4.6) pilot
|
|
84
|
+
|
|
85
|
+
The pilot this shape reproduces was assembled by hand for the refutation of the BCWW (4.6) inequality, before the substrate carried these types.
|
|
86
|
+
Artifacts: `~/bench-cache/bcww-formalization/` (`REPORT.md` is the check-phase report; `PRIORITY.md` records the search-coverage limits).
|
|
87
|
+
|
|
88
|
+
- Arm A (`armA/Statement.lean`) derived the Lean statement from the paper's LaTeX source (arXiv:1507.05650, display `eqn:cd2`).
|
|
89
|
+
- Arm B (`armB/Statement.lean`) derived it from the campaign's artifacts: discovery-lab KB pages plus the standalone verifier source.
|
|
90
|
+
- Both arms were blind to each other and committed before the counterexample was read.
|
|
91
|
+
- The checker was the Lean kernel: toolchain `leanprover/lean4:v4.33.0`, Mathlib pin `db584cd6d46c` (`check/Check.lean`, no `sorry`).
|
|
92
|
+
- Obligation: **proved** — `statement_equivalence` gives the iff at the level of the committed propositions, and the stronger `lhs46_entVec` shows the paper's linear form and the campaign's functional are equal as functions.
|
|
93
|
+
- The five-atom counterexample was then kernel-checked against both statements and violates both, with the exact value `(9*log2(3) - 14)/5` bits.
|
|
94
|
+
- The certification's assumption list is not empty: the diagonal embedding (classical distribution → quantum state) is argued in prose, not formalized.
|
|
95
|
+
That line is what an honest `VerdictCertification.assumptions` exists to carry.
|
|
96
|
+
|
|
97
|
+
In this package's terms: `defineEquivalenceCheck({ source: 'proof-kernel', artifact: 'arXiv:1507.05650 inequality (4.6) + five-atom counterexample record', arms: 2, blind: true })`, two `EquivalenceArm` records with `blindness: { toOtherArms: true, toOutcome: true }`, and an `EquivalenceObligation` with `status: 'proved'`, the lean4 `CheckerIdentity`, and the evidence digest of the kernel run.
|
|
98
|
+
|
|
99
|
+
## What this package refuses to own
|
|
100
|
+
|
|
101
|
+
- No Lean (or any checker) implementation ships here; the port is the boundary and the consumer binds its own kernel.
|
|
102
|
+
- No strategy selection: choosing proof-kernel over agreement for a task is a method decision, and methods live above the substrate (discovery's covenant).
|
|
103
|
+
- No certification laundering: a record missing its checker identity, evidence digest, blindness provenance, witness, or reason does not get a weaker record — it gets a thrown `EquivalenceProtocolError`.
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@tangle-network/agent-eval",
|
|
3
|
-
"version": "0.144.
|
|
3
|
+
"version": "0.144.7",
|
|
4
4
|
"description": "Evaluate and improve AI agents from runs, traces, judges, and feedback. Compare candidates, cluster failures, measure lift, and gate releases.",
|
|
5
5
|
"homepage": "https://github.com/tangle-network/agent-eval#readme",
|
|
6
6
|
"repository": {
|
|
@@ -134,6 +134,21 @@
|
|
|
134
134
|
"import": "./dist/hosted/index.js",
|
|
135
135
|
"default": "./dist/hosted/index.js"
|
|
136
136
|
},
|
|
137
|
+
"./trace-repair": {
|
|
138
|
+
"types": "./dist/trace-repair/index.d.ts",
|
|
139
|
+
"import": "./dist/trace-repair/index.js",
|
|
140
|
+
"default": "./dist/trace-repair/index.js"
|
|
141
|
+
},
|
|
142
|
+
"./experiment": {
|
|
143
|
+
"types": "./dist/experiment/index.d.ts",
|
|
144
|
+
"import": "./dist/experiment/index.js",
|
|
145
|
+
"default": "./dist/experiment/index.js"
|
|
146
|
+
},
|
|
147
|
+
"./trajectory-replay": {
|
|
148
|
+
"types": "./dist/trajectory-replay/index.d.ts",
|
|
149
|
+
"import": "./dist/trajectory-replay/index.js",
|
|
150
|
+
"default": "./dist/trajectory-replay/index.js"
|
|
151
|
+
},
|
|
137
152
|
"./openapi.json": {
|
|
138
153
|
"default": "./dist/openapi.json"
|
|
139
154
|
}
|
|
@@ -158,12 +173,14 @@
|
|
|
158
173
|
"test:watch": "vitest",
|
|
159
174
|
"typecheck": "tsc --noEmit",
|
|
160
175
|
"typecheck:examples": "tsc -p tsconfig.examples.json",
|
|
176
|
+
"typecheck:scripts": "tsc -p tsconfig.script.json",
|
|
161
177
|
"lint": "biome check src",
|
|
162
178
|
"format": "biome format --write src",
|
|
163
179
|
"check:skill": "node scripts/check-skill.mjs",
|
|
180
|
+
"check:model-ids": "node scripts/check-model-id-requests.mjs",
|
|
164
181
|
"check:analyst-benchmark": "node scripts/check-analyst-benchmark-implementation.mjs",
|
|
165
182
|
"openapi": "node dist/cli.js openapi --out dist/openapi.json",
|
|
166
|
-
"verify:package": "pnpm check:analyst-benchmark && pnpm run check:skill && publint && attw --pack --profile esm-only . && node scripts/verify-package-exports.mjs"
|
|
183
|
+
"verify:package": "pnpm check:analyst-benchmark && pnpm run check:skill && pnpm run check:model-ids && publint && attw --pack --profile esm-only . && node scripts/verify-package-exports.mjs"
|
|
167
184
|
},
|
|
168
185
|
"dependencies": {
|
|
169
186
|
"@asteasolutions/zod-to-openapi": "^9.1.0",
|