@tangle-network/agent-eval 0.179.0 → 0.181.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +66 -0
- package/README.md +119 -146
- package/dist/adapters/http.d.ts +2 -2
- package/dist/{agent-profile-B7yErX0q.d.ts → agent-profile-CivaSsSy.d.ts} +4 -4
- package/dist/{agent-profile-B7yErX0q.d.ts.map → agent-profile-CivaSsSy.d.ts.map} +1 -1
- package/dist/{agent-profile-cell-0gSi5ffD.js → agent-profile-cell-Cv6UA-W_.js} +20 -57
- package/dist/agent-profile-cell-Cv6UA-W_.js.map +1 -0
- package/dist/{agent-profile-cell-CTOZJUuE.d.ts → agent-profile-cell-s__adRnK.d.ts} +3 -3
- package/dist/agent-profile-cell-s__adRnK.d.ts.map +1 -0
- package/dist/analyst/index.d.ts +10 -10
- package/dist/analyst/index.js +4 -4
- package/dist/ast-CP9ae9B0.js +557 -0
- package/dist/ast-CP9ae9B0.js.map +1 -0
- package/dist/ast-hI-vjW6J.d.ts +457 -0
- package/dist/ast-hI-vjW6J.d.ts.map +1 -0
- package/dist/{benchmark-command-CY6Dg5t5.js → benchmark-command-B57n9vjz.js} +7 -6
- package/dist/{benchmark-command-CY6Dg5t5.js.map → benchmark-command-B57n9vjz.js.map} +1 -1
- package/dist/benchmarks/index.d.ts +4 -4
- package/dist/benchmarks/index.js +3 -3
- package/dist/campaign/index.d.ts +6 -6
- package/dist/campaign/index.js +8 -8
- package/dist/{campaign-BGEurASO.js → campaign-4_ppJW5X.js} +12 -12
- package/dist/{campaign-BGEurASO.js.map → campaign-4_ppJW5X.js.map} +1 -1
- package/dist/{campaign-evidence-D8DBLqLI.js → campaign-evidence-B8oF9xQ6.js} +515 -471
- package/dist/campaign-evidence-B8oF9xQ6.js.map +1 -0
- package/dist/cli.js +5 -8
- package/dist/cli.js.map +1 -1
- package/dist/{client-BlLY6o2w.js → client-CXE-U1SA.js} +3 -1
- package/dist/client-CXE-U1SA.js.map +1 -0
- package/dist/{client-CuQgX33c.d.ts → client-kh2jOjTK.d.ts} +4 -4
- package/dist/{client-CuQgX33c.d.ts.map → client-kh2jOjTK.d.ts.map} +1 -1
- package/dist/contract/index.d.ts +13 -13
- package/dist/contract/index.js +11 -10
- package/dist/contract/index.js.map +1 -1
- package/dist/{default-registry-IGDE9XIC.d.ts → default-registry-BwDSWVzg.d.ts} +6 -6
- package/dist/{default-registry-IGDE9XIC.d.ts.map → default-registry-BwDSWVzg.d.ts.map} +1 -1
- package/dist/{default-registry-BryMEmr8.js → default-registry-aL7xUrUz.js} +2 -2
- package/dist/{default-registry-BryMEmr8.js.map → default-registry-aL7xUrUz.js.map} +1 -1
- package/dist/{define-agent-eval-Cx4Ls9ta.d.ts → define-agent-eval-CwOWWQt_.d.ts} +33 -12
- package/dist/define-agent-eval-CwOWWQt_.d.ts.map +1 -0
- package/dist/{define-agent-eval-Dzidv34q.js → define-agent-eval-Ddu33JH9.js} +134 -67
- package/dist/define-agent-eval-Ddu33JH9.js.map +1 -0
- package/dist/{dspy-rlm-engine-xKiWmj_G.js → dspy-rlm-engine-S53V0HhE.js} +2 -2
- package/dist/{dspy-rlm-engine-xKiWmj_G.js.map → dspy-rlm-engine-S53V0HhE.js.map} +1 -1
- package/dist/{engine-CX8ReXkn.d.ts → engine-DS1cysJy.d.ts} +10 -7
- package/dist/engine-DS1cysJy.d.ts.map +1 -0
- package/dist/{eval-campaign-Cs-7MiCs.js → eval-campaign-aYdtjtJR.js} +4 -4
- package/dist/{eval-campaign-Cs-7MiCs.js.map → eval-campaign-aYdtjtJR.js.map} +1 -1
- package/dist/{exact-types-B7LC1EyX.d.ts → exact-types-BZDe0W2D.d.ts} +2 -2
- package/dist/{exact-types-B7LC1EyX.d.ts.map → exact-types-BZDe0W2D.d.ts.map} +1 -1
- package/dist/experiment/index.d.ts +27 -477
- package/dist/experiment/index.d.ts.map +1 -1
- package/dist/experiment/index.js +95 -559
- package/dist/experiment/index.js.map +1 -1
- package/dist/{experiment-tracker-B3TiF5-u.d.ts → experiment-tracker-C7PfnF4b.d.ts} +2 -2
- package/dist/{experiment-tracker-B3TiF5-u.d.ts.map → experiment-tracker-C7PfnF4b.d.ts.map} +1 -1
- package/dist/{external-optimizer-process-Dlz8YxrT.js → external-optimizer-process-QDRURJAM.js} +3 -3
- package/dist/{external-optimizer-process-Dlz8YxrT.js.map → external-optimizer-process-QDRURJAM.js.map} +1 -1
- package/dist/{external-optimizer-subprocess-q3VzlGAO.js → external-optimizer-subprocess-D4dzUBZI.js} +3 -2
- package/dist/{external-optimizer-subprocess-q3VzlGAO.js.map → external-optimizer-subprocess-D4dzUBZI.js.map} +1 -1
- package/dist/{feedback-trajectory-eHWNv5Aj.d.ts → feedback-trajectory-CXmtITBo.d.ts} +3 -3
- package/dist/{feedback-trajectory-eHWNv5Aj.d.ts.map → feedback-trajectory-CXmtITBo.d.ts.map} +1 -1
- package/dist/hosted/index.d.ts +2 -2
- package/dist/hosted/index.d.ts.map +1 -1
- package/dist/hosted/index.js +1 -1
- package/dist/{index-BxWvILU8.d.ts → index-Bp_6sj3x.d.ts} +109 -56
- package/dist/index-Bp_6sj3x.d.ts.map +1 -0
- package/dist/{index-e7LXeRVa.d.ts → index-CJ3LhKIX.d.ts} +2 -2
- package/dist/{index-e7LXeRVa.d.ts.map → index-CJ3LhKIX.d.ts.map} +1 -1
- package/dist/{index-CbLmrWCa.d.ts → index-DNntP4ch.d.ts} +8 -8
- package/dist/{index-CbLmrWCa.d.ts.map → index-DNntP4ch.d.ts.map} +1 -1
- package/dist/{index-DxNYmx4a.d.ts → index-DoykkxW0.d.ts} +11 -11
- package/dist/{index-DxNYmx4a.d.ts.map → index-DoykkxW0.d.ts.map} +1 -1
- package/dist/index.d.ts +28 -28
- package/dist/index.js +25 -16
- package/dist/index.js.map +1 -1
- package/dist/{insight-report-DETqPc_A.d.ts → insight-report-D1qa0HWs.d.ts} +9 -5
- package/dist/{insight-report-DETqPc_A.d.ts.map → insight-report-D1qa0HWs.d.ts.map} +1 -1
- package/dist/{integrity-DsHWCebQ.js → integrity-DH5ng72x.js} +2 -2
- package/dist/{integrity-DsHWCebQ.js.map → integrity-DH5ng72x.js.map} +1 -1
- package/dist/{integrity-BKTcA-HP.d.ts → integrity-rGOfSUle.d.ts} +2 -2
- package/dist/{integrity-BKTcA-HP.d.ts.map → integrity-rGOfSUle.d.ts.map} +1 -1
- package/dist/{ledger-core-Cs9f7385.js → journal-Cs9f7385.js} +1 -1
- package/dist/journal-Cs9f7385.js.map +1 -0
- package/dist/{judge-calibration-C5CbMYce.d.ts → judge-calibration-DFtEMlde.d.ts} +31 -2
- package/dist/judge-calibration-DFtEMlde.d.ts.map +1 -0
- package/dist/{judge-calibration-BnpVKtnb.js → judge-calibration-DYmaBtJr.js} +48 -2
- package/dist/{judge-calibration-BnpVKtnb.js.map → judge-calibration-DYmaBtJr.js.map} +1 -1
- package/dist/ledger-core/index.d.ts +1 -1
- package/dist/ledger-core/index.js +1 -1
- package/dist/{llm-judge-v80Kmu9g.js → llm-judge-DEFZeSiu.js} +645 -456
- package/dist/llm-judge-DEFZeSiu.js.map +1 -0
- package/dist/{matrix-DeMmnWrP.d.ts → matrix-CyhW-vgJ.d.ts} +2 -2
- package/dist/{matrix-DeMmnWrP.d.ts.map → matrix-CyhW-vgJ.d.ts.map} +1 -1
- package/dist/meta-eval/index.d.ts +138 -7
- package/dist/meta-eval/index.d.ts.map +1 -1
- package/dist/meta-eval/index.js +245 -97
- package/dist/meta-eval/index.js.map +1 -1
- package/dist/{mint-Cc1_zwRQ.js → mint-ySIIkKlV.js} +2 -2
- package/dist/{mint-Cc1_zwRQ.js.map → mint-ySIIkKlV.js.map} +1 -1
- package/dist/multishot/golden/index.d.ts +1 -1
- package/dist/multishot/index.d.ts +2 -2
- package/dist/openapi.json +1 -1
- package/dist/outcome-store-BXlkwMPR.js +131 -0
- package/dist/outcome-store-BXlkwMPR.js.map +1 -0
- package/dist/{outcome-store-BYHIuO0e.d.ts → outcome-store-CNt4iZ67.d.ts} +18 -25
- package/dist/outcome-store-CNt4iZ67.d.ts.map +1 -0
- package/dist/{paired-promotion-decision-CGzg0cI_.d.ts → paired-promotion-decision-DPsMQm-0.d.ts} +13 -7
- package/dist/{paired-promotion-decision-CGzg0cI_.d.ts.map → paired-promotion-decision-DPsMQm-0.d.ts.map} +1 -1
- package/dist/pipelines/index.js +1 -1
- package/dist/{produced-state-Cv0kJJuP.js → produced-state-BHboMaab.js} +3 -3
- package/dist/{produced-state-Cv0kJJuP.js.map → produced-state-BHboMaab.js.map} +1 -1
- package/dist/profile-cell.d.ts +1 -1
- package/dist/profile-cell.js +1 -1
- package/dist/{promotion-policy-DWOm70gx.js → promotion-policy-CDMMxzb6.js} +28 -40
- package/dist/promotion-policy-CDMMxzb6.js.map +1 -0
- package/dist/{registry-ByVld1-5.d.ts → registry-BRbB6Y0v.d.ts} +4 -4
- package/dist/{registry-ByVld1-5.d.ts.map → registry-BRbB6Y0v.d.ts.map} +1 -1
- package/dist/{release-confidence-BAcNYOf1.d.ts → release-confidence-BcqeQTHW.d.ts} +3 -3
- package/dist/{release-confidence-BAcNYOf1.d.ts.map → release-confidence-BcqeQTHW.d.ts.map} +1 -1
- package/dist/{release-confidence-BcGCclTB.js → release-confidence-DMg8n18l.js} +2 -2
- package/dist/{release-confidence-BcGCclTB.js.map → release-confidence-DMg8n18l.js.map} +1 -1
- package/dist/{report-command-DKlXfU5r.js → report-command-V1ecVgAv.js} +27 -3
- package/dist/report-command-V1ecVgAv.js.map +1 -0
- package/dist/reporting.d.ts +4 -4
- package/dist/reporting.js +3 -3
- package/dist/{researcher-jsW1X94L.d.ts → researcher-64T49THL.d.ts} +6 -6
- package/dist/{researcher-jsW1X94L.d.ts.map → researcher-64T49THL.d.ts.map} +1 -1
- package/dist/{reward-hacking-ZXEi9VCq.d.ts → reward-hacking-uzO_ihep.d.ts} +2 -2
- package/dist/{reward-hacking-ZXEi9VCq.d.ts.map → reward-hacking-uzO_ihep.d.ts.map} +1 -1
- package/dist/rl.d.ts +53 -99
- package/dist/rl.d.ts.map +1 -1
- package/dist/rl.js +182 -169
- package/dist/rl.js.map +1 -1
- package/dist/rollout/index.d.ts +1 -1
- package/dist/rollout/index.js +2 -2
- package/dist/{rollout-DmoJVqrF.js → rollout-B-UF5R6w.js} +2 -2
- package/dist/{rollout-DmoJVqrF.js.map → rollout-B-UF5R6w.js.map} +1 -1
- package/dist/rubric-predictive-validity-Bmj2_cll.d.ts +79 -0
- package/dist/rubric-predictive-validity-Bmj2_cll.d.ts.map +1 -0
- package/dist/rubric-predictive-validity-CCK-1B7w.js +178 -0
- package/dist/rubric-predictive-validity-CCK-1B7w.js.map +1 -0
- package/dist/{run-record-DTv1MdjK.d.ts → run-record-BiTWauyO.d.ts} +2 -2
- package/dist/{run-record-DTv1MdjK.d.ts.map → run-record-BiTWauyO.d.ts.map} +1 -1
- package/dist/run-record-Br-Yzt_k.js +464 -0
- package/dist/run-record-Br-Yzt_k.js.map +1 -0
- package/dist/{run-record-DQpSf7t-.js → run-record-DualPTn2.js} +2 -2
- package/dist/{run-record-DQpSf7t-.js.map → run-record-DualPTn2.js.map} +1 -1
- package/dist/{semantic-concept-judge-Bi6_iGqg.js → semantic-concept-judge-Bm5JDEKO.js} +3 -3
- package/dist/{semantic-concept-judge-Bi6_iGqg.js.map → semantic-concept-judge-Bm5JDEKO.js.map} +1 -1
- package/dist/{sequential-B5gXgcyp.js → sequential-DAsyV2T9.js} +42 -25
- package/dist/sequential-DAsyV2T9.js.map +1 -0
- package/dist/{series-convergence-DeG33RpC.d.ts → series-convergence-BnMs_uAr.d.ts} +3 -3
- package/dist/{series-convergence-DeG33RpC.d.ts.map → series-convergence-BnMs_uAr.d.ts.map} +1 -1
- package/dist/{skillopt-optimization-method-C3oYul8v.js → skillopt-optimization-method-CL_0aArC.js} +5 -5
- package/dist/{skillopt-optimization-method-C3oYul8v.js.map → skillopt-optimization-method-CL_0aArC.js.map} +1 -1
- package/dist/{statistical-heldout-0La5ZTlv.d.ts → statistical-heldout-CpVd6FmY.d.ts} +207 -144
- package/dist/statistical-heldout-CpVd6FmY.d.ts.map +1 -0
- package/dist/{store-tool-spans-4J1EDElP.d.ts → store-tool-spans-Dt-YdAuE.d.ts} +6 -6
- package/dist/{store-tool-spans-4J1EDElP.d.ts.map → store-tool-spans-Dt-YdAuE.d.ts.map} +1 -1
- package/dist/{summary-report-gMrbYawB.d.ts → summary-report-D1h4dlrK.d.ts} +3 -3
- package/dist/{summary-report-gMrbYawB.d.ts.map → summary-report-D1h4dlrK.d.ts.map} +1 -1
- package/dist/{summary-report-B16xy9Kd.js → summary-report-e-MaOAHV.js} +2 -2
- package/dist/{summary-report-B16xy9Kd.js.map → summary-report-e-MaOAHV.js.map} +1 -1
- package/dist/supervisor-run/index.d.ts +4 -2
- package/dist/supervisor-run/index.d.ts.map +1 -1
- package/dist/supervisor-run/index.js +3 -3
- package/dist/{terminal-record-Ce9_UjRz.js → terminal-record-BtPwKTSr.js} +58 -26
- package/dist/terminal-record-BtPwKTSr.js.map +1 -0
- package/dist/{tool-groups-2QA0S7dK.d.ts → tool-groups-B2bSNaJB.d.ts} +3 -3
- package/dist/tool-groups-B2bSNaJB.d.ts.map +1 -0
- package/dist/{tool-waste-B9tdWV6g.js → tool-waste-C7MU9u1e.js} +2 -2
- package/dist/{tool-waste-B9tdWV6g.js.map → tool-waste-C7MU9u1e.js.map} +1 -1
- package/dist/trace-repair/index.d.ts +2 -2
- package/dist/traces.d.ts +6 -6
- package/dist/traces.js +1 -1
- package/dist/{types-BmlkCrg0.d.ts → types-BvZoPTGa.d.ts} +3 -3
- package/dist/{types-BmlkCrg0.d.ts.map → types-BvZoPTGa.d.ts.map} +1 -1
- package/dist/{types-gvRsyJLh.d.ts → types-CBbLtr2J.d.ts} +38 -3
- package/dist/{types-gvRsyJLh.d.ts.map → types-CBbLtr2J.d.ts.map} +1 -1
- package/dist/{types-C34V4Vto.d.ts → types-CS0qk_Yp.d.ts} +4 -4
- package/dist/{types-C34V4Vto.d.ts.map → types-CS0qk_Yp.d.ts.map} +1 -1
- package/dist/{types-DzuaM493.d.ts → types-D7gEdPoQ.d.ts} +3 -3
- package/dist/{types-DzuaM493.d.ts.map → types-D7gEdPoQ.d.ts.map} +1 -1
- package/dist/{types-vUdAx2Cj.d.ts → types-lPkDQNqJ.d.ts} +20 -2
- package/dist/{types-vUdAx2Cj.d.ts.map → types-lPkDQNqJ.d.ts.map} +1 -1
- package/dist/wire/index.d.ts +2 -2
- package/docs/adapters-observability.md +14 -0
- package/docs/campaign-proposers.md +86 -128
- package/docs/charter.md +108 -112
- package/docs/concepts.md +157 -69
- package/docs/design/mlbenchmarks-book-review.md +440 -0
- package/docs/design/mlbenchmarks-review/observations.json +713 -0
- package/docs/design/mlbenchmarks-review/probes.mts +476 -0
- package/docs/design/mlbenchmarks-review/sources.json +200 -0
- package/docs/design/self-improvement-evidence-audit.md +263 -0
- package/docs/design.md +2 -1
- package/docs/eval-surface-map.md +95 -42
- package/docs/evaluation-integrity.md +220 -0
- package/docs/experiment.md +111 -55
- package/docs/feature-guide.md +5 -6
- package/docs/hosted-ingest-spec.md +4 -11
- package/docs/insight-report.md +187 -455
- package/docs/outcome-validity.md +182 -0
- package/docs/product-eval-adoption.md +1 -2
- package/docs/research-report-methodology.md +7 -7
- package/docs/search-history-receipts.md +8 -0
- package/docs/statistical-evidence.md +129 -0
- package/docs/verdicts.md +76 -49
- package/package.json +1 -1
- package/dist/agent-profile-cell-0gSi5ffD.js.map +0 -1
- package/dist/agent-profile-cell-CTOZJUuE.d.ts.map +0 -1
- package/dist/campaign-evidence-D8DBLqLI.js.map +0 -1
- package/dist/client-BlLY6o2w.js.map +0 -1
- package/dist/define-agent-eval-Cx4Ls9ta.d.ts.map +0 -1
- package/dist/define-agent-eval-Dzidv34q.js.map +0 -1
- package/dist/engine-CX8ReXkn.d.ts.map +0 -1
- package/dist/index-BxWvILU8.d.ts.map +0 -1
- package/dist/judge-calibration-C5CbMYce.d.ts.map +0 -1
- package/dist/ledger-core-Cs9f7385.js.map +0 -1
- package/dist/llm-judge-v80Kmu9g.js.map +0 -1
- package/dist/outcome-store-BYHIuO0e.d.ts.map +0 -1
- package/dist/outcome-store-ChBKlTd_.js +0 -75
- package/dist/outcome-store-ChBKlTd_.js.map +0 -1
- package/dist/promotion-policy-DWOm70gx.js.map +0 -1
- package/dist/report-command-DKlXfU5r.js.map +0 -1
- package/dist/rubric-predictive-validity-2D5Gw9z9.js +0 -131
- package/dist/rubric-predictive-validity-2D5Gw9z9.js.map +0 -1
- package/dist/rubric-predictive-validity-Dl1dvKCv.d.ts +0 -75
- package/dist/rubric-predictive-validity-Dl1dvKCv.d.ts.map +0 -1
- package/dist/run-record-CR63CpHK.js +0 -216
- package/dist/run-record-CR63CpHK.js.map +0 -1
- package/dist/sequential-B5gXgcyp.js.map +0 -1
- package/dist/statistical-heldout-0La5ZTlv.d.ts.map +0 -1
- package/dist/terminal-record-Ce9_UjRz.js.map +0 -1
- package/dist/tool-groups-2QA0S7dK.d.ts.map +0 -1
|
@@ -0,0 +1,713 @@
|
|
|
1
|
+
{
|
|
2
|
+
"reviewedBaseRevision": "fe1cc5111aab5d588bf7db3a3785325635937a91",
|
|
3
|
+
"sourceIdentity": {
|
|
4
|
+
"algorithm": "sha256-canonical-file-manifest",
|
|
5
|
+
"paths": [
|
|
6
|
+
"src",
|
|
7
|
+
"package.json",
|
|
8
|
+
"pnpm-lock.yaml",
|
|
9
|
+
"tsconfig.json"
|
|
10
|
+
],
|
|
11
|
+
"fileCount": 765,
|
|
12
|
+
"digest": "sha256:d743fe180fb7f89c2add26f3f66344a8586b2798f1edcaf8b881b914c10bc009",
|
|
13
|
+
"dependencyScope": "Records the manifest and lockfile; assumes dependencies were installed from that lockfile."
|
|
14
|
+
},
|
|
15
|
+
"diagnosticIdentity": {
|
|
16
|
+
"path": "docs/design/mlbenchmarks-review/probes.mts",
|
|
17
|
+
"sha256": "29d0bba079d8b76247bd3d92a24ea8b256080e22ee37db9ca28f0ac3330c793a"
|
|
18
|
+
},
|
|
19
|
+
"command": "pnpm exec tsx docs/design/mlbenchmarks-review/probes.mts",
|
|
20
|
+
"paidModelCalls": 0,
|
|
21
|
+
"execution": {
|
|
22
|
+
"kind": "offline deterministic diagnostic",
|
|
23
|
+
"modelCalls": 0,
|
|
24
|
+
"callbackImplementation": "Local arithmetic and string checks only; no provider clients are supplied.",
|
|
25
|
+
"temporaryRunStorage": "Allocated under the OS temporary directory and removed in finally.",
|
|
26
|
+
"outputPolicy": "Preserves current returned measurements; excludes temporary paths, run IDs, and wallclock fields.",
|
|
27
|
+
"assertionPolicy": "No assertions require the observed defects or policy boundaries to persist."
|
|
28
|
+
},
|
|
29
|
+
"probes": {
|
|
30
|
+
"holdoutReuse": {
|
|
31
|
+
"inputs": {
|
|
32
|
+
"calls": 2,
|
|
33
|
+
"trainCases": 6,
|
|
34
|
+
"finalCases": 6,
|
|
35
|
+
"generationsPerCall": 1,
|
|
36
|
+
"populationPerGeneration": 1,
|
|
37
|
+
"replicatesPerCase": 1,
|
|
38
|
+
"sameFinalPayloadsOnBothCalls": true,
|
|
39
|
+
"firstResultFedToSecondProposer": false,
|
|
40
|
+
"baselineSurface": "baseline",
|
|
41
|
+
"proposedSurface": "marker",
|
|
42
|
+
"scoring": "1 if artifact contains marker, otherwise 0",
|
|
43
|
+
"expectUsage": "off"
|
|
44
|
+
},
|
|
45
|
+
"results": {
|
|
46
|
+
"rounds": [
|
|
47
|
+
{
|
|
48
|
+
"round": 1,
|
|
49
|
+
"gateDecision": "ship",
|
|
50
|
+
"finalDispatches": 12,
|
|
51
|
+
"distinctFinalIds": 6,
|
|
52
|
+
"finalSplitDigest": "sha256:fe055aa9390b6483701c8686564c1d4d96426d6a545f59e405fa5350efae84fa",
|
|
53
|
+
"agentCallbacks": 24,
|
|
54
|
+
"judgeCallbacks": 24,
|
|
55
|
+
"proposerCallbacks": 1
|
|
56
|
+
},
|
|
57
|
+
{
|
|
58
|
+
"round": 2,
|
|
59
|
+
"gateDecision": "ship",
|
|
60
|
+
"finalDispatches": 12,
|
|
61
|
+
"distinctFinalIds": 6,
|
|
62
|
+
"finalSplitDigest": "sha256:fe055aa9390b6483701c8686564c1d4d96426d6a545f59e405fa5350efae84fa",
|
|
63
|
+
"agentCallbacks": 24,
|
|
64
|
+
"judgeCallbacks": 24,
|
|
65
|
+
"proposerCallbacks": 1
|
|
66
|
+
}
|
|
67
|
+
],
|
|
68
|
+
"accessPurposes": [
|
|
69
|
+
"debugging",
|
|
70
|
+
"evaluation"
|
|
71
|
+
],
|
|
72
|
+
"temporaryRunDirectoriesRemoved": true
|
|
73
|
+
},
|
|
74
|
+
"limitations": [
|
|
75
|
+
"Measures repeated final-set access and debugging access, not empirical overfitting.",
|
|
76
|
+
"The calls use deterministic local callbacks and independent temporary run directories.",
|
|
77
|
+
"Does not estimate false-promotion frequency or test a downstream access-control service."
|
|
78
|
+
]
|
|
79
|
+
},
|
|
80
|
+
"outcomeKeyOrder": {
|
|
81
|
+
"inputs": {
|
|
82
|
+
"n": 10,
|
|
83
|
+
"rows": [
|
|
84
|
+
{
|
|
85
|
+
"score": 0.1,
|
|
86
|
+
"retention": 0.9,
|
|
87
|
+
"csat": 0.1
|
|
88
|
+
},
|
|
89
|
+
{
|
|
90
|
+
"score": 0.2,
|
|
91
|
+
"retention": 0.8,
|
|
92
|
+
"csat": 0.2
|
|
93
|
+
},
|
|
94
|
+
{
|
|
95
|
+
"score": 0.3,
|
|
96
|
+
"retention": 0.7,
|
|
97
|
+
"csat": 0.3
|
|
98
|
+
},
|
|
99
|
+
{
|
|
100
|
+
"score": 0.4,
|
|
101
|
+
"retention": 0.6,
|
|
102
|
+
"csat": 0.4
|
|
103
|
+
},
|
|
104
|
+
{
|
|
105
|
+
"score": 0.5,
|
|
106
|
+
"retention": 0.5,
|
|
107
|
+
"csat": 0.5
|
|
108
|
+
},
|
|
109
|
+
{
|
|
110
|
+
"score": 0.6,
|
|
111
|
+
"retention": 0.4,
|
|
112
|
+
"csat": 0.6
|
|
113
|
+
},
|
|
114
|
+
{
|
|
115
|
+
"score": 0.7,
|
|
116
|
+
"retention": 0.30000000000000004,
|
|
117
|
+
"csat": 0.7
|
|
118
|
+
},
|
|
119
|
+
{
|
|
120
|
+
"score": 0.8,
|
|
121
|
+
"retention": 0.19999999999999996,
|
|
122
|
+
"csat": 0.8
|
|
123
|
+
},
|
|
124
|
+
{
|
|
125
|
+
"score": 0.9,
|
|
126
|
+
"retention": 0.09999999999999998,
|
|
127
|
+
"csat": 0.9
|
|
128
|
+
},
|
|
129
|
+
{
|
|
130
|
+
"score": 1,
|
|
131
|
+
"retention": 0,
|
|
132
|
+
"csat": 1
|
|
133
|
+
}
|
|
134
|
+
],
|
|
135
|
+
"outcomeRowsPerRun": 1,
|
|
136
|
+
"requestedMetric": "csat",
|
|
137
|
+
"expectedPearsonForRequestedMetric": 1,
|
|
138
|
+
"expectedSpearmanForRequestedMetric": 1,
|
|
139
|
+
"seed": 1,
|
|
140
|
+
"bootstrapIterations": 500
|
|
141
|
+
},
|
|
142
|
+
"results": [
|
|
143
|
+
{
|
|
144
|
+
"keyOrder": "retention-first",
|
|
145
|
+
"latest": {
|
|
146
|
+
"pairs": [
|
|
147
|
+
{
|
|
148
|
+
"evalMetric": "score",
|
|
149
|
+
"outcomeMetric": "csat",
|
|
150
|
+
"n": 10,
|
|
151
|
+
"pearson": -1,
|
|
152
|
+
"spearman": -1,
|
|
153
|
+
"pearsonCi95": {
|
|
154
|
+
"lower": -1.0000000000000002,
|
|
155
|
+
"upper": -0.9999999999999998
|
|
156
|
+
},
|
|
157
|
+
"verdict": "strong"
|
|
158
|
+
}
|
|
159
|
+
],
|
|
160
|
+
"joinedSamples": 10,
|
|
161
|
+
"skippedRuns": 0
|
|
162
|
+
},
|
|
163
|
+
"mean": {
|
|
164
|
+
"pairs": [
|
|
165
|
+
{
|
|
166
|
+
"evalMetric": "score",
|
|
167
|
+
"outcomeMetric": "csat",
|
|
168
|
+
"n": 10,
|
|
169
|
+
"pearson": 1,
|
|
170
|
+
"spearman": 1,
|
|
171
|
+
"pearsonCi95": {
|
|
172
|
+
"lower": 1,
|
|
173
|
+
"upper": 1
|
|
174
|
+
},
|
|
175
|
+
"verdict": "strong"
|
|
176
|
+
}
|
|
177
|
+
],
|
|
178
|
+
"joinedSamples": 10,
|
|
179
|
+
"skippedRuns": 0
|
|
180
|
+
}
|
|
181
|
+
},
|
|
182
|
+
{
|
|
183
|
+
"keyOrder": "csat-first",
|
|
184
|
+
"latest": {
|
|
185
|
+
"pairs": [
|
|
186
|
+
{
|
|
187
|
+
"evalMetric": "score",
|
|
188
|
+
"outcomeMetric": "csat",
|
|
189
|
+
"n": 10,
|
|
190
|
+
"pearson": 1,
|
|
191
|
+
"spearman": 1,
|
|
192
|
+
"pearsonCi95": {
|
|
193
|
+
"lower": 1,
|
|
194
|
+
"upper": 1
|
|
195
|
+
},
|
|
196
|
+
"verdict": "strong"
|
|
197
|
+
}
|
|
198
|
+
],
|
|
199
|
+
"joinedSamples": 10,
|
|
200
|
+
"skippedRuns": 0
|
|
201
|
+
},
|
|
202
|
+
"mean": {
|
|
203
|
+
"pairs": [
|
|
204
|
+
{
|
|
205
|
+
"evalMetric": "score",
|
|
206
|
+
"outcomeMetric": "csat",
|
|
207
|
+
"n": 10,
|
|
208
|
+
"pearson": 1,
|
|
209
|
+
"spearman": 1,
|
|
210
|
+
"pearsonCi95": {
|
|
211
|
+
"lower": 1,
|
|
212
|
+
"upper": 1
|
|
213
|
+
},
|
|
214
|
+
"verdict": "strong"
|
|
215
|
+
}
|
|
216
|
+
],
|
|
217
|
+
"joinedSamples": 10,
|
|
218
|
+
"skippedRuns": 0
|
|
219
|
+
}
|
|
220
|
+
}
|
|
221
|
+
],
|
|
222
|
+
"limitations": [
|
|
223
|
+
"Tests metric selection and JSON key order with constructed data, not deployment validity.",
|
|
224
|
+
"Each run has one outcome row, so latest and mean refer to the same requested observation."
|
|
225
|
+
]
|
|
226
|
+
},
|
|
227
|
+
"adaptationPairing": {
|
|
228
|
+
"inputs": {
|
|
229
|
+
"scenariosA": [
|
|
230
|
+
{
|
|
231
|
+
"scenarioId": "easy-only",
|
|
232
|
+
"score": 0.9
|
|
233
|
+
}
|
|
234
|
+
],
|
|
235
|
+
"scenariosB": [
|
|
236
|
+
{
|
|
237
|
+
"scenarioId": "hard-only",
|
|
238
|
+
"score": 0.1
|
|
239
|
+
}
|
|
240
|
+
],
|
|
241
|
+
"ks": [
|
|
242
|
+
0,
|
|
243
|
+
1
|
|
244
|
+
],
|
|
245
|
+
"reps": 1,
|
|
246
|
+
"observationsPerArm": 2,
|
|
247
|
+
"commonScenarios": 0,
|
|
248
|
+
"sameRunnerForBothArms": true,
|
|
249
|
+
"bootstrapSeed": 1
|
|
250
|
+
},
|
|
251
|
+
"results": {
|
|
252
|
+
"perK": [
|
|
253
|
+
{
|
|
254
|
+
"k": 0,
|
|
255
|
+
"deltaMean": 0.8,
|
|
256
|
+
"aLow": 0.9,
|
|
257
|
+
"aHigh": 0.9,
|
|
258
|
+
"bLow": 0.1,
|
|
259
|
+
"bHigh": 0.1
|
|
260
|
+
},
|
|
261
|
+
{
|
|
262
|
+
"k": 1,
|
|
263
|
+
"deltaMean": 0.8,
|
|
264
|
+
"aLow": 0.9,
|
|
265
|
+
"aHigh": 0.9,
|
|
266
|
+
"bLow": 0.1,
|
|
267
|
+
"bHigh": 0.1
|
|
268
|
+
}
|
|
269
|
+
],
|
|
270
|
+
"areaDelta": 0.8,
|
|
271
|
+
"firstPassKDelta": null,
|
|
272
|
+
"verdict": "a_better",
|
|
273
|
+
"rationale": "mean per-k delta=0.800, area delta=0.800"
|
|
274
|
+
},
|
|
275
|
+
"limitations": [
|
|
276
|
+
"The two arms differ in task difficulty; zero task identities overlap.",
|
|
277
|
+
"This probes the adaptation helper, not the separately implemented campaign paired comparison."
|
|
278
|
+
]
|
|
279
|
+
},
|
|
280
|
+
"contaminationDisplay": {
|
|
281
|
+
"inputs": {
|
|
282
|
+
"n": 12,
|
|
283
|
+
"originalScorePerCase": 1,
|
|
284
|
+
"perturbedScorePerCase": 0.4,
|
|
285
|
+
"observationsPerCasePerCondition": 1,
|
|
286
|
+
"modelTrainingExposure": "No model is used; scores are constructed fixture values."
|
|
287
|
+
},
|
|
288
|
+
"results": {
|
|
289
|
+
"perScenario": [
|
|
290
|
+
{
|
|
291
|
+
"scenarioId": "case-0",
|
|
292
|
+
"originalScore": 1,
|
|
293
|
+
"perturbedScore": 0.4,
|
|
294
|
+
"delta": -0.6,
|
|
295
|
+
"qValue": 0.4
|
|
296
|
+
},
|
|
297
|
+
{
|
|
298
|
+
"scenarioId": "case-1",
|
|
299
|
+
"originalScore": 1,
|
|
300
|
+
"perturbedScore": 0.4,
|
|
301
|
+
"delta": -0.6,
|
|
302
|
+
"qValue": 0.4
|
|
303
|
+
},
|
|
304
|
+
{
|
|
305
|
+
"scenarioId": "case-2",
|
|
306
|
+
"originalScore": 1,
|
|
307
|
+
"perturbedScore": 0.4,
|
|
308
|
+
"delta": -0.6,
|
|
309
|
+
"qValue": 0.4
|
|
310
|
+
},
|
|
311
|
+
{
|
|
312
|
+
"scenarioId": "case-3",
|
|
313
|
+
"originalScore": 1,
|
|
314
|
+
"perturbedScore": 0.4,
|
|
315
|
+
"delta": -0.6,
|
|
316
|
+
"qValue": 0.4
|
|
317
|
+
},
|
|
318
|
+
{
|
|
319
|
+
"scenarioId": "case-4",
|
|
320
|
+
"originalScore": 1,
|
|
321
|
+
"perturbedScore": 0.4,
|
|
322
|
+
"delta": -0.6,
|
|
323
|
+
"qValue": 0.4
|
|
324
|
+
},
|
|
325
|
+
{
|
|
326
|
+
"scenarioId": "case-5",
|
|
327
|
+
"originalScore": 1,
|
|
328
|
+
"perturbedScore": 0.4,
|
|
329
|
+
"delta": -0.6,
|
|
330
|
+
"qValue": 0.4
|
|
331
|
+
},
|
|
332
|
+
{
|
|
333
|
+
"scenarioId": "case-6",
|
|
334
|
+
"originalScore": 1,
|
|
335
|
+
"perturbedScore": 0.4,
|
|
336
|
+
"delta": -0.6,
|
|
337
|
+
"qValue": 0.4
|
|
338
|
+
},
|
|
339
|
+
{
|
|
340
|
+
"scenarioId": "case-7",
|
|
341
|
+
"originalScore": 1,
|
|
342
|
+
"perturbedScore": 0.4,
|
|
343
|
+
"delta": -0.6,
|
|
344
|
+
"qValue": 0.4
|
|
345
|
+
},
|
|
346
|
+
{
|
|
347
|
+
"scenarioId": "case-8",
|
|
348
|
+
"originalScore": 1,
|
|
349
|
+
"perturbedScore": 0.4,
|
|
350
|
+
"delta": -0.6,
|
|
351
|
+
"qValue": 0.4
|
|
352
|
+
},
|
|
353
|
+
{
|
|
354
|
+
"scenarioId": "case-9",
|
|
355
|
+
"originalScore": 1,
|
|
356
|
+
"perturbedScore": 0.4,
|
|
357
|
+
"delta": -0.6,
|
|
358
|
+
"qValue": 0.4
|
|
359
|
+
},
|
|
360
|
+
{
|
|
361
|
+
"scenarioId": "case-10",
|
|
362
|
+
"originalScore": 1,
|
|
363
|
+
"perturbedScore": 0.4,
|
|
364
|
+
"delta": -0.6,
|
|
365
|
+
"qValue": 0.4
|
|
366
|
+
},
|
|
367
|
+
{
|
|
368
|
+
"scenarioId": "case-11",
|
|
369
|
+
"originalScore": 1,
|
|
370
|
+
"perturbedScore": 0.4,
|
|
371
|
+
"delta": -0.6,
|
|
372
|
+
"qValue": 0.4
|
|
373
|
+
}
|
|
374
|
+
],
|
|
375
|
+
"pairedTest": {
|
|
376
|
+
"w": 0,
|
|
377
|
+
"p": 0.00048828125,
|
|
378
|
+
"method": "exact",
|
|
379
|
+
"pFloor": 0.00048828125,
|
|
380
|
+
"nNonZero": 12
|
|
381
|
+
},
|
|
382
|
+
"medianDelta": -0.6,
|
|
383
|
+
"meanDelta": -0.5999999999999999,
|
|
384
|
+
"contaminationSuspected": true,
|
|
385
|
+
"reason": "paired p=0.0005 < 0.05 and median drop -0.6000 ≥ 0.05",
|
|
386
|
+
"n": 12
|
|
387
|
+
},
|
|
388
|
+
"limitations": [
|
|
389
|
+
"The global Wilcoxon test measures the constructed paired difference; it does not identify contamination as its cause.",
|
|
390
|
+
"Per-item qValue uses BH on 1 - abs(delta), without a per-item sampling null; it is a display aid in the inspected source.",
|
|
391
|
+
"The per-item qValues do not drive the global contaminationSuspected result."
|
|
392
|
+
]
|
|
393
|
+
},
|
|
394
|
+
"negativeOutcomeDirection": {
|
|
395
|
+
"inputs": {
|
|
396
|
+
"n": 8,
|
|
397
|
+
"rows": [
|
|
398
|
+
{
|
|
399
|
+
"quality": 0,
|
|
400
|
+
"successRate": 1
|
|
401
|
+
},
|
|
402
|
+
{
|
|
403
|
+
"quality": 0.14285714285714285,
|
|
404
|
+
"successRate": 0.8571428571428572
|
|
405
|
+
},
|
|
406
|
+
{
|
|
407
|
+
"quality": 0.2857142857142857,
|
|
408
|
+
"successRate": 0.7142857142857143
|
|
409
|
+
},
|
|
410
|
+
{
|
|
411
|
+
"quality": 0.42857142857142855,
|
|
412
|
+
"successRate": 0.5714285714285714
|
|
413
|
+
},
|
|
414
|
+
{
|
|
415
|
+
"quality": 0.5714285714285714,
|
|
416
|
+
"successRate": 0.4285714285714286
|
|
417
|
+
},
|
|
418
|
+
{
|
|
419
|
+
"quality": 0.7142857142857143,
|
|
420
|
+
"successRate": 0.2857142857142857
|
|
421
|
+
},
|
|
422
|
+
{
|
|
423
|
+
"quality": 0.8571428571428571,
|
|
424
|
+
"successRate": 0.1428571428571429
|
|
425
|
+
},
|
|
426
|
+
{
|
|
427
|
+
"quality": 1,
|
|
428
|
+
"successRate": 0
|
|
429
|
+
}
|
|
430
|
+
],
|
|
431
|
+
"rubric": "quality",
|
|
432
|
+
"outcome": "success_rate",
|
|
433
|
+
"desiredOutcomeDirection": "increase",
|
|
434
|
+
"seed": 1,
|
|
435
|
+
"bootstrapResamples": 100,
|
|
436
|
+
"researcherBootstrapResamples": 500,
|
|
437
|
+
"researcherSeed": "Derived deterministically by the validity helper",
|
|
438
|
+
"researcherFailureThreshold": 0.5
|
|
439
|
+
},
|
|
440
|
+
"results": {
|
|
441
|
+
"report": {
|
|
442
|
+
"pairs": [
|
|
443
|
+
{
|
|
444
|
+
"rubric": "quality",
|
|
445
|
+
"outcome": "success_rate",
|
|
446
|
+
"n": 8,
|
|
447
|
+
"pearson": -1,
|
|
448
|
+
"spearman": -1,
|
|
449
|
+
"ci95": {
|
|
450
|
+
"low": -1.0000000000000002,
|
|
451
|
+
"high": -0.9999999999999998
|
|
452
|
+
},
|
|
453
|
+
"verdict": "load_bearing"
|
|
454
|
+
}
|
|
455
|
+
],
|
|
456
|
+
"ranked": [
|
|
457
|
+
{
|
|
458
|
+
"rubric": "quality",
|
|
459
|
+
"bestOutcome": "success_rate",
|
|
460
|
+
"spearman": -1,
|
|
461
|
+
"pearson": -1,
|
|
462
|
+
"n": 8,
|
|
463
|
+
"verdict": "load_bearing"
|
|
464
|
+
}
|
|
465
|
+
],
|
|
466
|
+
"joinedSamples": 8,
|
|
467
|
+
"skippedRuns": 0,
|
|
468
|
+
"rubricsWithoutData": []
|
|
469
|
+
},
|
|
470
|
+
"researcher": {
|
|
471
|
+
"report": {
|
|
472
|
+
"pairs": [
|
|
473
|
+
{
|
|
474
|
+
"rubric": "quality",
|
|
475
|
+
"outcome": "success_rate",
|
|
476
|
+
"n": 8,
|
|
477
|
+
"pearson": -1,
|
|
478
|
+
"spearman": -1,
|
|
479
|
+
"ci95": {
|
|
480
|
+
"low": -1.0000000000000002,
|
|
481
|
+
"high": -0.9999999999999998
|
|
482
|
+
},
|
|
483
|
+
"verdict": "load_bearing"
|
|
484
|
+
}
|
|
485
|
+
],
|
|
486
|
+
"ranked": [
|
|
487
|
+
{
|
|
488
|
+
"rubric": "quality",
|
|
489
|
+
"bestOutcome": "success_rate",
|
|
490
|
+
"spearman": -1,
|
|
491
|
+
"pearson": -1,
|
|
492
|
+
"n": 8,
|
|
493
|
+
"verdict": "load_bearing"
|
|
494
|
+
}
|
|
495
|
+
],
|
|
496
|
+
"joinedSamples": 8,
|
|
497
|
+
"skippedRuns": 0,
|
|
498
|
+
"rubricsWithoutData": []
|
|
499
|
+
},
|
|
500
|
+
"failureGroups": 1,
|
|
501
|
+
"failures": [
|
|
502
|
+
{
|
|
503
|
+
"code": "low-score-same-candidate",
|
|
504
|
+
"description": "same-candidate scored < 0.5 on 4 run(s) (mean 0.214)",
|
|
505
|
+
"samples": 4
|
|
506
|
+
}
|
|
507
|
+
],
|
|
508
|
+
"proposedChanges": [
|
|
509
|
+
{
|
|
510
|
+
"kind": "reviewer_prompt",
|
|
511
|
+
"payload": {
|
|
512
|
+
"rubric": "quality",
|
|
513
|
+
"action": "up-weight",
|
|
514
|
+
"spearman": -1,
|
|
515
|
+
"bestOutcome": "success_rate"
|
|
516
|
+
},
|
|
517
|
+
"rationale": "predictive-validity Spearman=-1.000 vs success_rate (load-bearing); recommend up-weighting",
|
|
518
|
+
"expectedDelta": 0.05
|
|
519
|
+
}
|
|
520
|
+
]
|
|
521
|
+
}
|
|
522
|
+
},
|
|
523
|
+
"limitations": [
|
|
524
|
+
"Magnitude-based bucketing is intentional in existing tests, despite contradictory interface prose.",
|
|
525
|
+
"A negative association can be desirable for an outcome such as failure rate; direction needs explicit interpretation.",
|
|
526
|
+
"The researcher recommends increasing rubric weight despite its negative association with desired success rate; it does not execute or deploy that recommendation.",
|
|
527
|
+
"Constructed perfect correlation establishes neither causal validity nor held-out predictive performance."
|
|
528
|
+
]
|
|
529
|
+
},
|
|
530
|
+
"sequentialDependence": {
|
|
531
|
+
"inputs": {
|
|
532
|
+
"alpha": 0.05,
|
|
533
|
+
"minN": 5,
|
|
534
|
+
"maxN": 100,
|
|
535
|
+
"shuffleSeed": 1337,
|
|
536
|
+
"branchesEnumerated": 2,
|
|
537
|
+
"cellsPerBranch": 100,
|
|
538
|
+
"independentRandomSignsPerExperiment": 1,
|
|
539
|
+
"dataGeneratingProcess": "Draw one fair sign Z; set all 100 paired deltas equal to Z.",
|
|
540
|
+
"exchangeable": true,
|
|
541
|
+
"marginalMeanDelta": 0,
|
|
542
|
+
"conditionalMeanAfterFirstObservation": "Z, not necessarily <= 0"
|
|
543
|
+
},
|
|
544
|
+
"results": {
|
|
545
|
+
"branches": [
|
|
546
|
+
{
|
|
547
|
+
"commonDelta": -1,
|
|
548
|
+
"probability": 0.5,
|
|
549
|
+
"result": {
|
|
550
|
+
"decision": "hold",
|
|
551
|
+
"reasons": [
|
|
552
|
+
"sequentialPairedGate: undecided at pre-registered maxN=100 (e-value 1.00 < 1/α=20.00). This is NOT evidence of no effect — the effect may be real but smaller than this budget can detect; re-register with a larger N to test that"
|
|
553
|
+
],
|
|
554
|
+
"contributingGates": [
|
|
555
|
+
{
|
|
556
|
+
"name": "sequentialPairedGate",
|
|
557
|
+
"status": "fail",
|
|
558
|
+
"detail": {
|
|
559
|
+
"wealth": 1,
|
|
560
|
+
"n": 100,
|
|
561
|
+
"decided": false,
|
|
562
|
+
"alpha": 0.05,
|
|
563
|
+
"maxBet": 0.5,
|
|
564
|
+
"nullMean": 0.5,
|
|
565
|
+
"threshold": 20,
|
|
566
|
+
"sumX": 0,
|
|
567
|
+
"varSum": 0.15877048244745845,
|
|
568
|
+
"decision": "undecided-at-maxN",
|
|
569
|
+
"minN": 5,
|
|
570
|
+
"maxN": 100,
|
|
571
|
+
"scale": 1,
|
|
572
|
+
"shuffleSeed": 1337,
|
|
573
|
+
"direction": "increase",
|
|
574
|
+
"minEffect": 0,
|
|
575
|
+
"pairedN": 100
|
|
576
|
+
}
|
|
577
|
+
}
|
|
578
|
+
],
|
|
579
|
+
"delta": -1
|
|
580
|
+
}
|
|
581
|
+
},
|
|
582
|
+
{
|
|
583
|
+
"commonDelta": 1,
|
|
584
|
+
"probability": 0.5,
|
|
585
|
+
"result": {
|
|
586
|
+
"decision": "ship",
|
|
587
|
+
"reasons": [
|
|
588
|
+
"sequentialPairedGate: e-value 22.74 ≥ 1/α=20.00 at n=15 (minN=5): the paired improvement exceeds 0 at anytime-valid level α=0.05"
|
|
589
|
+
],
|
|
590
|
+
"contributingGates": [
|
|
591
|
+
{
|
|
592
|
+
"name": "sequentialPairedGate",
|
|
593
|
+
"status": "pass",
|
|
594
|
+
"detail": {
|
|
595
|
+
"wealth": 22.737367544323206,
|
|
596
|
+
"n": 15,
|
|
597
|
+
"decided": true,
|
|
598
|
+
"alpha": 0.05,
|
|
599
|
+
"maxBet": 0.5,
|
|
600
|
+
"nullMean": 0.5,
|
|
601
|
+
"threshold": 20,
|
|
602
|
+
"decidedAtN": 15,
|
|
603
|
+
"sumX": 15,
|
|
604
|
+
"varSum": 0.14608663336124675,
|
|
605
|
+
"decision": "promote",
|
|
606
|
+
"minN": 5,
|
|
607
|
+
"maxN": 100,
|
|
608
|
+
"scale": 1,
|
|
609
|
+
"shuffleSeed": 1337,
|
|
610
|
+
"direction": "increase",
|
|
611
|
+
"minEffect": 0,
|
|
612
|
+
"pairedN": 100
|
|
613
|
+
}
|
|
614
|
+
}
|
|
615
|
+
],
|
|
616
|
+
"delta": 1
|
|
617
|
+
}
|
|
618
|
+
}
|
|
619
|
+
],
|
|
620
|
+
"promotionProbabilityUnderMarginalZeroProcess": 0.5
|
|
621
|
+
},
|
|
622
|
+
"limitations": [
|
|
623
|
+
"Enumerates both equiprobable branches exactly; this is not a Monte Carlo estimate.",
|
|
624
|
+
"The process violates the conditional-mean null required by the e-process core.",
|
|
625
|
+
"This refutes sufficiency of exchangeability and shuffling, not the valid e-process theorem or any measured production dataset."
|
|
626
|
+
]
|
|
627
|
+
},
|
|
628
|
+
"powerFloor": {
|
|
629
|
+
"inputs": {
|
|
630
|
+
"gate": {
|
|
631
|
+
"kind": "power-floor",
|
|
632
|
+
"target": 0.8,
|
|
633
|
+
"effectGrid": [
|
|
634
|
+
0.01,
|
|
635
|
+
1
|
|
636
|
+
],
|
|
637
|
+
"sim": {
|
|
638
|
+
"trials": 1,
|
|
639
|
+
"resamples": 1,
|
|
640
|
+
"seed": 1
|
|
641
|
+
}
|
|
642
|
+
},
|
|
643
|
+
"curve": [
|
|
644
|
+
{
|
|
645
|
+
"effect": 0.01,
|
|
646
|
+
"power": 0.1
|
|
647
|
+
},
|
|
648
|
+
{
|
|
649
|
+
"effect": 1,
|
|
650
|
+
"power": 1
|
|
651
|
+
}
|
|
652
|
+
],
|
|
653
|
+
"practicalEffectForInterpretation": 0.01,
|
|
654
|
+
"curveSource": "Supplied deterministic fixture; no power simulation is run."
|
|
655
|
+
},
|
|
656
|
+
"results": {
|
|
657
|
+
"id": "floor",
|
|
658
|
+
"passed": true,
|
|
659
|
+
"evidence": {
|
|
660
|
+
"target": 0.8,
|
|
661
|
+
"maxPower": 1,
|
|
662
|
+
"curve": [
|
|
663
|
+
{
|
|
664
|
+
"effect": 0.01,
|
|
665
|
+
"power": 0.1
|
|
666
|
+
},
|
|
667
|
+
{
|
|
668
|
+
"effect": 1,
|
|
669
|
+
"power": 1
|
|
670
|
+
}
|
|
671
|
+
]
|
|
672
|
+
}
|
|
673
|
+
},
|
|
674
|
+
"limitations": [
|
|
675
|
+
"The inspected gate documents a maximum-over-grid structural feasibility check; this output matches that contract.",
|
|
676
|
+
"Passing this gate does not establish target power at the practical effect of 0.01.",
|
|
677
|
+
"The fixture powers are inputs, not measured or simulated power estimates."
|
|
678
|
+
]
|
|
679
|
+
}
|
|
680
|
+
},
|
|
681
|
+
"separateVerificationAtReviewedBase": {
|
|
682
|
+
"provenance": "Historical checks at reviewedBaseRevision; this diagnostic does not rerun them.",
|
|
683
|
+
"sourceRevision": "fe1cc5111aab5d588bf7db3a3785325635937a91",
|
|
684
|
+
"checks": [
|
|
685
|
+
{
|
|
686
|
+
"command": "pnpm typecheck",
|
|
687
|
+
"result": "passed"
|
|
688
|
+
},
|
|
689
|
+
{
|
|
690
|
+
"command": "pnpm build",
|
|
691
|
+
"result": "passed"
|
|
692
|
+
},
|
|
693
|
+
{
|
|
694
|
+
"command": "pnpm verify:package",
|
|
695
|
+
"result": "passed"
|
|
696
|
+
}
|
|
697
|
+
],
|
|
698
|
+
"tests": {
|
|
699
|
+
"command": "pnpm test -- tests/experiment/preregistration-acceptance.test.ts tests/experiment/power.test.ts tests/contamination-guard.test.ts tests/rl-predictive-validity-researcher.test.ts tests/rubric-predictive-validity.test.ts tests/meta-eval.test.ts",
|
|
700
|
+
"observedScope": "The package command expanded to the full Vitest suite.",
|
|
701
|
+
"files": {
|
|
702
|
+
"passed": 399,
|
|
703
|
+
"skipped": 2
|
|
704
|
+
},
|
|
705
|
+
"tests": {
|
|
706
|
+
"passed": 5876,
|
|
707
|
+
"skipped": 3
|
|
708
|
+
},
|
|
709
|
+
"result": "passed"
|
|
710
|
+
},
|
|
711
|
+
"limitation": "Passing repository checks do not establish correctness of the counterexample behaviors recorded above."
|
|
712
|
+
}
|
|
713
|
+
}
|