@tangle-network/agent-eval 0.180.0 → 0.181.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +60 -0
- package/README.md +119 -159
- package/dist/adapters/http.d.ts +2 -2
- package/dist/{agent-profile-B7yErX0q.d.ts → agent-profile-CivaSsSy.d.ts} +4 -4
- package/dist/{agent-profile-B7yErX0q.d.ts.map → agent-profile-CivaSsSy.d.ts.map} +1 -1
- package/dist/{agent-profile-cell-0gSi5ffD.js → agent-profile-cell-Cv6UA-W_.js} +20 -57
- package/dist/agent-profile-cell-Cv6UA-W_.js.map +1 -0
- package/dist/{agent-profile-cell-CTOZJUuE.d.ts → agent-profile-cell-s__adRnK.d.ts} +3 -3
- package/dist/agent-profile-cell-s__adRnK.d.ts.map +1 -0
- package/dist/analyst/index.d.ts +10 -10
- package/dist/analyst/index.js +3 -3
- package/dist/ast-CP9ae9B0.js +557 -0
- package/dist/ast-CP9ae9B0.js.map +1 -0
- package/dist/ast-hI-vjW6J.d.ts +457 -0
- package/dist/ast-hI-vjW6J.d.ts.map +1 -0
- package/dist/{benchmark-command-D4vpnAdO.js → benchmark-command-B57n9vjz.js} +7 -6
- package/dist/{benchmark-command-D4vpnAdO.js.map → benchmark-command-B57n9vjz.js.map} +1 -1
- package/dist/benchmarks/index.d.ts +4 -4
- package/dist/benchmarks/index.js +3 -3
- package/dist/campaign/index.d.ts +6 -6
- package/dist/campaign/index.js +8 -8
- package/dist/{campaign-BGEurASO.js → campaign-4_ppJW5X.js} +12 -12
- package/dist/{campaign-BGEurASO.js.map → campaign-4_ppJW5X.js.map} +1 -1
- package/dist/{campaign-evidence-D8DBLqLI.js → campaign-evidence-B8oF9xQ6.js} +515 -471
- package/dist/campaign-evidence-B8oF9xQ6.js.map +1 -0
- package/dist/cli.js +4 -7
- package/dist/cli.js.map +1 -1
- package/dist/{client-BlLY6o2w.js → client-CXE-U1SA.js} +3 -1
- package/dist/client-CXE-U1SA.js.map +1 -0
- package/dist/{client-CuQgX33c.d.ts → client-kh2jOjTK.d.ts} +4 -4
- package/dist/{client-CuQgX33c.d.ts.map → client-kh2jOjTK.d.ts.map} +1 -1
- package/dist/contract/index.d.ts +13 -13
- package/dist/contract/index.js +10 -9
- package/dist/contract/index.js.map +1 -1
- package/dist/{default-registry-IGDE9XIC.d.ts → default-registry-BwDSWVzg.d.ts} +6 -6
- package/dist/{default-registry-IGDE9XIC.d.ts.map → default-registry-BwDSWVzg.d.ts.map} +1 -1
- package/dist/{define-agent-eval-Cx4Ls9ta.d.ts → define-agent-eval-CwOWWQt_.d.ts} +33 -12
- package/dist/define-agent-eval-CwOWWQt_.d.ts.map +1 -0
- package/dist/{define-agent-eval-Dzidv34q.js → define-agent-eval-Ddu33JH9.js} +134 -67
- package/dist/define-agent-eval-Ddu33JH9.js.map +1 -0
- package/dist/{dspy-rlm-engine-xKiWmj_G.js → dspy-rlm-engine-S53V0HhE.js} +2 -2
- package/dist/{dspy-rlm-engine-xKiWmj_G.js.map → dspy-rlm-engine-S53V0HhE.js.map} +1 -1
- package/dist/{engine-CX8ReXkn.d.ts → engine-DS1cysJy.d.ts} +10 -7
- package/dist/engine-DS1cysJy.d.ts.map +1 -0
- package/dist/{eval-campaign-Cs-7MiCs.js → eval-campaign-aYdtjtJR.js} +4 -4
- package/dist/{eval-campaign-Cs-7MiCs.js.map → eval-campaign-aYdtjtJR.js.map} +1 -1
- package/dist/{exact-types-B7LC1EyX.d.ts → exact-types-BZDe0W2D.d.ts} +2 -2
- package/dist/{exact-types-B7LC1EyX.d.ts.map → exact-types-BZDe0W2D.d.ts.map} +1 -1
- package/dist/experiment/index.d.ts +27 -477
- package/dist/experiment/index.d.ts.map +1 -1
- package/dist/experiment/index.js +95 -559
- package/dist/experiment/index.js.map +1 -1
- package/dist/{experiment-tracker-B3TiF5-u.d.ts → experiment-tracker-C7PfnF4b.d.ts} +2 -2
- package/dist/{experiment-tracker-B3TiF5-u.d.ts.map → experiment-tracker-C7PfnF4b.d.ts.map} +1 -1
- package/dist/{external-optimizer-process-Dlz8YxrT.js → external-optimizer-process-QDRURJAM.js} +3 -3
- package/dist/{external-optimizer-process-Dlz8YxrT.js.map → external-optimizer-process-QDRURJAM.js.map} +1 -1
- package/dist/{external-optimizer-subprocess-q3VzlGAO.js → external-optimizer-subprocess-D4dzUBZI.js} +3 -2
- package/dist/{external-optimizer-subprocess-q3VzlGAO.js.map → external-optimizer-subprocess-D4dzUBZI.js.map} +1 -1
- package/dist/{feedback-trajectory-eHWNv5Aj.d.ts → feedback-trajectory-CXmtITBo.d.ts} +3 -3
- package/dist/{feedback-trajectory-eHWNv5Aj.d.ts.map → feedback-trajectory-CXmtITBo.d.ts.map} +1 -1
- package/dist/hosted/index.d.ts +2 -2
- package/dist/hosted/index.d.ts.map +1 -1
- package/dist/hosted/index.js +1 -1
- package/dist/{index-BxWvILU8.d.ts → index-Bp_6sj3x.d.ts} +109 -56
- package/dist/index-Bp_6sj3x.d.ts.map +1 -0
- package/dist/{index-e7LXeRVa.d.ts → index-CJ3LhKIX.d.ts} +2 -2
- package/dist/{index-e7LXeRVa.d.ts.map → index-CJ3LhKIX.d.ts.map} +1 -1
- package/dist/{index-CiUjjEIa.d.ts → index-DNntP4ch.d.ts} +7 -7
- package/dist/{index-CiUjjEIa.d.ts.map → index-DNntP4ch.d.ts.map} +1 -1
- package/dist/{index-DxNYmx4a.d.ts → index-DoykkxW0.d.ts} +11 -11
- package/dist/{index-DxNYmx4a.d.ts.map → index-DoykkxW0.d.ts.map} +1 -1
- package/dist/index.d.ts +28 -28
- package/dist/index.js +24 -15
- package/dist/index.js.map +1 -1
- package/dist/{insight-report-DETqPc_A.d.ts → insight-report-D1qa0HWs.d.ts} +9 -5
- package/dist/{insight-report-DETqPc_A.d.ts.map → insight-report-D1qa0HWs.d.ts.map} +1 -1
- package/dist/{integrity-BKTcA-HP.d.ts → integrity-rGOfSUle.d.ts} +2 -2
- package/dist/{integrity-BKTcA-HP.d.ts.map → integrity-rGOfSUle.d.ts.map} +1 -1
- package/dist/{ledger-core-Cs9f7385.js → journal-Cs9f7385.js} +1 -1
- package/dist/journal-Cs9f7385.js.map +1 -0
- package/dist/{judge-calibration-C5CbMYce.d.ts → judge-calibration-DFtEMlde.d.ts} +31 -2
- package/dist/judge-calibration-DFtEMlde.d.ts.map +1 -0
- package/dist/{judge-calibration-BnpVKtnb.js → judge-calibration-DYmaBtJr.js} +48 -2
- package/dist/{judge-calibration-BnpVKtnb.js.map → judge-calibration-DYmaBtJr.js.map} +1 -1
- package/dist/ledger-core/index.d.ts +1 -1
- package/dist/ledger-core/index.js +1 -1
- package/dist/{llm-judge-v80Kmu9g.js → llm-judge-DEFZeSiu.js} +645 -456
- package/dist/llm-judge-DEFZeSiu.js.map +1 -0
- package/dist/{matrix-DeMmnWrP.d.ts → matrix-CyhW-vgJ.d.ts} +2 -2
- package/dist/{matrix-DeMmnWrP.d.ts.map → matrix-CyhW-vgJ.d.ts.map} +1 -1
- package/dist/meta-eval/index.d.ts +138 -7
- package/dist/meta-eval/index.d.ts.map +1 -1
- package/dist/meta-eval/index.js +245 -97
- package/dist/meta-eval/index.js.map +1 -1
- package/dist/{mint-Cc1_zwRQ.js → mint-ySIIkKlV.js} +2 -2
- package/dist/{mint-Cc1_zwRQ.js.map → mint-ySIIkKlV.js.map} +1 -1
- package/dist/multishot/golden/index.d.ts +1 -1
- package/dist/multishot/index.d.ts +2 -2
- package/dist/openapi.json +1 -1
- package/dist/outcome-store-BXlkwMPR.js +131 -0
- package/dist/outcome-store-BXlkwMPR.js.map +1 -0
- package/dist/{outcome-store-BYHIuO0e.d.ts → outcome-store-CNt4iZ67.d.ts} +18 -25
- package/dist/outcome-store-CNt4iZ67.d.ts.map +1 -0
- package/dist/{paired-promotion-decision-CGzg0cI_.d.ts → paired-promotion-decision-DPsMQm-0.d.ts} +13 -7
- package/dist/{paired-promotion-decision-CGzg0cI_.d.ts.map → paired-promotion-decision-DPsMQm-0.d.ts.map} +1 -1
- package/dist/pipelines/index.js +1 -1
- package/dist/{produced-state-Cv0kJJuP.js → produced-state-BHboMaab.js} +3 -3
- package/dist/{produced-state-Cv0kJJuP.js.map → produced-state-BHboMaab.js.map} +1 -1
- package/dist/profile-cell.d.ts +1 -1
- package/dist/profile-cell.js +1 -1
- package/dist/{promotion-policy-DWOm70gx.js → promotion-policy-CDMMxzb6.js} +28 -40
- package/dist/promotion-policy-CDMMxzb6.js.map +1 -0
- package/dist/{registry-ByVld1-5.d.ts → registry-BRbB6Y0v.d.ts} +4 -4
- package/dist/{registry-ByVld1-5.d.ts.map → registry-BRbB6Y0v.d.ts.map} +1 -1
- package/dist/{release-confidence-BAcNYOf1.d.ts → release-confidence-BcqeQTHW.d.ts} +3 -3
- package/dist/{release-confidence-BAcNYOf1.d.ts.map → release-confidence-BcqeQTHW.d.ts.map} +1 -1
- package/dist/{release-confidence-BcGCclTB.js → release-confidence-DMg8n18l.js} +2 -2
- package/dist/{release-confidence-BcGCclTB.js.map → release-confidence-DMg8n18l.js.map} +1 -1
- package/dist/reporting.d.ts +4 -4
- package/dist/reporting.js +3 -3
- package/dist/{researcher-jsW1X94L.d.ts → researcher-64T49THL.d.ts} +6 -6
- package/dist/{researcher-jsW1X94L.d.ts.map → researcher-64T49THL.d.ts.map} +1 -1
- package/dist/{reward-hacking-ZXEi9VCq.d.ts → reward-hacking-uzO_ihep.d.ts} +2 -2
- package/dist/{reward-hacking-ZXEi9VCq.d.ts.map → reward-hacking-uzO_ihep.d.ts.map} +1 -1
- package/dist/rl.d.ts +53 -99
- package/dist/rl.d.ts.map +1 -1
- package/dist/rl.js +182 -169
- package/dist/rl.js.map +1 -1
- package/dist/rollout/index.d.ts +1 -1
- package/dist/rollout/index.js +2 -2
- package/dist/{rollout-DmoJVqrF.js → rollout-B-UF5R6w.js} +2 -2
- package/dist/{rollout-DmoJVqrF.js.map → rollout-B-UF5R6w.js.map} +1 -1
- package/dist/rubric-predictive-validity-Bmj2_cll.d.ts +79 -0
- package/dist/rubric-predictive-validity-Bmj2_cll.d.ts.map +1 -0
- package/dist/rubric-predictive-validity-CCK-1B7w.js +178 -0
- package/dist/rubric-predictive-validity-CCK-1B7w.js.map +1 -0
- package/dist/{run-record-DTv1MdjK.d.ts → run-record-BiTWauyO.d.ts} +2 -2
- package/dist/{run-record-DTv1MdjK.d.ts.map → run-record-BiTWauyO.d.ts.map} +1 -1
- package/dist/run-record-Br-Yzt_k.js +464 -0
- package/dist/run-record-Br-Yzt_k.js.map +1 -0
- package/dist/{run-record-DQpSf7t-.js → run-record-DualPTn2.js} +2 -2
- package/dist/{run-record-DQpSf7t-.js.map → run-record-DualPTn2.js.map} +1 -1
- package/dist/{semantic-concept-judge-Bi6_iGqg.js → semantic-concept-judge-Bm5JDEKO.js} +3 -3
- package/dist/{semantic-concept-judge-Bi6_iGqg.js.map → semantic-concept-judge-Bm5JDEKO.js.map} +1 -1
- package/dist/{sequential-B5gXgcyp.js → sequential-DAsyV2T9.js} +42 -25
- package/dist/sequential-DAsyV2T9.js.map +1 -0
- package/dist/{series-convergence-DeG33RpC.d.ts → series-convergence-BnMs_uAr.d.ts} +3 -3
- package/dist/{series-convergence-DeG33RpC.d.ts.map → series-convergence-BnMs_uAr.d.ts.map} +1 -1
- package/dist/{skillopt-optimization-method-C3oYul8v.js → skillopt-optimization-method-CL_0aArC.js} +5 -5
- package/dist/{skillopt-optimization-method-C3oYul8v.js.map → skillopt-optimization-method-CL_0aArC.js.map} +1 -1
- package/dist/{statistical-heldout-0La5ZTlv.d.ts → statistical-heldout-CpVd6FmY.d.ts} +207 -144
- package/dist/statistical-heldout-CpVd6FmY.d.ts.map +1 -0
- package/dist/{store-tool-spans-4J1EDElP.d.ts → store-tool-spans-Dt-YdAuE.d.ts} +6 -6
- package/dist/{store-tool-spans-4J1EDElP.d.ts.map → store-tool-spans-Dt-YdAuE.d.ts.map} +1 -1
- package/dist/{summary-report-gMrbYawB.d.ts → summary-report-D1h4dlrK.d.ts} +3 -3
- package/dist/{summary-report-gMrbYawB.d.ts.map → summary-report-D1h4dlrK.d.ts.map} +1 -1
- package/dist/{summary-report-B16xy9Kd.js → summary-report-e-MaOAHV.js} +2 -2
- package/dist/{summary-report-B16xy9Kd.js.map → summary-report-e-MaOAHV.js.map} +1 -1
- package/dist/{tool-groups-2QA0S7dK.d.ts → tool-groups-B2bSNaJB.d.ts} +3 -3
- package/dist/tool-groups-B2bSNaJB.d.ts.map +1 -0
- package/dist/{tool-waste-B9tdWV6g.js → tool-waste-C7MU9u1e.js} +2 -2
- package/dist/{tool-waste-B9tdWV6g.js.map → tool-waste-C7MU9u1e.js.map} +1 -1
- package/dist/trace-repair/index.d.ts +2 -2
- package/dist/traces.d.ts +6 -6
- package/dist/traces.js +1 -1
- package/dist/{types-BmlkCrg0.d.ts → types-BvZoPTGa.d.ts} +3 -3
- package/dist/{types-BmlkCrg0.d.ts.map → types-BvZoPTGa.d.ts.map} +1 -1
- package/dist/{types-gvRsyJLh.d.ts → types-CBbLtr2J.d.ts} +38 -3
- package/dist/{types-gvRsyJLh.d.ts.map → types-CBbLtr2J.d.ts.map} +1 -1
- package/dist/{types-C34V4Vto.d.ts → types-CS0qk_Yp.d.ts} +4 -4
- package/dist/{types-C34V4Vto.d.ts.map → types-CS0qk_Yp.d.ts.map} +1 -1
- package/dist/{types-DzuaM493.d.ts → types-D7gEdPoQ.d.ts} +3 -3
- package/dist/{types-DzuaM493.d.ts.map → types-D7gEdPoQ.d.ts.map} +1 -1
- package/dist/wire/index.d.ts +2 -2
- package/docs/adapters-observability.md +14 -0
- package/docs/campaign-proposers.md +86 -128
- package/docs/charter.md +108 -112
- package/docs/concepts.md +157 -69
- package/docs/design/mlbenchmarks-book-review.md +440 -0
- package/docs/design/mlbenchmarks-review/observations.json +713 -0
- package/docs/design/mlbenchmarks-review/probes.mts +476 -0
- package/docs/design/mlbenchmarks-review/sources.json +200 -0
- package/docs/design/self-improvement-evidence-audit.md +263 -0
- package/docs/design.md +2 -1
- package/docs/eval-surface-map.md +95 -42
- package/docs/evaluation-integrity.md +220 -0
- package/docs/experiment.md +111 -55
- package/docs/feature-guide.md +5 -6
- package/docs/hosted-ingest-spec.md +4 -11
- package/docs/insight-report.md +187 -455
- package/docs/outcome-validity.md +182 -0
- package/docs/product-eval-adoption.md +1 -2
- package/docs/research-report-methodology.md +7 -7
- package/docs/search-history-receipts.md +8 -0
- package/docs/statistical-evidence.md +129 -0
- package/docs/verdicts.md +76 -49
- package/package.json +1 -1
- package/dist/agent-profile-cell-0gSi5ffD.js.map +0 -1
- package/dist/agent-profile-cell-CTOZJUuE.d.ts.map +0 -1
- package/dist/campaign-evidence-D8DBLqLI.js.map +0 -1
- package/dist/client-BlLY6o2w.js.map +0 -1
- package/dist/define-agent-eval-Cx4Ls9ta.d.ts.map +0 -1
- package/dist/define-agent-eval-Dzidv34q.js.map +0 -1
- package/dist/engine-CX8ReXkn.d.ts.map +0 -1
- package/dist/index-BxWvILU8.d.ts.map +0 -1
- package/dist/judge-calibration-C5CbMYce.d.ts.map +0 -1
- package/dist/ledger-core-Cs9f7385.js.map +0 -1
- package/dist/llm-judge-v80Kmu9g.js.map +0 -1
- package/dist/outcome-store-BYHIuO0e.d.ts.map +0 -1
- package/dist/outcome-store-ChBKlTd_.js +0 -75
- package/dist/outcome-store-ChBKlTd_.js.map +0 -1
- package/dist/promotion-policy-DWOm70gx.js.map +0 -1
- package/dist/rubric-predictive-validity-2D5Gw9z9.js +0 -131
- package/dist/rubric-predictive-validity-2D5Gw9z9.js.map +0 -1
- package/dist/rubric-predictive-validity-Dl1dvKCv.d.ts +0 -75
- package/dist/rubric-predictive-validity-Dl1dvKCv.d.ts.map +0 -1
- package/dist/run-record-CR63CpHK.js +0 -216
- package/dist/run-record-CR63CpHK.js.map +0 -1
- package/dist/sequential-B5gXgcyp.js.map +0 -1
- package/dist/statistical-heldout-0La5ZTlv.d.ts.map +0 -1
- package/dist/tool-groups-2QA0S7dK.d.ts.map +0 -1
|
@@ -1,11 +1,264 @@
|
|
|
1
|
-
import { t as AgentEvalError } from "./errors-Dngq5h35.js";
|
|
1
|
+
import { s as ValidationError, t as AgentEvalError } from "./errors-Dngq5h35.js";
|
|
2
2
|
import { a as hashCanonical, i as compareCodeUnits, r as canonicalString } from "./canonical-DPyQ_rpt.js";
|
|
3
|
-
import {
|
|
4
|
-
import {
|
|
3
|
+
import { r as pairedBootstrap } from "./paired-tests-C8iCsioC.js";
|
|
4
|
+
import { n as campaignCellExecutionEvidence, o as projectCampaignCellQuality, s as decidePairedPromotion } from "./run-record-Br-Yzt_k.js";
|
|
5
5
|
import { n as contentHash } from "./verdict-cache-B3eCVQtY.js";
|
|
6
|
-
import {
|
|
6
|
+
import { t as FileLedgerJournal } from "./journal-Cs9f7385.js";
|
|
7
|
+
import { g as readField } from "./ast-CP9ae9B0.js";
|
|
7
8
|
import { createHash } from "node:crypto";
|
|
9
|
+
import { z } from "zod";
|
|
8
10
|
import { join } from "node:path";
|
|
11
|
+
//#region src/campaign/gates/statistical-heldout.ts
|
|
12
|
+
/**
|
|
13
|
+
* Held-out inference pairs execution cells and judge identities before taking
|
|
14
|
+
* means within registered independent units. Repetitions can improve a unit's
|
|
15
|
+
* precision without increasing n. Ungrouped inference concerns independently
|
|
16
|
+
* sampled execution cells conditional on a fixed scenario roster.
|
|
17
|
+
*
|
|
18
|
+
* The shared paired decision rule selects the estimator and statistical test.
|
|
19
|
+
* Dimension reports retain missing coverage so required safety checks cannot
|
|
20
|
+
* pass through absent evidence. Thresholds use the judge's native score scale.
|
|
21
|
+
*/
|
|
22
|
+
/** Tie fraction at/above which a gate annotates its verdict with the tie share.
|
|
23
|
+
* Tie-domination of the median bites structurally at >= 0.5 (the median is then
|
|
24
|
+
* 0 by construction); 0.4 is a softer warn threshold that flags a run APPROACHING
|
|
25
|
+
* that regime, so an operator sees it before the median goes fully blind. */
|
|
26
|
+
const TIE_WARN_FRACTION = .4;
|
|
27
|
+
/** Campaign cell IDs append a numeric repetition after the scenario's full ID. */
|
|
28
|
+
function scenarioIdFromCellId(cellId) {
|
|
29
|
+
const separator = cellId.lastIndexOf(":");
|
|
30
|
+
const repetition = cellId.slice(separator + 1);
|
|
31
|
+
if (separator < 1 || cellId.trim() !== cellId || !/^(0|[1-9]\d*)$/.test(repetition) || !Number.isSafeInteger(Number(repetition))) throw new Error(`pairHoldout: malformed cellId '${cellId}'; expected scenarioId:rep`);
|
|
32
|
+
return cellId.slice(0, separator);
|
|
33
|
+
}
|
|
34
|
+
/** Preserve cell pairing before taking equal-weight independent-unit means. */
|
|
35
|
+
function aggregatePairedHoldout(paired, independentUnitByScenarioId) {
|
|
36
|
+
if (paired.before.length !== paired.after.length || paired.before.length !== paired.cellIds.length) throw new Error("aggregatePairedHoldout: scores and cellIds must have the same length");
|
|
37
|
+
if (new Set(paired.cellIds).size !== paired.cellIds.length) throw new Error("aggregatePairedHoldout: duplicate cellIds cannot count as new observations");
|
|
38
|
+
if (paired.before.some((value) => !Number.isFinite(value)) || paired.after.some((value) => !Number.isFinite(value))) throw new Error("aggregatePairedHoldout: paired scores must be finite");
|
|
39
|
+
if (independentUnitByScenarioId === void 0) return {
|
|
40
|
+
before: [...paired.before],
|
|
41
|
+
after: [...paired.after],
|
|
42
|
+
unitIds: [...paired.cellIds]
|
|
43
|
+
};
|
|
44
|
+
const scenarioIds = paired.cellIds.map(scenarioIdFromCellId);
|
|
45
|
+
const groups = /* @__PURE__ */ new Map();
|
|
46
|
+
for (let i = 0; i < paired.cellIds.length; i++) {
|
|
47
|
+
const scenarioId = scenarioIds[i];
|
|
48
|
+
const unitId = independentUnitByScenarioId.get(scenarioId);
|
|
49
|
+
if (typeof unitId !== "string" || unitId.length === 0 || unitId.trim() !== unitId) throw new Error(`aggregatePairedHoldout: missing independent unit for scenario '${scenarioId}'`);
|
|
50
|
+
const group = groups.get(unitId) ?? {
|
|
51
|
+
before: 0,
|
|
52
|
+
after: 0,
|
|
53
|
+
n: 0
|
|
54
|
+
};
|
|
55
|
+
group.before += paired.before[i];
|
|
56
|
+
group.after += paired.after[i];
|
|
57
|
+
group.n += 1;
|
|
58
|
+
groups.set(unitId, group);
|
|
59
|
+
}
|
|
60
|
+
const unitIds = [...groups.keys()].sort();
|
|
61
|
+
return {
|
|
62
|
+
before: unitIds.map((id) => groups.get(id).before / groups.get(id).n),
|
|
63
|
+
after: unitIds.map((id) => groups.get(id).after / groups.get(id).n),
|
|
64
|
+
unitIds
|
|
65
|
+
};
|
|
66
|
+
}
|
|
67
|
+
/**
|
|
68
|
+
* Pair candidate vs baseline holdout observations by FULL cellId. `select`
|
|
69
|
+
* pulls the scalar from a cell's judge reports (composite, or a named
|
|
70
|
+
* dimension); a cell contributes the mean of `select` across its judges. Cells
|
|
71
|
+
* whose scenario is not in `scenarioIds`, or where `select` is undefined for
|
|
72
|
+
* every judge on both sides, are skipped. The selected judge IDs must agree
|
|
73
|
+
* within each pair. Throws when the two maps disagree on holdout cell IDs — a
|
|
74
|
+
* load-bearing invariant: the baseline + winner holdout campaigns run the same
|
|
75
|
+
* scenarios with the same seed base, so their cellIds MUST align; a mismatch
|
|
76
|
+
* means a silent pairing bug, not a soft fallback.
|
|
77
|
+
*/
|
|
78
|
+
function pairHoldout(candidate, baseline, scenarioIds, select) {
|
|
79
|
+
const cellValues = (byCell, cellId) => {
|
|
80
|
+
const scores = byCell.get(cellId);
|
|
81
|
+
const values = /* @__PURE__ */ new Map();
|
|
82
|
+
if (!scores) return values;
|
|
83
|
+
for (const [judgeId, s] of Object.entries(scores)) {
|
|
84
|
+
if (s.failed === true) throw new Error(`pairHoldout: cell '${cellId}' contains a failed judge score`);
|
|
85
|
+
const v = select(s);
|
|
86
|
+
if (typeof v === "number" && !Number.isFinite(v)) throw new Error(`pairHoldout: cell '${cellId}' contains a non-finite selected score`);
|
|
87
|
+
if (typeof v === "number") values.set(judgeId, v);
|
|
88
|
+
}
|
|
89
|
+
return values;
|
|
90
|
+
};
|
|
91
|
+
const inScope = (cellId) => scenarioIds.has(scenarioIdFromCellId(cellId));
|
|
92
|
+
const candCells = [...candidate.keys()].filter(inScope).sort();
|
|
93
|
+
const baseCells = [...baseline.keys()].filter(inScope).sort();
|
|
94
|
+
if (candCells.length !== baseCells.length || candCells.some((c, i) => c !== baseCells[i])) throw new Error(`pairHoldout: candidate/baseline holdout cells do not align — candidate=[${candCells.join(",")}] baseline=[${baseCells.join(",")}]. Both holdout campaigns must run the same scenarios with the same seed base.`);
|
|
95
|
+
const before = [];
|
|
96
|
+
const after = [];
|
|
97
|
+
const cellIds = [];
|
|
98
|
+
for (const cellId of candCells) {
|
|
99
|
+
const b = cellValues(baseline, cellId);
|
|
100
|
+
const a = cellValues(candidate, cellId);
|
|
101
|
+
if (b.size === 0 && a.size === 0) continue;
|
|
102
|
+
if (b.size === 0 || a.size === 0) throw new Error(`pairHoldout: cell '${cellId}' has a selected score on only one arm`);
|
|
103
|
+
if (b.size !== a.size || [...b.keys()].some((id) => !a.has(id))) throw new Error(`pairHoldout: cell '${cellId}' selected judge IDs do not align`);
|
|
104
|
+
const judgeIds = [...b.keys()].sort();
|
|
105
|
+
before.push(meanSelectedScores(judgeIds.map((id) => b.get(id))));
|
|
106
|
+
after.push(meanSelectedScores(judgeIds.map((id) => a.get(id))));
|
|
107
|
+
cellIds.push(cellId);
|
|
108
|
+
}
|
|
109
|
+
return {
|
|
110
|
+
before,
|
|
111
|
+
after,
|
|
112
|
+
cellIds
|
|
113
|
+
};
|
|
114
|
+
}
|
|
115
|
+
function meanSelectedScores(values) {
|
|
116
|
+
const first = values[0];
|
|
117
|
+
return values.every((value) => value === first) ? first : values.reduce((sum, value) => sum + value, 0) / values.length;
|
|
118
|
+
}
|
|
119
|
+
/**
|
|
120
|
+
* Significance of the held-out composite lift: ship only when the lower bound
|
|
121
|
+
* of the interval the outcome's shape ADMITS exceeds `deltaThreshold` (default
|
|
122
|
+
* 0 ⇒ "confidently positive"). Interpret `deltaThreshold` in the judge's native
|
|
123
|
+
* scale.
|
|
124
|
+
*
|
|
125
|
+
* The decision is delegated whole to {@link decidePairedPromotion}, the one
|
|
126
|
+
* copy of the rule (`src/paired-promotion-decision.ts`), which `HeldOutGate`
|
|
127
|
+
* also calls. That module's header carries the measurements; the short version
|
|
128
|
+
* is three guards a bare `bootstrap.low > threshold` does not have:
|
|
129
|
+
*
|
|
130
|
+
* - a two-point (pass/fail) outcome decides on Tango's SCORE interval, the
|
|
131
|
+
* only paired-binary construction that stays valid at a nonzero margin;
|
|
132
|
+
* - McNemar's exact test VETOES at any non-negative threshold;
|
|
133
|
+
* - a ZERO-WIDTH interval is refused rather than promoted, in either
|
|
134
|
+
* direction — [0,0] clears every negative threshold and [g,g] clears every
|
|
135
|
+
* threshold below g, and both are an absence of evidence, not a result.
|
|
136
|
+
*
|
|
137
|
+
* Measured on this function before those guards landed, at a nominal 5 %:
|
|
138
|
+
* 14.60 % false promotion at n = 40 on a paired-binary noninferiority boundary,
|
|
139
|
+
* and 88.50 % at n = 6 under a bounded asymmetric null whose true mean paired
|
|
140
|
+
* delta is exactly 0.
|
|
141
|
+
*
|
|
142
|
+
* Continuous mean targets require bootstrap eligibility. Explicit median
|
|
143
|
+
* targets can use the exact sign test at its confidence-dependent minimum.
|
|
144
|
+
*/
|
|
145
|
+
function heldoutSignificance(paired, opts = {}) {
|
|
146
|
+
const deltaThreshold = opts.deltaThreshold ?? 0;
|
|
147
|
+
const confidence = opts.confidence ?? .95;
|
|
148
|
+
const resamples = opts.resamples ?? 2e3;
|
|
149
|
+
const seed = opts.seed ?? 1337;
|
|
150
|
+
const statistic = opts.statistic ?? "mean";
|
|
151
|
+
const observations = aggregatePairedHoldout(paired, opts.independentUnitByScenarioId);
|
|
152
|
+
const decision = decidePairedPromotion(observations.before, observations.after, {
|
|
153
|
+
confidence,
|
|
154
|
+
resamples,
|
|
155
|
+
statistic,
|
|
156
|
+
seed,
|
|
157
|
+
threshold: deltaThreshold,
|
|
158
|
+
minPairs: opts.minProductiveRuns
|
|
159
|
+
});
|
|
160
|
+
const bootstrap = decision.bootstrap ?? pairedBootstrap(observations.before, observations.after, {
|
|
161
|
+
confidence,
|
|
162
|
+
resamples,
|
|
163
|
+
statistic,
|
|
164
|
+
seed
|
|
165
|
+
});
|
|
166
|
+
const medianBootstrap = statistic === "median" ? bootstrap : pairedBootstrap(observations.before, observations.after, {
|
|
167
|
+
confidence,
|
|
168
|
+
resamples,
|
|
169
|
+
statistic: "median",
|
|
170
|
+
seed
|
|
171
|
+
});
|
|
172
|
+
const n = observations.before.length;
|
|
173
|
+
let ties = 0;
|
|
174
|
+
for (let i = 0; i < n; i += 1) {
|
|
175
|
+
const after = observations.after[i];
|
|
176
|
+
const before = observations.before[i];
|
|
177
|
+
if (Math.abs(after - before) < 1e-9) ties += 1;
|
|
178
|
+
}
|
|
179
|
+
const tieFraction = n === 0 ? 0 : ties / n;
|
|
180
|
+
return {
|
|
181
|
+
paired,
|
|
182
|
+
bootstrap,
|
|
183
|
+
medianBootstrap,
|
|
184
|
+
decision,
|
|
185
|
+
decisionStatistic: decision.statistic,
|
|
186
|
+
mcnemar: decision.mcnemar,
|
|
187
|
+
tieFraction,
|
|
188
|
+
n,
|
|
189
|
+
pairedCellN: paired.cellIds.length,
|
|
190
|
+
observationUnit: opts.independentUnitByScenarioId === void 0 ? "cell" : "registered",
|
|
191
|
+
unitIds: observations.unitIds,
|
|
192
|
+
minimumRequired: decision.minimumPairs,
|
|
193
|
+
decisionMethod: decision.method,
|
|
194
|
+
pValue: decision.pValue,
|
|
195
|
+
significant: decision.promote,
|
|
196
|
+
fewRuns: !decision.sufficient
|
|
197
|
+
};
|
|
198
|
+
}
|
|
199
|
+
/** Detect the native scale of a set of scores: 0-100 when any magnitude clears
|
|
200
|
+
* 1.5, else [0,1]. Used to auto-scale the regression tolerance so a default
|
|
201
|
+
* expressed for [0,1] is not silently a no-op on a 0-100 dimension. */
|
|
202
|
+
function detectScale(values) {
|
|
203
|
+
return values.some((v) => Math.abs(v) > 1.5) ? 100 : 1;
|
|
204
|
+
}
|
|
205
|
+
/**
|
|
206
|
+
* Report required-dimension evidence after full pairing and optional unit means.
|
|
207
|
+
* A bootstrap floor breach or a shared paired test supporting a drop marks
|
|
208
|
+
* regression. These two criteria are distinct; `ci` records the shared
|
|
209
|
+
* estimator and `bootstrap` records the floor interval. Missing observations
|
|
210
|
+
* and insufficient n remain explicit for the caller's evidence policy.
|
|
211
|
+
* The default tolerance is 0.05 on [0,1] and 5 on a detected 0-100 scale.
|
|
212
|
+
*/
|
|
213
|
+
function dimensionRegressions(candidate, baseline, scenarioIds, criticalDimensions, opts = {}) {
|
|
214
|
+
const out = [];
|
|
215
|
+
const expectedCellIds = [...baseline.keys()].filter((cellId) => scenarioIds.has(scenarioIdFromCellId(cellId))).sort();
|
|
216
|
+
for (const dim of criticalDimensions) {
|
|
217
|
+
const paired = pairHoldout(candidate, baseline, scenarioIds, (s) => s.dimensions[dim]);
|
|
218
|
+
if (paired.before.length === 0) continue;
|
|
219
|
+
const observations = aggregatePairedHoldout(paired, opts.independentUnitByScenarioId);
|
|
220
|
+
const measuredCells = new Set(paired.cellIds);
|
|
221
|
+
const measuredScenarios = new Set(paired.cellIds.map(scenarioIdFromCellId));
|
|
222
|
+
const tolerance = opts.tolerance ?? .05 * detectScale([...paired.before, ...paired.after]);
|
|
223
|
+
const bootstrapStatistic = opts.statistic ?? "mean";
|
|
224
|
+
const shared = {
|
|
225
|
+
confidence: opts.confidence ?? .95,
|
|
226
|
+
resamples: opts.resamples ?? 2e3,
|
|
227
|
+
statistic: bootstrapStatistic,
|
|
228
|
+
seed: opts.seed ?? 1337,
|
|
229
|
+
minPairs: opts.minProductiveRuns
|
|
230
|
+
};
|
|
231
|
+
const guard = decidePairedPromotion(observations.before, observations.after, shared);
|
|
232
|
+
const regression = decidePairedPromotion(observations.after, observations.before, {
|
|
233
|
+
...shared,
|
|
234
|
+
threshold: tolerance
|
|
235
|
+
});
|
|
236
|
+
const bootstrap = guard.bootstrap ?? pairedBootstrap(observations.before, observations.after, shared);
|
|
237
|
+
out.push({
|
|
238
|
+
dimension: dim,
|
|
239
|
+
bootstrap,
|
|
240
|
+
bootstrapStatistic,
|
|
241
|
+
ci: {
|
|
242
|
+
low: guard.low,
|
|
243
|
+
high: guard.high
|
|
244
|
+
},
|
|
245
|
+
decisionStatistic: guard.statistic,
|
|
246
|
+
mcnemar: guard.mcnemar,
|
|
247
|
+
indeterminate: guard.indeterminate,
|
|
248
|
+
regressed: bootstrap.low < -tolerance || regression.promote,
|
|
249
|
+
tolerance,
|
|
250
|
+
n: observations.before.length,
|
|
251
|
+
pairedCellN: paired.cellIds.length,
|
|
252
|
+
observationUnit: opts.independentUnitByScenarioId === void 0 ? "cell" : "registered",
|
|
253
|
+
minimumRequired: guard.minimumPairs,
|
|
254
|
+
fewRuns: !guard.sufficient,
|
|
255
|
+
missingCellIds: expectedCellIds.filter((cellId) => !measuredCells.has(cellId)),
|
|
256
|
+
missingScenarioIds: [...scenarioIds].filter((id) => !measuredScenarios.has(id)).sort()
|
|
257
|
+
});
|
|
258
|
+
}
|
|
259
|
+
return out;
|
|
260
|
+
}
|
|
261
|
+
//#endregion
|
|
9
262
|
//#region src/campaign/coverage.ts
|
|
10
263
|
/** Reject campaign designs whose denominator cannot be identified exactly. */
|
|
11
264
|
function assertCampaignDesign(scenarios, reps) {
|
|
@@ -401,457 +654,227 @@ function surfaceDispatchRef(surface, executionRef = "anonymous") {
|
|
|
401
654
|
return `surface:${executionRef}:${surfaceContentHash(surface)}`;
|
|
402
655
|
}
|
|
403
656
|
//#endregion
|
|
404
|
-
//#region src/
|
|
405
|
-
|
|
406
|
-
|
|
407
|
-
|
|
408
|
-
|
|
409
|
-
|
|
657
|
+
//#region src/experiment/claim.ts
|
|
658
|
+
const nonEmpty = z.string().min(1).refine((value) => value.trim() === value);
|
|
659
|
+
const claimSchema = z.object({
|
|
660
|
+
use: z.enum([
|
|
661
|
+
"development",
|
|
662
|
+
"comparison",
|
|
663
|
+
"certification"
|
|
664
|
+
]),
|
|
665
|
+
population: z.object({
|
|
666
|
+
id: nonEmpty,
|
|
667
|
+
description: nonEmpty
|
|
668
|
+
}).strict(),
|
|
669
|
+
samplingFrame: nonEmpty,
|
|
670
|
+
independentUnit: nonEmpty,
|
|
671
|
+
generalization: z.enum(["fixed-roster", "new-units"]),
|
|
672
|
+
minimumEffect: z.number().finite().positive().optional()
|
|
673
|
+
}).strict();
|
|
674
|
+
/** Validate a claim before binding it to a sealed experiment or final evidence. */
|
|
675
|
+
function defineEvaluationClaim(input) {
|
|
676
|
+
const parsed = claimSchema.safeParse(input);
|
|
677
|
+
if (!parsed.success) throw new ValidationError(`invalid evaluation claim: ${parsed.error.message}`);
|
|
678
|
+
return Object.freeze({
|
|
679
|
+
...parsed.data,
|
|
680
|
+
population: Object.freeze(parsed.data.population)
|
|
681
|
+
});
|
|
410
682
|
}
|
|
411
|
-
/**
|
|
412
|
-
|
|
413
|
-
|
|
414
|
-
|
|
415
|
-
|
|
416
|
-
|
|
417
|
-
|
|
418
|
-
|
|
419
|
-
|
|
420
|
-
|
|
421
|
-
*
|
|
422
|
-
* When every paired delta is identical the resample distribution is a point
|
|
423
|
-
* mass and the interval collapses: `[0, 0]` when all pairs tie, `[g, g]` on n
|
|
424
|
-
* identical deltas of g. Neither says the effect is certain — both say the
|
|
425
|
-
* sample carries no information about how far the estimate could be wrong, and
|
|
426
|
-
* `low > threshold` then answers on the point estimate alone. It fails in both
|
|
427
|
-
* directions: `[0, 0]` clears every NEGATIVE threshold, which is how a
|
|
428
|
-
* tie-dominated pass/fail comparison laundered a regression into a
|
|
429
|
-
* noninferiority pass, and `[g, g]` clears every threshold below g with no
|
|
430
|
-
* spread behind it. Under a bounded asymmetric null whose true mean paired
|
|
431
|
-
* delta is exactly 0 — 2 % of pairs dropping by 1.0, the rest gaining 0.0204 —
|
|
432
|
-
* every sample that misses the drop is exactly that shape, and deciding on
|
|
433
|
-
* `low > 0` promoted 65.65 % of samples at n = 20 against a nominal 5 %.
|
|
434
|
-
*
|
|
435
|
-
* So `indeterminate` is reported and `significant` is false whenever the
|
|
436
|
-
* interval has zero width, on BOTH paths: at small n the exact sign test is a
|
|
437
|
-
* test of the MEDIAN and a zero-spread sample is precisely where it stops
|
|
438
|
-
* saying anything about the mean the caller is thresholding.
|
|
439
|
-
*
|
|
440
|
-
* `threshold` may be negative — that is a noninferiority margin, and it is the
|
|
441
|
-
* regime the zero-width hole is worst in. For a two-point (pass/fail) outcome
|
|
442
|
-
* the percentile bootstrap is not a valid interval at a nonzero margin at all;
|
|
443
|
-
* use {@link decidePairedPromotion}, which routes those to Tango's score
|
|
444
|
-
* interval, rather than thresholding this function's bootstrap directly.
|
|
445
|
-
*/
|
|
446
|
-
function pairedDeltaTest(before, after, options = {}) {
|
|
447
|
-
const threshold = options.threshold ?? 0;
|
|
448
|
-
if (!Number.isFinite(threshold)) throw new Error(`pairedDeltaTest: threshold must be finite, got ${threshold}`);
|
|
449
|
-
const exactMinimum = minimumPairsForPairedDeltaTest(options.confidence ?? .95);
|
|
450
|
-
const requestedMinimum = options.minPairs ?? exactMinimum;
|
|
451
|
-
if (!Number.isInteger(requestedMinimum) || requestedMinimum < 1) throw new Error(`pairedDeltaTest: minPairs must be a positive integer, got ${requestedMinimum}`);
|
|
452
|
-
const minimumPairs = Math.max(requestedMinimum, exactMinimum);
|
|
453
|
-
const bootstrap = pairedBootstrap(before, after, options);
|
|
454
|
-
const sufficient = bootstrap.n >= minimumPairs;
|
|
455
|
-
const indeterminate = bootstrap.n > 0 && (!Number.isFinite(bootstrap.low) || !Number.isFinite(bootstrap.high) || bootstrap.low === bootstrap.high);
|
|
456
|
-
if (bootstrap.gateEligible) return {
|
|
457
|
-
bootstrap,
|
|
458
|
-
method: "bootstrap-ci",
|
|
459
|
-
pValue: null,
|
|
460
|
-
minimumPairs,
|
|
461
|
-
sufficient,
|
|
462
|
-
indeterminate,
|
|
463
|
-
significant: sufficient && !indeterminate && bootstrap.low > threshold
|
|
464
|
-
};
|
|
465
|
-
const exact = pairedSignTest(before.map((value, index) => after[index] - value - threshold), "greater");
|
|
466
|
-
const estimate = options.statistic === "mean" ? bootstrap.mean : bootstrap.median;
|
|
683
|
+
/** Repetitions retain their denominator without becoming additional independent units. */
|
|
684
|
+
function summarizeEvaluationUnits(claim, rows) {
|
|
685
|
+
const validated = defineEvaluationClaim(claim);
|
|
686
|
+
const counts = /* @__PURE__ */ new Map();
|
|
687
|
+
for (const [index, row] of rows.entries()) {
|
|
688
|
+
if (row === null || typeof row !== "object" || Array.isArray(row)) throw new ValidationError(`evaluation claim: row ${index} must be an object`);
|
|
689
|
+
const value = readField(row, validated.independentUnit);
|
|
690
|
+
if (typeof value !== "string" || !value.trim() || value.trim() !== value) throw new ValidationError(`evaluation claim: row ${index} needs a nonempty string at '${validated.independentUnit}'`);
|
|
691
|
+
counts.set(value, (counts.get(value) ?? 0) + 1);
|
|
692
|
+
}
|
|
467
693
|
return {
|
|
468
|
-
|
|
469
|
-
|
|
470
|
-
|
|
471
|
-
|
|
472
|
-
|
|
473
|
-
|
|
474
|
-
significant: sufficient && !indeterminate && estimate > threshold && exact.pValue <= (1 - bootstrap.confidence) / 2
|
|
694
|
+
observations: rows.length,
|
|
695
|
+
independentUnits: counts.size,
|
|
696
|
+
units: [...counts].sort(([left], [right]) => compareCodeUnits(left, right)).map(([id, observations]) => ({
|
|
697
|
+
id,
|
|
698
|
+
observations
|
|
699
|
+
}))
|
|
475
700
|
};
|
|
476
701
|
}
|
|
477
702
|
//#endregion
|
|
478
|
-
//#region src/
|
|
479
|
-
|
|
480
|
-
|
|
481
|
-
|
|
482
|
-
|
|
483
|
-
|
|
484
|
-
|
|
485
|
-
|
|
486
|
-
|
|
487
|
-
|
|
488
|
-
|
|
489
|
-
|
|
490
|
-
|
|
491
|
-
|
|
492
|
-
|
|
493
|
-
|
|
494
|
-
|
|
495
|
-
|
|
496
|
-
|
|
497
|
-
|
|
498
|
-
|
|
499
|
-
|
|
500
|
-
|
|
501
|
-
|
|
502
|
-
|
|
503
|
-
|
|
504
|
-
|
|
505
|
-
|
|
506
|
-
|
|
507
|
-
|
|
508
|
-
|
|
509
|
-
|
|
510
|
-
|
|
511
|
-
|
|
512
|
-
|
|
513
|
-
|
|
514
|
-
|
|
515
|
-
|
|
516
|
-
|
|
517
|
-
* of evidence. Measured on the composable gate before this change, under a
|
|
518
|
-
* bounded asymmetric null whose true mean paired delta is exactly 0: 88.50 %
|
|
519
|
-
* false promotion at n = 6 and 65.65 % at n = 20 against a nominal 5 %.
|
|
520
|
-
*
|
|
521
|
-
* Orthogonal to the small-sample switch inside {@link pairedDeltaTest}: that
|
|
522
|
-
* picks the TEST from the sample size (bootstrap CI at n >= 20, pre-registered
|
|
523
|
-
* exact sign test below it), this picks the ESTIMATOR from the outcome's shape.
|
|
524
|
-
* Both are needed — an exact sign test applied to a tie-pinned median is still
|
|
525
|
-
* blind, and a mean bootstrap CI at n = 6 is still not a valid test.
|
|
526
|
-
*/
|
|
527
|
-
/**
|
|
528
|
-
* Which estimator {@link decidePairedPromotion} would use on this data, and the
|
|
529
|
-
* shape facts behind it — for callers that must report the shape on a path
|
|
530
|
-
* where no interval is computed at all (an early rejection, or zero pairs).
|
|
531
|
-
* Cheap: no bootstrap, no interval.
|
|
532
|
-
*/
|
|
533
|
-
function pairedDecisionShape(before, after, statistic = "mean") {
|
|
534
|
-
const tieFraction = before.length === 0 ? null : pairedDeltaTieFraction(before, after);
|
|
535
|
-
if (statistic === "median") return {
|
|
536
|
-
statistic: "median_bootstrap",
|
|
537
|
-
binaryScale: null,
|
|
538
|
-
tieFraction
|
|
539
|
-
};
|
|
540
|
-
const binaryScale = pairedBinaryScale(before, after);
|
|
541
|
-
if (binaryScale !== null) return {
|
|
542
|
-
statistic: "paired_risk_difference",
|
|
543
|
-
binaryScale,
|
|
544
|
-
tieFraction
|
|
545
|
-
};
|
|
546
|
-
return {
|
|
547
|
-
statistic: "mean_bootstrap",
|
|
548
|
-
binaryScale: null,
|
|
549
|
-
tieFraction
|
|
550
|
-
};
|
|
551
|
-
}
|
|
552
|
-
/**
|
|
553
|
-
* Decide whether a paired candidate-minus-baseline delta clears a promotion
|
|
554
|
-
* threshold. `before` is the baseline arm, `after` the candidate arm, paired by
|
|
555
|
-
* position. Throws on unequal lengths.
|
|
556
|
-
*/
|
|
557
|
-
function decidePairedPromotion(before, after, options = {}) {
|
|
558
|
-
if (before.length !== after.length) throw new Error(`decidePairedPromotion: unequal sample sizes (${before.length} vs ${after.length})`);
|
|
559
|
-
const threshold = options.threshold ?? 0;
|
|
560
|
-
if (!Number.isFinite(threshold)) throw new Error(`decidePairedPromotion: threshold must be finite, got ${threshold}`);
|
|
561
|
-
const confidence = options.confidence ?? .95;
|
|
562
|
-
const exactMinimum = minimumPairsForPairedDeltaTest(confidence);
|
|
563
|
-
const requestedMinimum = options.minPairs ?? exactMinimum;
|
|
564
|
-
if (!Number.isInteger(requestedMinimum) || requestedMinimum < 1) throw new Error(`decidePairedPromotion: minPairs must be a positive integer, got ${requestedMinimum}`);
|
|
565
|
-
const minimumPairs = Math.max(requestedMinimum, exactMinimum);
|
|
566
|
-
const n = before.length;
|
|
567
|
-
const sufficient = n >= minimumPairs;
|
|
568
|
-
const { binaryScale, tieFraction } = pairedDecisionShape(before, after, options.statistic);
|
|
569
|
-
let core;
|
|
570
|
-
if (binaryScale !== null) {
|
|
571
|
-
const unitControl = before.map((v) => v / binaryScale);
|
|
572
|
-
const unitTreatment = after.map((v) => v / binaryScale);
|
|
573
|
-
const exact = pairedRiskDifferenceExact(unitControl, unitTreatment, confidence);
|
|
574
|
-
const score = pairedRiskDifferenceScore(unitControl, unitTreatment, confidence);
|
|
575
|
-
const low = score.lower * binaryScale;
|
|
576
|
-
core = {
|
|
577
|
-
statistic: "paired_risk_difference",
|
|
578
|
-
method: "score-interval",
|
|
579
|
-
delta: score.riskDifference * binaryScale,
|
|
580
|
-
low,
|
|
581
|
-
high: score.upper * binaryScale,
|
|
582
|
-
bootstrap: null,
|
|
583
|
-
mcnemar: {
|
|
584
|
-
b: exact.b,
|
|
585
|
-
c: exact.c,
|
|
586
|
-
nDiscordant: exact.nDiscordant,
|
|
587
|
-
pValue: exact.pValue
|
|
588
|
-
},
|
|
589
|
-
pValue: null,
|
|
590
|
-
clearsThreshold: low > threshold,
|
|
591
|
-
label: "success-rate",
|
|
592
|
-
methodDetail: ""
|
|
593
|
-
};
|
|
594
|
-
} else {
|
|
595
|
-
const bootstrapStatistic = options.statistic === "median" ? "median" : "mean";
|
|
596
|
-
const test = pairedDeltaTest(before, after, {
|
|
597
|
-
confidence,
|
|
598
|
-
resamples: options.resamples,
|
|
599
|
-
statistic: bootstrapStatistic,
|
|
600
|
-
seed: options.seed,
|
|
601
|
-
threshold,
|
|
602
|
-
minPairs: options.minPairs
|
|
603
|
-
});
|
|
604
|
-
const ci = test.bootstrap;
|
|
605
|
-
core = {
|
|
606
|
-
statistic: bootstrapStatistic === "mean" ? "mean_bootstrap" : "median_bootstrap",
|
|
607
|
-
method: test.method,
|
|
608
|
-
delta: bootstrapStatistic === "mean" ? ci.mean : ci.median,
|
|
609
|
-
low: ci.low,
|
|
610
|
-
high: ci.high,
|
|
611
|
-
bootstrap: ci,
|
|
612
|
-
mcnemar: null,
|
|
613
|
-
pValue: test.pValue,
|
|
614
|
-
clearsThreshold: test.significant,
|
|
615
|
-
label: bootstrapStatistic,
|
|
616
|
-
methodDetail: test.method === "exact-sign" ? ` Below ${test.minimumPairs} pairs the interval is descriptive only; the decision is the exact one-sided sign test, p=${fmt(test.pValue ?? 1)}.` : ""
|
|
617
|
-
};
|
|
703
|
+
//#region src/experiment/final-evidence.ts
|
|
704
|
+
const identity = z.string().min(1).refine((value) => value.trim() === value);
|
|
705
|
+
const digest = z.string().regex(/^sha256:[a-f0-9]{64}$/);
|
|
706
|
+
const reservationSchema = z.object({
|
|
707
|
+
requestId: identity,
|
|
708
|
+
claimDigest: digest,
|
|
709
|
+
populationId: identity,
|
|
710
|
+
inputDigest: digest,
|
|
711
|
+
unitIds: z.array(identity).min(1)
|
|
712
|
+
}).strict();
|
|
713
|
+
const measurementSchema = z.object({
|
|
714
|
+
evaluatorDigest: digest,
|
|
715
|
+
candidateDigests: z.array(digest).min(1)
|
|
716
|
+
}).strict();
|
|
717
|
+
const eventSchema = z.discriminatedUnion("kind", [z.object({
|
|
718
|
+
kind: z.literal("reserved"),
|
|
719
|
+
eventId: identity,
|
|
720
|
+
reservation: reservationSchema
|
|
721
|
+
}).strict(), z.object({
|
|
722
|
+
kind: z.literal("exposed"),
|
|
723
|
+
eventId: identity,
|
|
724
|
+
requestId: identity,
|
|
725
|
+
measurement: measurementSchema
|
|
726
|
+
}).strict()]);
|
|
727
|
+
const schema = "agent-eval.final-evidence.v1";
|
|
728
|
+
const entrySchema = z.object({
|
|
729
|
+
schema: z.literal(schema),
|
|
730
|
+
sequence: z.number().int().nonnegative(),
|
|
731
|
+
previousHash: digest.nullable(),
|
|
732
|
+
event: eventSchema,
|
|
733
|
+
entryHash: digest
|
|
734
|
+
}).strict();
|
|
735
|
+
/** Retains the distinction between invalid input, consumed evidence, and unavailable storage. */
|
|
736
|
+
var FinalEvidenceError = class extends Error {
|
|
737
|
+
kind;
|
|
738
|
+
constructor(kind, message) {
|
|
739
|
+
super(message);
|
|
740
|
+
this.kind = kind;
|
|
741
|
+
this.name = "FinalEvidenceError";
|
|
618
742
|
}
|
|
619
|
-
|
|
620
|
-
|
|
621
|
-
|
|
743
|
+
};
|
|
744
|
+
/** A consumed or conflicting dataset cannot authorize another adaptive decision. */
|
|
745
|
+
var FinalEvidenceConflictError = class extends FinalEvidenceError {
|
|
746
|
+
constructor(message) {
|
|
747
|
+
super("conflict", message);
|
|
748
|
+
this.name = "FinalEvidenceConflictError";
|
|
749
|
+
}
|
|
750
|
+
};
|
|
751
|
+
function invalid(message) {
|
|
752
|
+
return /* @__PURE__ */ new TypeError(`final evidence: ${message}`);
|
|
753
|
+
}
|
|
754
|
+
function normalizeReservation(input) {
|
|
755
|
+
const parsed = reservationSchema.parse(input);
|
|
756
|
+
if (new Set(parsed.unitIds).size !== parsed.unitIds.length) throw invalid("unitIds must be unique independent source identities");
|
|
622
757
|
return {
|
|
623
|
-
|
|
624
|
-
|
|
625
|
-
confidence,
|
|
626
|
-
binaryScale,
|
|
627
|
-
tieFraction,
|
|
628
|
-
minimumPairs,
|
|
629
|
-
sufficient,
|
|
630
|
-
indeterminate,
|
|
631
|
-
indeterminateCause,
|
|
632
|
-
exactTestVetoes,
|
|
633
|
-
promote: sufficient && !indeterminate && core.clearsThreshold && !exactTestVetoes,
|
|
634
|
-
...core
|
|
758
|
+
...parsed,
|
|
759
|
+
unitIds: [...parsed.unitIds].sort(compareCodeUnits)
|
|
635
760
|
};
|
|
636
761
|
}
|
|
637
|
-
function
|
|
638
|
-
|
|
639
|
-
}
|
|
640
|
-
//#endregion
|
|
641
|
-
//#region src/campaign/gates/statistical-heldout.ts
|
|
642
|
-
/**
|
|
643
|
-
* Statistical held-out promotion machinery — the trustworthy core the
|
|
644
|
-
* point-estimate `heldout-delta` gate lacked.
|
|
645
|
-
*
|
|
646
|
-
* The shipped false positive it prevents: a winner re-scored against the
|
|
647
|
-
* baseline on the holdout read run-to-run model NOISE (e.g. 91 vs 95) as a
|
|
648
|
-
* "+4 lift" and shipped, because the gate compared point estimates with no
|
|
649
|
-
* confidence interval. Here we pair candidate vs baseline holdout observations
|
|
650
|
-
* and bootstrap a CI on the paired delta — a candidate ships only when the CI
|
|
651
|
-
* lower bound clears the effect-size threshold (the gain is real at the
|
|
652
|
-
* confidence level, not noise), and is blocked when a critical dimension
|
|
653
|
-
* (e.g. `hallucination_free` for a legal agent) significantly regresses even if
|
|
654
|
-
* the net composite rose (anti-Goodhart).
|
|
655
|
-
*
|
|
656
|
-
* Two traps this module is built around (both produce a NEW false positive if
|
|
657
|
-
* gotten wrong):
|
|
658
|
-
* 1. PAIRING GRANULARITY — pairs by FULL `cellId` (`scenario:rep`), never by
|
|
659
|
-
* `scenarioId` (which averages reps away and destroys the within-pair
|
|
660
|
-
* variance reduction that makes a paired bootstrap tighter than unpaired).
|
|
661
|
-
* One paired observation per cell ⇒ reps multiply n.
|
|
662
|
-
* 2. SCALE — a judge may emit composites/dimensions on [0,1] or 0-100. The
|
|
663
|
-
* threshold + tolerance are interpreted in the judge's NATIVE scale; the
|
|
664
|
-
* per-dimension tolerance auto-scales off the observed baseline magnitudes
|
|
665
|
-
* so `-0.10` on [0,1] doesn't silently become a no-op on a 0-100 dimension.
|
|
666
|
-
*/
|
|
667
|
-
/** Tie fraction at/above which a gate annotates its verdict with the tie share.
|
|
668
|
-
* Tie-domination of the median bites structurally at >= 0.5 (the median is then
|
|
669
|
-
* 0 by construction); 0.4 is a softer warn threshold that flags a run APPROACHING
|
|
670
|
-
* that regime, so an operator sees it before the median goes fully blind. */
|
|
671
|
-
const TIE_WARN_FRACTION = .4;
|
|
672
|
-
/**
|
|
673
|
-
* Pair candidate vs baseline holdout observations by FULL cellId. `select`
|
|
674
|
-
* pulls the scalar from a cell's judge reports (composite, or a named
|
|
675
|
-
* dimension); a cell contributes the mean of `select` across its judges. Cells
|
|
676
|
-
* whose scenario is not in `scenarioIds`, or where `select` is undefined for
|
|
677
|
-
* every judge on either side, are skipped on BOTH sides so the arrays stay
|
|
678
|
-
* paired. Throws when the two maps disagree on which holdout cells exist — a
|
|
679
|
-
* load-bearing invariant: the baseline + winner holdout campaigns run the same
|
|
680
|
-
* scenarios with the same seed base, so their cellIds MUST align; a mismatch
|
|
681
|
-
* means a silent pairing bug, not a soft fallback.
|
|
682
|
-
*/
|
|
683
|
-
function pairHoldout(candidate, baseline, scenarioIds, select) {
|
|
684
|
-
const cellValue = (byCell, cellId) => {
|
|
685
|
-
const scores = byCell.get(cellId);
|
|
686
|
-
if (!scores) return void 0;
|
|
687
|
-
const vals = [];
|
|
688
|
-
for (const s of Object.values(scores)) {
|
|
689
|
-
if (s.failed === true) throw new Error(`pairHoldout: cell '${cellId}' contains a failed judge score`);
|
|
690
|
-
const v = select(s);
|
|
691
|
-
if (typeof v === "number" && !Number.isFinite(v)) throw new Error(`pairHoldout: cell '${cellId}' contains a non-finite selected score`);
|
|
692
|
-
if (typeof v === "number") vals.push(v);
|
|
693
|
-
}
|
|
694
|
-
if (vals.length === 0) return void 0;
|
|
695
|
-
return vals.reduce((a, b) => a + b, 0) / vals.length;
|
|
696
|
-
};
|
|
697
|
-
const inScope = (cellId) => scenarioIds.has(cellId.split(":")[0] ?? "");
|
|
698
|
-
const candCells = [...candidate.keys()].filter(inScope).sort();
|
|
699
|
-
const baseCells = [...baseline.keys()].filter(inScope).sort();
|
|
700
|
-
if (candCells.length !== baseCells.length || candCells.some((c, i) => c !== baseCells[i])) throw new Error(`pairHoldout: candidate/baseline holdout cells do not align — candidate=[${candCells.join(",")}] baseline=[${baseCells.join(",")}]. Both holdout campaigns must run the same scenarios with the same seed base.`);
|
|
701
|
-
const before = [];
|
|
702
|
-
const after = [];
|
|
703
|
-
const cellIds = [];
|
|
704
|
-
for (const cellId of candCells) {
|
|
705
|
-
const b = cellValue(baseline, cellId);
|
|
706
|
-
const a = cellValue(candidate, cellId);
|
|
707
|
-
if (b === void 0 && a === void 0) continue;
|
|
708
|
-
if (b === void 0 || a === void 0) throw new Error(`pairHoldout: cell '${cellId}' has a selected score on only one arm`);
|
|
709
|
-
before.push(b);
|
|
710
|
-
after.push(a);
|
|
711
|
-
cellIds.push(cellId);
|
|
712
|
-
}
|
|
762
|
+
function normalizeMeasurement(input) {
|
|
763
|
+
const parsed = measurementSchema.parse(input);
|
|
713
764
|
return {
|
|
714
|
-
|
|
715
|
-
|
|
716
|
-
cellIds
|
|
765
|
+
...parsed,
|
|
766
|
+
candidateDigests: [...new Set(parsed.candidateDigests)].sort(compareCodeUnits)
|
|
717
767
|
};
|
|
718
768
|
}
|
|
719
|
-
|
|
720
|
-
* Significance of the held-out composite lift: ship only when the lower bound
|
|
721
|
-
* of the interval the outcome's shape ADMITS exceeds `deltaThreshold` (default
|
|
722
|
-
* 0 ⇒ "confidently positive"). Interpret `deltaThreshold` in the judge's native
|
|
723
|
-
* scale.
|
|
724
|
-
*
|
|
725
|
-
* The decision is delegated whole to {@link decidePairedPromotion}, the one
|
|
726
|
-
* copy of the rule (`src/paired-promotion-decision.ts`), which `HeldOutGate`
|
|
727
|
-
* also calls. That module's header carries the measurements; the short version
|
|
728
|
-
* is three guards a bare `bootstrap.low > threshold` does not have:
|
|
729
|
-
*
|
|
730
|
-
* - a two-point (pass/fail) outcome decides on Tango's SCORE interval, the
|
|
731
|
-
* only paired-binary construction that stays valid at a nonzero margin;
|
|
732
|
-
* - McNemar's exact test VETOES at any non-negative threshold;
|
|
733
|
-
* - a ZERO-WIDTH interval is refused rather than promoted, in either
|
|
734
|
-
* direction — [0,0] clears every negative threshold and [g,g] clears every
|
|
735
|
-
* threshold below g, and both are an absence of evidence, not a result.
|
|
736
|
-
*
|
|
737
|
-
* Measured on this function before those guards landed, at a nominal 5 %:
|
|
738
|
-
* 14.60 % false promotion at n = 40 on a paired-binary noninferiority boundary,
|
|
739
|
-
* and 88.50 % at n = 6 under a bounded asymmetric null whose true mean paired
|
|
740
|
-
* delta is exactly 0.
|
|
741
|
-
*
|
|
742
|
-
* At small n, where the percentile bootstrap is descriptive only, a
|
|
743
|
-
* pre-registered exact sign test still carries the bootstrap path.
|
|
744
|
-
*/
|
|
745
|
-
function heldoutSignificance(paired, opts = {}) {
|
|
746
|
-
const deltaThreshold = opts.deltaThreshold ?? 0;
|
|
747
|
-
const confidence = opts.confidence ?? .95;
|
|
748
|
-
const resamples = opts.resamples ?? 2e3;
|
|
749
|
-
const seed = opts.seed ?? 1337;
|
|
750
|
-
const statistic = opts.statistic ?? "mean";
|
|
751
|
-
const decision = decidePairedPromotion(paired.before, paired.after, {
|
|
752
|
-
confidence,
|
|
753
|
-
resamples,
|
|
754
|
-
statistic,
|
|
755
|
-
seed,
|
|
756
|
-
threshold: deltaThreshold,
|
|
757
|
-
minPairs: opts.minProductiveRuns
|
|
758
|
-
});
|
|
759
|
-
const bootstrap = decision.bootstrap ?? pairedBootstrap(paired.before, paired.after, {
|
|
760
|
-
confidence,
|
|
761
|
-
resamples,
|
|
762
|
-
statistic,
|
|
763
|
-
seed
|
|
764
|
-
});
|
|
765
|
-
const medianBootstrap = statistic === "median" ? bootstrap : pairedBootstrap(paired.before, paired.after, {
|
|
766
|
-
confidence,
|
|
767
|
-
resamples,
|
|
768
|
-
statistic: "median",
|
|
769
|
-
seed
|
|
770
|
-
});
|
|
771
|
-
const n = paired.before.length;
|
|
772
|
-
let ties = 0;
|
|
773
|
-
for (let i = 0; i < n; i += 1) {
|
|
774
|
-
const after = paired.after[i] ?? 0;
|
|
775
|
-
const before = paired.before[i] ?? 0;
|
|
776
|
-
if (Math.abs(after - before) < 1e-9) ties += 1;
|
|
777
|
-
}
|
|
778
|
-
const tieFraction = n === 0 ? 0 : ties / n;
|
|
769
|
+
function codec() {
|
|
779
770
|
return {
|
|
780
|
-
|
|
781
|
-
|
|
782
|
-
|
|
783
|
-
|
|
784
|
-
|
|
785
|
-
|
|
786
|
-
|
|
787
|
-
|
|
788
|
-
|
|
789
|
-
|
|
790
|
-
|
|
791
|
-
|
|
792
|
-
|
|
771
|
+
subject: "final evidence ledger",
|
|
772
|
+
header: { schema },
|
|
773
|
+
integrityError: (message, options) => new Error(message, options),
|
|
774
|
+
conflictError: (message) => new FinalEvidenceConflictError(message),
|
|
775
|
+
parseEntry: (raw, context) => {
|
|
776
|
+
const decoded = entrySchema.safeParse(raw);
|
|
777
|
+
if (!decoded.success) throw new Error(`final evidence ledger ${context.path}:${context.line} is invalid: ${decoded.error.message}`);
|
|
778
|
+
const parsed = decoded.data;
|
|
779
|
+
return {
|
|
780
|
+
...parsed,
|
|
781
|
+
previousHash: parsed.previousHash,
|
|
782
|
+
entryHash: parsed.entryHash
|
|
783
|
+
};
|
|
784
|
+
},
|
|
785
|
+
checkEntryHeader: (entry) => {
|
|
786
|
+
if (entry.schema !== schema) throw invalid("unsupported ledger schema");
|
|
787
|
+
},
|
|
788
|
+
createProjector: () => {
|
|
789
|
+
const records = /* @__PURE__ */ new Map();
|
|
790
|
+
const owners = /* @__PURE__ */ new Map();
|
|
791
|
+
const inputOwners = /* @__PURE__ */ new Map();
|
|
792
|
+
return {
|
|
793
|
+
apply: (entry) => {
|
|
794
|
+
const event = entry.event;
|
|
795
|
+
if (event.kind === "reserved") {
|
|
796
|
+
const reservation = normalizeReservation(event.reservation);
|
|
797
|
+
if (event.eventId !== `reserve:${reservation.requestId}` || records.has(reservation.requestId)) throw invalid("invalid or duplicate reservation identity");
|
|
798
|
+
const inputOwner = inputOwners.get(reservation.inputDigest);
|
|
799
|
+
if (inputOwner !== void 0) throw new FinalEvidenceConflictError(`final input is already reserved by '${inputOwner}'`);
|
|
800
|
+
for (const unitId of reservation.unitIds) {
|
|
801
|
+
const owner = owners.get(unitId);
|
|
802
|
+
if (owner !== void 0) throw new FinalEvidenceConflictError(`final unit '${unitId}' is already reserved by '${owner}'`);
|
|
803
|
+
owners.set(unitId, reservation.requestId);
|
|
804
|
+
}
|
|
805
|
+
inputOwners.set(reservation.inputDigest, reservation.requestId);
|
|
806
|
+
records.set(reservation.requestId, {
|
|
807
|
+
reservation,
|
|
808
|
+
reservationHash: entry.entryHash,
|
|
809
|
+
exposure: null
|
|
810
|
+
});
|
|
811
|
+
} else {
|
|
812
|
+
const record = records.get(event.requestId);
|
|
813
|
+
if (event.eventId !== `expose:${event.requestId}` || !record || record.exposure !== null) throw invalid("exposure requires one unexposed reservation");
|
|
814
|
+
record.exposure = {
|
|
815
|
+
measurement: normalizeMeasurement(event.measurement),
|
|
816
|
+
entryHash: entry.entryHash
|
|
817
|
+
};
|
|
818
|
+
}
|
|
819
|
+
},
|
|
820
|
+
finish: () => [...records.values()]
|
|
821
|
+
};
|
|
822
|
+
}
|
|
793
823
|
};
|
|
794
824
|
}
|
|
795
|
-
|
|
796
|
-
|
|
797
|
-
|
|
798
|
-
|
|
799
|
-
|
|
800
|
-
}
|
|
801
|
-
|
|
802
|
-
|
|
803
|
-
|
|
804
|
-
|
|
805
|
-
|
|
806
|
-
|
|
807
|
-
|
|
808
|
-
* The interval comes from {@link decidePairedPromotion}, so a pass/fail
|
|
809
|
-
* dimension is judged on Tango's score interval rather than a percentile
|
|
810
|
-
* bootstrap of the mean — `tolerance` is a NONZERO margin, and the bootstrap
|
|
811
|
-
* is not a valid interval at one. That matters most here because this guard
|
|
812
|
-
* fails OPEN by construction: `tolerance` is positive, so an interval pinned at
|
|
813
|
-
* [0,0] never satisfies `low < −tolerance` and a real regression on a safety
|
|
814
|
-
* dimension would be reported as `regressed: false`. On the median it fails the
|
|
815
|
-
* same way for the same reason — when most pairs tie, which is automatic for a
|
|
816
|
-
* pass/fail dimension on {0,1} and on the 0-100 encoding `detectScale` exists
|
|
817
|
-
* to support, the median CI collapses to [0,0]. Pass `statistic: 'median'` to
|
|
818
|
-
* restore the pre-0.134 behaviour. */
|
|
819
|
-
function dimensionRegressions(candidate, baseline, scenarioIds, criticalDimensions, opts = {}) {
|
|
820
|
-
const out = [];
|
|
821
|
-
for (const dim of criticalDimensions) {
|
|
822
|
-
const paired = pairHoldout(candidate, baseline, scenarioIds, (s) => s.dimensions[dim]);
|
|
823
|
-
if (paired.before.length === 0) continue;
|
|
824
|
-
const tolerance = opts.tolerance ?? .05 * detectScale([...paired.before, ...paired.after]);
|
|
825
|
-
const bootstrapStatistic = opts.statistic ?? "mean";
|
|
826
|
-
const shared = {
|
|
827
|
-
confidence: opts.confidence ?? .95,
|
|
828
|
-
resamples: opts.resamples ?? 2e3,
|
|
829
|
-
statistic: bootstrapStatistic,
|
|
830
|
-
seed: opts.seed ?? 1337
|
|
825
|
+
async function outcome(operation) {
|
|
826
|
+
try {
|
|
827
|
+
return {
|
|
828
|
+
succeeded: true,
|
|
829
|
+
value: await operation()
|
|
830
|
+
};
|
|
831
|
+
} catch (error) {
|
|
832
|
+
return {
|
|
833
|
+
succeeded: false,
|
|
834
|
+
error: {
|
|
835
|
+
kind: error instanceof FinalEvidenceConflictError ? "conflict" : error instanceof TypeError || error instanceof z.ZodError ? "invalid" : "unavailable",
|
|
836
|
+
message: error instanceof Error ? error.message : String(error)
|
|
837
|
+
}
|
|
831
838
|
};
|
|
832
|
-
const guard = decidePairedPromotion(paired.before, paired.after, shared);
|
|
833
|
-
const regression = decidePairedPromotion(paired.after, paired.before, {
|
|
834
|
-
...shared,
|
|
835
|
-
threshold: tolerance
|
|
836
|
-
});
|
|
837
|
-
const bootstrap = guard.bootstrap ?? pairedBootstrap(paired.before, paired.after, shared);
|
|
838
|
-
out.push({
|
|
839
|
-
dimension: dim,
|
|
840
|
-
bootstrap,
|
|
841
|
-
bootstrapStatistic,
|
|
842
|
-
ci: {
|
|
843
|
-
low: guard.low,
|
|
844
|
-
high: guard.high
|
|
845
|
-
},
|
|
846
|
-
decisionStatistic: guard.statistic,
|
|
847
|
-
mcnemar: guard.mcnemar,
|
|
848
|
-
indeterminate: guard.indeterminate,
|
|
849
|
-
regressed: bootstrap.low < -tolerance || regression.promote,
|
|
850
|
-
tolerance,
|
|
851
|
-
n: paired.before.length
|
|
852
|
-
});
|
|
853
839
|
}
|
|
854
|
-
|
|
840
|
+
}
|
|
841
|
+
/** Uses the shared locked journal and requires its trusted head on every reopen. */
|
|
842
|
+
function openFinalEvidenceLedger(options) {
|
|
843
|
+
if (!options.path.trim()) throw invalid("path is empty");
|
|
844
|
+
const journal = new FileLedgerJournal(options.path, codec(), { requireTrustedHead: true });
|
|
845
|
+
return {
|
|
846
|
+
reserve: (input) => outcome(async () => {
|
|
847
|
+
const reservation = normalizeReservation(input);
|
|
848
|
+
const result = await journal.append({
|
|
849
|
+
kind: "reserved",
|
|
850
|
+
eventId: `reserve:${reservation.requestId}`,
|
|
851
|
+
reservation
|
|
852
|
+
}, { pinHead: true });
|
|
853
|
+
const record = result.projection.find((item) => item.reservation.requestId === reservation.requestId);
|
|
854
|
+
if (!record) throw invalid("reserved record is missing after append");
|
|
855
|
+
return {
|
|
856
|
+
record,
|
|
857
|
+
replayed: !result.appended
|
|
858
|
+
};
|
|
859
|
+
}),
|
|
860
|
+
expose: (requestId, input) => outcome(async () => {
|
|
861
|
+
identity.parse(requestId);
|
|
862
|
+
const measurement = normalizeMeasurement(input);
|
|
863
|
+
const result = await journal.append({
|
|
864
|
+
kind: "exposed",
|
|
865
|
+
eventId: `expose:${requestId}`,
|
|
866
|
+
requestId,
|
|
867
|
+
measurement
|
|
868
|
+
}, { pinHead: true });
|
|
869
|
+
const record = result.projection.find((item) => item.reservation.requestId === requestId);
|
|
870
|
+
if (!record) throw invalid("exposed record is missing after append");
|
|
871
|
+
return {
|
|
872
|
+
record,
|
|
873
|
+
replayed: !result.appended
|
|
874
|
+
};
|
|
875
|
+
}),
|
|
876
|
+
read: () => outcome(async () => (await journal.replay()).projection)
|
|
877
|
+
};
|
|
855
878
|
}
|
|
856
879
|
//#endregion
|
|
857
880
|
//#region src/campaign/gates/power-preflight.ts
|
|
@@ -862,10 +885,7 @@ function zFor(confidence) {
|
|
|
862
885
|
if (confidence >= .9) return 1.645;
|
|
863
886
|
return 1.282;
|
|
864
887
|
}
|
|
865
|
-
/** Estimate
|
|
866
|
-
* ship at a given budget, from the baseline holdout composites — call it BEFORE
|
|
867
|
-
* spending a search to learn whether the effect you are hunting is even
|
|
868
|
-
* observable at this holdout size and worker variance. */
|
|
888
|
+
/** Estimate detectable lift from baseline independent observations before budgeting a comparison. */
|
|
869
889
|
function powerPreflight(opts) {
|
|
870
890
|
const composites = opts.baselineComposites.filter((v) => Number.isFinite(v));
|
|
871
891
|
if (composites.length < 3) throw new Error(`powerPreflight: need >= 3 finite baseline composites to estimate variance, got ${composites.length}`);
|
|
@@ -881,8 +901,8 @@ function powerPreflight(opts) {
|
|
|
881
901
|
const scaleAssumed = composites.every((v) => v >= -.001 && v <= 1.5);
|
|
882
902
|
const headroom = Math.max(0, 1 - mean);
|
|
883
903
|
const underpowered = scaleAssumed && mde > headroom;
|
|
884
|
-
const sharedChannelCaveat = opts.sharedScorerChannel ? "Holdout and gate share one scoring channel:
|
|
885
|
-
const recommendation = underpowered ? `UNDERPOWERED:
|
|
904
|
+
const sharedChannelCaveat = opts.sharedScorerChannel ? "Holdout and gate share one scoring channel: systematic judge bias remains outside this estimate. An independent second scoring channel can help test that bias." : void 0;
|
|
905
|
+
const recommendation = underpowered ? `UNDERPOWERED under this approximation: detectable lift ${mde.toFixed(3)} exceeds the ${headroom.toFixed(3)} headroom above the baseline (${mean.toFixed(3)}). Raise paired n using independent observations to ~${Math.ceil((z * Math.SQRT2 * sd / Math.max(headroom - deltaThreshold, .01)) ** 2)} or reduce observation variance. Recheck with measured paired deltas.` : `Approximate detectable lift at n=${n}: ${mde.toFixed(3)} (baseline sd ${sd.toFixed(3)}). Compare this estimate with the effect you expect, then recheck using measured paired deltas.`;
|
|
886
906
|
return {
|
|
887
907
|
n,
|
|
888
908
|
sd,
|
|
@@ -1202,6 +1222,35 @@ function parseReflectionResponse(raw, maxProposals) {
|
|
|
1202
1222
|
* "composite of a campaign" and "per-scenario / per-dimension breakdown" so
|
|
1203
1223
|
* the optimizers cannot drift on how a surface's score is computed.
|
|
1204
1224
|
*/
|
|
1225
|
+
/** Reduce complete paired cells using the same observation units as held-out inference. */
|
|
1226
|
+
function pairedCampaignComposites(baseline, candidate, independentUnitByScenarioId) {
|
|
1227
|
+
const scoresByCell = (campaign) => {
|
|
1228
|
+
const scores = /* @__PURE__ */ new Map();
|
|
1229
|
+
for (const cell of campaign.cells) {
|
|
1230
|
+
if (scores.has(cell.cellId)) throw new Error(`pairedCampaignComposites: duplicate cell '${cell.cellId}'`);
|
|
1231
|
+
const quality = projectCampaignCellQuality(cell);
|
|
1232
|
+
scores.set(cell.cellId, quality.score === void 0 ? {} : quality.successfulJudgeScores);
|
|
1233
|
+
}
|
|
1234
|
+
return scores;
|
|
1235
|
+
};
|
|
1236
|
+
const scenarioIds = new Set([...baseline.cells, ...candidate.cells].map((cell) => cell.scenarioId));
|
|
1237
|
+
const paired = pairHoldout(scoresByCell(candidate), scoresByCell(baseline), scenarioIds, (score) => score.composite);
|
|
1238
|
+
const observations = aggregatePairedHoldout(paired, independentUnitByScenarioId);
|
|
1239
|
+
const scoredCellIds = new Set(paired.cellIds);
|
|
1240
|
+
if (observations.before.length === 0) throw new Error("pairedCampaignComposites: campaigns have no paired quality scores");
|
|
1241
|
+
return {
|
|
1242
|
+
before: observations.before,
|
|
1243
|
+
after: observations.after,
|
|
1244
|
+
beforeMean: observations.before.reduce((sum, score) => sum + score, 0) / observations.before.length,
|
|
1245
|
+
afterMean: observations.after.reduce((sum, score) => sum + score, 0) / observations.after.length,
|
|
1246
|
+
observations: {
|
|
1247
|
+
pairedCellN: paired.cellIds.length,
|
|
1248
|
+
unitIds: observations.unitIds,
|
|
1249
|
+
unscoredCellIds: baseline.cells.filter((cell) => !scoredCellIds.has(cell.cellId)).map((cell) => cell.cellId).sort(),
|
|
1250
|
+
...independentUnitByScenarioId ? { independentUnitByScenarioId: Object.fromEntries([...scenarioIds].sort().map((id) => [id, independentUnitByScenarioId.get(id)])) } : {}
|
|
1251
|
+
}
|
|
1252
|
+
};
|
|
1253
|
+
}
|
|
1205
1254
|
/** Mean composite across cells with complete task-quality evidence.
|
|
1206
1255
|
* Partial judge results remain on their cells but never enter this value.
|
|
1207
1256
|
* A campaign with no complete score has no numeric mean and fails loudly. */
|
|
@@ -1344,6 +1393,8 @@ function loopProvenanceArgsFromResult(input) {
|
|
|
1344
1393
|
...result.holdout === "deferred" ? { holdout: "deferred" } : {},
|
|
1345
1394
|
baselineOnHoldout: result.baselineOnHoldout,
|
|
1346
1395
|
winnerOnHoldout: result.winnerOnHoldout,
|
|
1396
|
+
...result.claim ? { claim: result.claim } : {},
|
|
1397
|
+
...input.independentUnitByScenarioId ? { independentUnitByScenarioId: input.independentUnitByScenarioId } : {},
|
|
1347
1398
|
...result.neutralizedSurface && result.neutralizedOnHoldout ? {
|
|
1348
1399
|
neutralizedSurface: result.neutralizedSurface,
|
|
1349
1400
|
neutralizedOnHoldout: result.neutralizedOnHoldout
|
|
@@ -1353,9 +1404,6 @@ function loopProvenanceArgsFromResult(input) {
|
|
|
1353
1404
|
totalDurationMs: input.totalDurationMs
|
|
1354
1405
|
};
|
|
1355
1406
|
}
|
|
1356
|
-
function meanHoldoutComposite(campaign) {
|
|
1357
|
-
return campaignMeanComposite(campaign);
|
|
1358
|
-
}
|
|
1359
1407
|
/** Build the durable provenance record from a completed loop result. */
|
|
1360
1408
|
function buildLoopProvenanceRecord(args) {
|
|
1361
1409
|
if (!args.runId.trim() || !args.runDir.trim()) throw new Error("buildLoopProvenanceRecord: runId and runDir must be non-empty");
|
|
@@ -1422,15 +1470,18 @@ function buildLoopProvenanceRecord(args) {
|
|
|
1422
1470
|
}
|
|
1423
1471
|
if (surfaceHash(args.winnerSurface) !== incumbentSurfaceHash) throw new Error("buildLoopProvenanceRecord: winner surface does not match the final promoted incumbent");
|
|
1424
1472
|
const holdoutDeferred = args.holdout === "deferred";
|
|
1473
|
+
const claim = args.claim ? defineEvaluationClaim(args.claim) : void 0;
|
|
1474
|
+
if (claim && !holdoutDeferred && args.independentUnitByScenarioId === void 0) throw new Error("buildLoopProvenanceRecord: a measured claim requires its independent-unit map");
|
|
1425
1475
|
if (args.baselineOnHoldout.splitDigest !== args.winnerOnHoldout.splitDigest) throw new Error("buildLoopProvenanceRecord: baseline and winner use different holdout splits");
|
|
1426
1476
|
if (args.neutralizedSurface === void 0 !== (args.neutralizedOnHoldout === void 0)) throw new Error("buildLoopProvenanceRecord: neutralized surface and campaign must be supplied together");
|
|
1427
1477
|
if (args.neutralizedOnHoldout && args.neutralizedOnHoldout.splitDigest !== args.baselineOnHoldout.splitDigest) throw new Error("buildLoopProvenanceRecord: neutralized campaign uses a different holdout split");
|
|
1428
1478
|
if (holdoutDeferred && args.neutralizedOnHoldout) throw new Error("buildLoopProvenanceRecord: a deferred holdout cannot include a neutralized measurement");
|
|
1429
|
-
const
|
|
1479
|
+
const holdoutScores = holdoutDeferred ? void 0 : pairedCampaignComposites(args.baselineOnHoldout, args.winnerOnHoldout, args.independentUnitByScenarioId);
|
|
1480
|
+
const holdoutMeasurement = holdoutScores === void 0 ? { kind: "deferred" } : {
|
|
1430
1481
|
kind: "measured",
|
|
1431
|
-
baseline:
|
|
1432
|
-
winner:
|
|
1433
|
-
...args.neutralizedOnHoldout ? { neutralized:
|
|
1482
|
+
baseline: holdoutScores.beforeMean,
|
|
1483
|
+
winner: holdoutScores.afterMean,
|
|
1484
|
+
...args.neutralizedOnHoldout ? { neutralized: pairedCampaignComposites(args.baselineOnHoldout, args.neutralizedOnHoldout, args.independentUnitByScenarioId).afterMean } : {}
|
|
1434
1485
|
};
|
|
1435
1486
|
const diff = surfaceContentHash(args.baselineSurface) === surfaceContentHash(args.winnerSurface) ? "" : renderSurfaceDiff(args.winnerSurface, args.baselineSurface);
|
|
1436
1487
|
const recordWithoutDigest = {
|
|
@@ -1451,6 +1502,7 @@ function buildLoopProvenanceRecord(args) {
|
|
|
1451
1502
|
splitDigest: args.baselineOnHoldout.splitDigest,
|
|
1452
1503
|
baselineCampaignDigest: campaignMeasurementDigest(args.baselineOnHoldout),
|
|
1453
1504
|
winnerCampaignDigest: campaignMeasurementDigest(args.winnerOnHoldout),
|
|
1505
|
+
...holdoutScores ? { observations: holdoutScores.observations } : {},
|
|
1454
1506
|
...args.neutralizedSurface && args.neutralizedOnHoldout && holdoutMeasurement.kind === "measured" && holdoutMeasurement.neutralized !== void 0 ? { neutralized: {
|
|
1455
1507
|
contentHash: surfaceContentHash(args.neutralizedSurface),
|
|
1456
1508
|
campaignDigest: campaignMeasurementDigest(args.neutralizedOnHoldout),
|
|
@@ -1461,6 +1513,7 @@ function buildLoopProvenanceRecord(args) {
|
|
|
1461
1513
|
costReceiptsDigest: canonicalDigest([...args.costReceipts].sort((left, right) => compareCodeUnits(left.callId, right.callId)))
|
|
1462
1514
|
},
|
|
1463
1515
|
baselineSearchComposite,
|
|
1516
|
+
...claim ? { claim } : {},
|
|
1464
1517
|
gate: {
|
|
1465
1518
|
decision: args.gate.decision,
|
|
1466
1519
|
reasons: args.gate.reasons,
|
|
@@ -1890,15 +1943,18 @@ function attest(report, provenance) {
|
|
|
1890
1943
|
* canonicalizes) is a verification failure with the cause in `reason`, not a
|
|
1891
1944
|
* crash — verifiers run in pipelines that must record WHY, not die.
|
|
1892
1945
|
*
|
|
1893
|
-
*
|
|
1894
|
-
*
|
|
1895
|
-
* them instead of accidentally treating old metadata as cryptographic proof.
|
|
1946
|
+
* Both the report hash and the provenance envelope must verify.
|
|
1947
|
+
* Missing envelope hashes cannot establish provenance and fail verification.
|
|
1896
1948
|
*/
|
|
1897
1949
|
function verifyAttestation(report, attested) {
|
|
1898
1950
|
if (attested.algorithm !== "sha256/canonical-json") return {
|
|
1899
1951
|
valid: false,
|
|
1900
1952
|
reason: `unknown algorithm '${attested.algorithm}' — this verifier only checks '${ATTESTATION_ALGORITHM}'`
|
|
1901
1953
|
};
|
|
1954
|
+
if (typeof attested.envelopeHash !== "string" || !/^[0-9a-f]{64}$/.test(attested.envelopeHash)) return {
|
|
1955
|
+
valid: false,
|
|
1956
|
+
reason: "attestation envelope hash is missing or invalid"
|
|
1957
|
+
};
|
|
1902
1958
|
let recomputed;
|
|
1903
1959
|
try {
|
|
1904
1960
|
recomputed = contentHash(report);
|
|
@@ -1912,10 +1968,6 @@ function verifyAttestation(report, attested) {
|
|
|
1912
1968
|
valid: false,
|
|
1913
1969
|
reason: `report hash mismatch: attested ${attested.reportHash}, recomputed ${recomputed}`
|
|
1914
1970
|
};
|
|
1915
|
-
if (attested.envelopeHash === void 0) return {
|
|
1916
|
-
valid: true,
|
|
1917
|
-
legacyUnboundProvenance: true
|
|
1918
|
-
};
|
|
1919
1971
|
let envelopeHash;
|
|
1920
1972
|
try {
|
|
1921
1973
|
envelopeHash = contentHash(envelopeMaterial(attested.reportHash, attested.provenance, attested.algorithm));
|
|
@@ -1989,10 +2041,8 @@ function createEvidenceReceipt(input, provenance) {
|
|
|
1989
2041
|
});
|
|
1990
2042
|
}
|
|
1991
2043
|
/**
|
|
1992
|
-
* Verify
|
|
1993
|
-
*
|
|
1994
|
-
* model versions, input commitment provenance, or creation record must invalidate the
|
|
1995
|
-
* evidence rather than merely annotating it as legacy.
|
|
2044
|
+
* Verify evidence identity and its provenance envelope. Changing the evaluator,
|
|
2045
|
+
* model versions, input commitment, or creation record invalidates the evidence.
|
|
1996
2046
|
*/
|
|
1997
2047
|
function verifyEvidenceReceipt(receipt) {
|
|
1998
2048
|
if (receipt.binding.schemaVersion !== "1.0.0") return {
|
|
@@ -2018,13 +2068,7 @@ function verifyEvidenceReceipt(receipt) {
|
|
|
2018
2068
|
reason: error instanceof Error ? error.message : String(error)
|
|
2019
2069
|
};
|
|
2020
2070
|
}
|
|
2021
|
-
|
|
2022
|
-
if (!verification.valid) return verification;
|
|
2023
|
-
if (verification.legacyUnboundProvenance === true || receipt.attestation.envelopeHash === void 0) return {
|
|
2024
|
-
valid: false,
|
|
2025
|
-
reason: "evidence receipt provenance is not bound by an attestation envelope"
|
|
2026
|
-
};
|
|
2027
|
-
return { valid: true };
|
|
2071
|
+
return verifyAttestation(receipt.binding, receipt.attestation);
|
|
2028
2072
|
}
|
|
2029
2073
|
/**
|
|
2030
2074
|
* Promotion may choose a stricter policy, but this primitive makes the basic separation
|
|
@@ -2078,6 +2122,6 @@ function createCampaignEvidenceReceipt(input) {
|
|
|
2078
2122
|
});
|
|
2079
2123
|
}
|
|
2080
2124
|
//#endregion
|
|
2081
|
-
export {
|
|
2125
|
+
export { formatCoverageFailures as $, FinalEvidenceConflictError as A, surfaceDispatchRef as B, campaignMeanCompositeOrNull as C, parseReflectionResponse as D, buildReflectionPrompt as E, assertCodeSurfaceIdentity as F, summarizeBackendIntegrity as G, surfaceHashMatches as H, codeSurfaceIdentityMaterial as I, assertCompleteCampaign as J, assertCampaignDesign as K, componentSurfaceIdentityMaterial as L, openFinalEvidenceLedger as M, defineEvaluationClaim as N, recoverTruncatedJson as O, summarizeEvaluationUnits as P, campaignSplitDigestFromIdentities as Q, renderSurfaceDiff as R, campaignMeanComposite as S, pairedCampaignComposites as T, BackendIntegrityError as U, surfaceHash as V, assertRealBackend as W, campaignScenarioIdentity as X, campaignCoverage as Y, campaignSplitDigest as Z, provenanceRecordPath as _, createEvidenceReceipt as a, pairHoldout as at, assertFiniteRankKey as b, ATTESTATION_ALGORITHM as c, buildLoopProvenanceRecord as d, TIE_WARN_FRACTION as et, campaignMeasurementDigest as f, loopProvenanceSpans as g, loopProvenanceArgsFromResult as h, INDEPENDENT_EVIDENCE_AUTHORITY_KINDS as i, heldoutSignificance as it, FinalEvidenceError as j, powerPreflight as k, attest as l, emitLoopProvenance as m, EVIDENCE_AUTHORITY_KINDS as n, detectScale as nt, isIndependentEvidence as o, canonicalDigest as p, assertCampaignSplitIdentity as q, EVIDENCE_RECEIPT_VERSION as r, dimensionRegressions as rt, verifyEvidenceReceipt as s, createCampaignEvidenceReceipt as t, aggregatePairedHoldout as tt, verifyAttestation as u, provenanceSpansPath as v, compareRankKeys as w, campaignBreakdown as x, verifyLoopProvenanceRecord as y, surfaceContentHash as z };
|
|
2082
2126
|
|
|
2083
|
-
//# sourceMappingURL=campaign-evidence-
|
|
2127
|
+
//# sourceMappingURL=campaign-evidence-B8oF9xQ6.js.map
|