@tangle-network/agent-eval 0.144.6 → 0.144.7
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +24 -0
- package/README.md +2 -0
- package/dist/{benchmark-J9Qe6j2_.d.ts → agent-profile-CgDTo40f.d.ts} +91 -4
- package/dist/agent-profile-CgDTo40f.d.ts.map +1 -0
- package/dist/analyst/index.d.ts +470 -88
- package/dist/analyst/index.d.ts.map +1 -1
- package/dist/analyst/index.js +24 -5
- package/dist/analyst/index.js.map +1 -1
- package/dist/{analyze-runs-DWIvOAGk.js → analyze-runs-BNNK7irB.js} +3 -3
- package/dist/{analyze-runs-DWIvOAGk.js.map → analyze-runs-BNNK7irB.js.map} +1 -1
- package/dist/{baseline-D_fT6277.d.ts → baseline-CavEbRyH.d.ts} +2 -23
- package/dist/baseline-CavEbRyH.d.ts.map +1 -0
- package/dist/{benchmark-command-CQd78YHt.js → benchmark-command-BCafwNrf.js} +662 -605
- package/dist/benchmark-command-BCafwNrf.js.map +1 -0
- package/dist/benchmarks/index.d.ts +1 -1
- package/dist/benchmarks/index.js +1 -1
- package/dist/{benchmarks-BEOkuvIg.js → benchmarks-CDSolHq7.js} +4 -4
- package/dist/{benchmarks-BEOkuvIg.js.map → benchmarks-CDSolHq7.js.map} +1 -1
- package/dist/campaign/index.d.ts +7 -5
- package/dist/campaign/index.js +5 -3
- package/dist/{campaign-CXsdyym7.js → campaign-Tdy3h62h.js} +17 -301
- package/dist/campaign-Tdy3h62h.js.map +1 -0
- package/dist/cli.js +2 -2
- package/dist/{client-0JI64ovJ.d.ts → client-DjXROWpx.d.ts} +3 -3
- package/dist/{client-0JI64ovJ.d.ts.map → client-DjXROWpx.d.ts.map} +1 -1
- package/dist/{completion-verifier-CBiee74w.d.ts → completion-verifier-foUCLif_.d.ts} +5 -5
- package/dist/{completion-verifier-CBiee74w.d.ts.map → completion-verifier-foUCLif_.d.ts.map} +1 -1
- package/dist/contract/index.d.ts +9 -8
- package/dist/contract/index.d.ts.map +1 -1
- package/dist/contract/index.js +7 -6
- package/dist/contract/index.js.map +1 -1
- package/dist/control.d.ts +2 -2
- package/dist/control.js +1 -1
- package/dist/counterfactual-CWPTrMH7.js +126 -0
- package/dist/counterfactual-CWPTrMH7.js.map +1 -0
- package/dist/counterfactual-CxmxAONP.d.ts +72 -0
- package/dist/counterfactual-CxmxAONP.d.ts.map +1 -0
- package/dist/{default-registry-J9m-_tya.d.ts → default-registry-BZhStdjl.d.ts} +6 -5
- package/dist/default-registry-BZhStdjl.d.ts.map +1 -0
- package/dist/{default-registry-Dta70shL.js → default-registry-BaQXW1Ow.js} +2 -2
- package/dist/{default-registry-Dta70shL.js.map → default-registry-BaQXW1Ow.js.map} +1 -1
- package/dist/{dspy-rlm-engine-19FQEMBK.js → dspy-rlm-engine-BiN49gK6.js} +4 -3
- package/dist/{dspy-rlm-engine-19FQEMBK.js.map → dspy-rlm-engine-BiN49gK6.js.map} +1 -1
- package/dist/{tool-groups-CK0JCkqO.d.ts → engine-nB64f48I.d.ts} +18 -31
- package/dist/engine-nB64f48I.d.ts.map +1 -0
- package/dist/{eval-campaign-CfLQQs9B.js → eval-campaign-DNjCvAm-.js} +7 -6
- package/dist/{eval-campaign-CfLQQs9B.js.map → eval-campaign-DNjCvAm-.js.map} +1 -1
- package/dist/{exact-types-CBYF5MGd.d.ts → exact-types-Djvzosly.d.ts} +2 -2
- package/dist/{exact-types-CBYF5MGd.d.ts.map → exact-types-Djvzosly.d.ts.map} +1 -1
- package/dist/exec-BLtYZdWo.js +49 -0
- package/dist/exec-BLtYZdWo.js.map +1 -0
- package/dist/experiment/index.d.ts +802 -0
- package/dist/experiment/index.d.ts.map +1 -0
- package/dist/experiment/index.js +1108 -0
- package/dist/experiment/index.js.map +1 -0
- package/dist/experiment-tracker-CnRICnMl.js +500 -0
- package/dist/experiment-tracker-CnRICnMl.js.map +1 -0
- package/dist/experiment-tracker-IMntXr6J.d.ts +299 -0
- package/dist/experiment-tracker-IMntXr6J.d.ts.map +1 -0
- package/dist/{external-optimizer-contracts-Q9c0L3eY.d.ts → external-optimizer-contracts-CZuJNcT5.d.ts} +2 -2
- package/dist/{external-optimizer-contracts-Q9c0L3eY.d.ts.map → external-optimizer-contracts-CZuJNcT5.d.ts.map} +1 -1
- package/dist/{feedback-trajectory-GgoS0-MK.d.ts → feedback-trajectory-Rh280oXo.d.ts} +3 -3
- package/dist/{feedback-trajectory-GgoS0-MK.d.ts.map → feedback-trajectory-Rh280oXo.d.ts.map} +1 -1
- package/dist/fuzz.d.ts +1 -1
- package/dist/hosted/index.d.ts +3 -3
- package/dist/{index-BZUe-ODI.d.ts → index-BZ3-y4YL.d.ts} +3 -3
- package/dist/{index-BZUe-ODI.d.ts.map → index-BZ3-y4YL.d.ts.map} +1 -1
- package/dist/{index-DSC51roc.d.ts → index-C3ssXVLv.d.ts} +2 -2
- package/dist/{index-DSC51roc.d.ts.map → index-C3ssXVLv.d.ts.map} +1 -1
- package/dist/{index-4XwggC10.d.ts → index-C5HOo4ZF2.d.ts} +4 -4
- package/dist/index-C5HOo4ZF2.d.ts.map +1 -0
- package/dist/{index-B6-B0zTB.d.ts → index-CvXXlyz7.d.ts} +2 -2
- package/dist/{index-B6-B0zTB.d.ts.map → index-CvXXlyz7.d.ts.map} +1 -1
- package/dist/{index-Dx1kF3Ez.d.ts → index-CwDrUMe0.d.ts} +2 -2
- package/dist/{index-Dx1kF3Ez.d.ts.map → index-CwDrUMe0.d.ts.map} +1 -1
- package/dist/{index-BIL5vxxt.d.ts → index-Sh2I0DRc.d.ts} +11 -645
- package/dist/index-Sh2I0DRc.d.ts.map +1 -0
- package/dist/index.d.ts +214 -404
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +233 -649
- package/dist/index.js.map +1 -1
- package/dist/{insight-report-DqEsugpr.d.ts → insight-report-C6h6F_4L.d.ts} +3 -3
- package/dist/{insight-report-DqEsugpr.d.ts.map → insight-report-C6h6F_4L.d.ts.map} +1 -1
- package/dist/{integrity-CNGUaGBY.d.ts → integrity-BuqEKu-x.d.ts} +2 -2
- package/dist/{integrity-CNGUaGBY.d.ts.map → integrity-BuqEKu-x.d.ts.map} +1 -1
- package/dist/integrity-MLzHOfV9.js +141 -0
- package/dist/integrity-MLzHOfV9.js.map +1 -0
- package/dist/kind-factory-BHIgPmzS.js.map +1 -1
- package/dist/{llm-client-Dv5BiKLE.js → llm-client-DzvMUsS_.js} +24 -6
- package/dist/llm-client-DzvMUsS_.js.map +1 -0
- package/dist/matrix/index.d.ts +2 -2
- package/dist/meta-eval/index.d.ts +1 -1
- package/dist/{mint-DD-0oQTA.js → mint-CGEkzPLf.js} +2 -2
- package/dist/{mint-DD-0oQTA.js.map → mint-CGEkzPLf.js.map} +1 -1
- package/dist/{multi-layer-verifier-BHY1gWAc.d.ts → multi-layer-verifier-DnAqwl0h.d.ts} +2 -2
- package/dist/{multi-layer-verifier-BHY1gWAc.d.ts.map → multi-layer-verifier-DnAqwl0h.d.ts.map} +1 -1
- package/dist/multishot/index.d.ts +2 -2
- package/dist/openapi.json +1 -1
- package/dist/paired-promotion-decision-B6zJ3gYM.d.ts +114 -0
- package/dist/paired-promotion-decision-B6zJ3gYM.d.ts.map +1 -0
- package/dist/pipelines/index.d.ts +2 -1
- package/dist/pipelines/index.d.ts.map +1 -1
- package/dist/pre-registration-DakwTRXk.js +96 -0
- package/dist/pre-registration-DakwTRXk.js.map +1 -0
- package/dist/prime-bridge-transport-6feEglLf.d.ts +49 -0
- package/dist/prime-bridge-transport-6feEglLf.d.ts.map +1 -0
- package/dist/prime-protocol-BfSalTfR.js +453 -0
- package/dist/prime-protocol-BfSalTfR.js.map +1 -0
- package/dist/profile-cell.js +242 -1
- package/dist/profile-cell.js.map +1 -0
- package/dist/promotion-policy-ChWhTDBH.d.ts +289 -0
- package/dist/promotion-policy-ChWhTDBH.d.ts.map +1 -0
- package/dist/promotion-policy-CrLrmys8.js +682 -0
- package/dist/promotion-policy-CrLrmys8.js.map +1 -0
- package/dist/{propose-review-control-BciUCZoh.js → propose-review-control-qgWLA6E9.js} +2 -2
- package/dist/{propose-review-control-BciUCZoh.js.map → propose-review-control-qgWLA6E9.js.map} +1 -1
- package/dist/{release-report-ChOgpIoQ.d.ts → release-report-CI8uisI1.d.ts} +2 -2
- package/dist/{release-report-ChOgpIoQ.d.ts.map → release-report-CI8uisI1.d.ts.map} +1 -1
- package/dist/{release-report-Sl0xfkFv.js → release-report-Dy39cbFF.js} +3 -3
- package/dist/{release-report-Sl0xfkFv.js.map → release-report-Dy39cbFF.js.map} +1 -1
- package/dist/{replay-Krvb114g.d.ts → replay-DFf-teiC.d.ts} +5 -4
- package/dist/replay-DFf-teiC.d.ts.map +1 -0
- package/dist/{replay-GW61ezMW.js → replay-DyBLaKFc.js} +3 -3
- package/dist/{replay-GW61ezMW.js.map → replay-DyBLaKFc.js.map} +1 -1
- package/dist/reporting.d.ts +3 -3
- package/dist/reporting.js +2 -2
- package/dist/{researcher-xLeNcpKX.d.ts → researcher-BoaxeCzP.d.ts} +4 -4
- package/dist/{researcher-xLeNcpKX.d.ts.map → researcher-BoaxeCzP.d.ts.map} +1 -1
- package/dist/{reward-hacking-CyuzxKly.js → reward-hacking-B2tPL9a3.js} +12 -3
- package/dist/reward-hacking-B2tPL9a3.js.map +1 -0
- package/dist/{reward-hacking-RZgnGWlx.d.ts → reward-hacking-Cf1PtEOz.d.ts} +33 -3
- package/dist/reward-hacking-Cf1PtEOz.d.ts.map +1 -0
- package/dist/rl.d.ts +17 -7
- package/dist/rl.d.ts.map +1 -1
- package/dist/rl.js +16 -7
- package/dist/rl.js.map +1 -1
- package/dist/rollout/index.d.ts +2 -2
- package/dist/rollout/index.js +2 -2
- package/dist/{rollout-C2fD1cf4.js → rollout-2ECTXb2N.js} +2 -2
- package/dist/{rollout-C2fD1cf4.js.map → rollout-2ECTXb2N.js.map} +1 -1
- package/dist/{run-evidence-C6G41MSI.d.ts → run-evidence-BDFFai9R.d.ts} +2 -2
- package/dist/{run-evidence-C6G41MSI.d.ts.map → run-evidence-BDFFai9R.d.ts.map} +1 -1
- package/dist/{run-record-CWN8-VsV.js → run-record-DqOw5X6_.js} +2 -2
- package/dist/{run-record-CWN8-VsV.js.map → run-record-DqOw5X6_.js.map} +1 -1
- package/dist/{schema-Cef2cFmb.d.ts → schema-Cef2cFmb2.d.ts} +1 -1
- package/dist/schema-Cef2cFmb2.d.ts.map +1 -0
- package/dist/{semantic-concept-judge-DwF6n05O.js → semantic-concept-judge-Bmrq6yqU.js} +2 -2
- package/dist/{semantic-concept-judge-DwF6n05O.js.map → semantic-concept-judge-Bmrq6yqU.js.map} +1 -1
- package/dist/sequential-D-BLJBKU.js +299 -0
- package/dist/sequential-D-BLJBKU.js.map +1 -0
- package/dist/{server-D6XJQHw7.js → server-iu0ede49.js} +2 -2
- package/dist/{server-D6XJQHw7.js.map → server-iu0ede49.js.map} +1 -1
- package/dist/{single-run-lock-D5iN0Xzb.js → single-run-lock-BMQEv1wG.js} +7 -161
- package/dist/single-run-lock-BMQEv1wG.js.map +1 -0
- package/dist/{skill-usage-GlOphAhX.d.ts → skill-usage-CJlWEUFt.d.ts} +10 -10
- package/dist/{skill-usage-GlOphAhX.d.ts.map → skill-usage-CJlWEUFt.d.ts.map} +1 -1
- package/dist/{skillopt-optimization-method-7S43rbDB.d.ts → skillopt-optimization-method-BO7NAl3b.d.ts} +8 -294
- package/dist/skillopt-optimization-method-BO7NAl3b.d.ts.map +1 -0
- package/dist/{skillopt-optimization-method-Bfb-vBKe.js → skillopt-optimization-method-CQwZ-ZX8.js} +8 -652
- package/dist/skillopt-optimization-method-CQwZ-ZX8.js.map +1 -0
- package/dist/statistical-heldout-Dn9ruizm.d.ts +555 -0
- package/dist/statistical-heldout-Dn9ruizm.d.ts.map +1 -0
- package/dist/{statistics-C-dm-J6H.d.ts → statistics-D6Uebe_4.d.ts} +2 -2
- package/dist/{statistics-C-dm-J6H.d.ts.map → statistics-D6Uebe_4.d.ts.map} +1 -1
- package/dist/steps-BArUxhna.d.ts +51 -0
- package/dist/steps-BArUxhna.d.ts.map +1 -0
- package/dist/{integrity-fdt8XPAv.js → store-DNe_Uv1Q.js} +2 -140
- package/dist/store-DNe_Uv1Q.js.map +1 -0
- package/dist/{summary-report-B0cAyA7N.d.ts → summary-report-DuUS_i7W.d.ts} +3 -114
- package/dist/summary-report-DuUS_i7W.d.ts.map +1 -0
- package/dist/{summary-report-9A5y7EsK.js → summary-report-Lf-5I7xh.js} +2 -2
- package/dist/{summary-report-9A5y7EsK.js.map → summary-report-Lf-5I7xh.js.map} +1 -1
- package/dist/supervisor-run/index.d.ts +2 -2
- package/dist/supervisor-run/index.js +1 -1
- package/dist/{supervisor-run-DiyQVczd.js → supervisor-run-D_sokXcO.js} +32 -8
- package/dist/{supervisor-run-DiyQVczd.js.map → supervisor-run-D_sokXcO.js.map} +1 -1
- package/dist/tool-groups-DIVBnJyl.d.ts +28 -0
- package/dist/tool-groups-DIVBnJyl.d.ts.map +1 -0
- package/dist/trace-repair/index.d.ts +2102 -0
- package/dist/trace-repair/index.d.ts.map +1 -0
- package/dist/trace-repair/index.js +3878 -0
- package/dist/trace-repair/index.js.map +1 -0
- package/dist/traces.d.ts +6 -6
- package/dist/traces.js +3 -2
- package/dist/trajectory-YC15QDYQ.d.ts +24 -0
- package/dist/trajectory-YC15QDYQ.d.ts.map +1 -0
- package/dist/trajectory-replay/index.d.ts +781 -0
- package/dist/trajectory-replay/index.d.ts.map +1 -0
- package/dist/trajectory-replay/index.js +2103 -0
- package/dist/trajectory-replay/index.js.map +1 -0
- package/dist/{types-XMVEdrE_.d.ts → types-D216SgwM.d.ts} +24 -6
- package/dist/types-D216SgwM.d.ts.map +1 -0
- package/dist/{types-yLK8gXE9.d.ts → types-D4mog56g.d.ts} +2 -2
- package/dist/{types-yLK8gXE9.d.ts.map → types-D4mog56g.d.ts.map} +1 -1
- package/dist/{types-BhP9q0Fq.d.ts → types-DF_Udrp-.d.ts} +52 -3
- package/dist/{types-BhP9q0Fq.d.ts.map → types-DF_Udrp-.d.ts.map} +1 -1
- package/dist/{types-DOZyvsFU.d.ts → types-DYuNHo9R.d.ts} +3 -3
- package/dist/{types-DOZyvsFU.d.ts.map → types-DYuNHo9R.d.ts.map} +1 -1
- package/dist/usage-receipt-EVI8B8Xu.js.map +1 -1
- package/dist/verdict-DExhxfgR.d.ts +201 -0
- package/dist/verdict-DExhxfgR.d.ts.map +1 -0
- package/dist/verdict-cache-BCcOh0kF.js +159 -0
- package/dist/verdict-cache-BCcOh0kF.js.map +1 -0
- package/dist/wire/index.d.ts +2 -2
- package/dist/wire/index.js +1 -1
- package/docs/charter.md +112 -0
- package/docs/experiment.md +104 -0
- package/docs/prime-analyst.md +1 -0
- package/docs/trace-analysis.md +26 -0
- package/docs/trace-repair-admission.md +194 -0
- package/docs/trace-repair-analyst-arms.md +121 -0
- package/docs/trace-repair-continuation.md +107 -0
- package/docs/trace-repair-grader.md +163 -0
- package/docs/trajectory-replay.md +110 -0
- package/docs/verification-strategies.md +103 -0
- package/package.json +19 -2
- package/dist/agent-profile-cell-CbfBm2g6.js +0 -335
- package/dist/agent-profile-cell-CbfBm2g6.js.map +0 -1
- package/dist/baseline-D_fT6277.d.ts.map +0 -1
- package/dist/benchmark-J9Qe6j2_.d.ts.map +0 -1
- package/dist/benchmark-command-CQd78YHt.js.map +0 -1
- package/dist/campaign-CXsdyym7.js.map +0 -1
- package/dist/default-registry-J9m-_tya.d.ts.map +0 -1
- package/dist/index-4XwggC10.d.ts.map +0 -1
- package/dist/index-BIL5vxxt.d.ts.map +0 -1
- package/dist/integrity-fdt8XPAv.js.map +0 -1
- package/dist/llm-client-Dv5BiKLE.js.map +0 -1
- package/dist/replay-Krvb114g.d.ts.map +0 -1
- package/dist/reward-hacking-CyuzxKly.js.map +0 -1
- package/dist/reward-hacking-RZgnGWlx.d.ts.map +0 -1
- package/dist/schema-Cef2cFmb.d.ts.map +0 -1
- package/dist/single-run-lock-D5iN0Xzb.js.map +0 -1
- package/dist/skillopt-optimization-method-7S43rbDB.d.ts.map +0 -1
- package/dist/skillopt-optimization-method-Bfb-vBKe.js.map +0 -1
- package/dist/summary-report-B0cAyA7N.d.ts.map +0 -1
- package/dist/tool-groups-CK0JCkqO.d.ts.map +0 -1
- package/dist/types-XMVEdrE_.d.ts.map +0 -1
- package/dist/verdict-Dps8_okt.d.ts +0 -37
- package/dist/verdict-Dps8_okt.d.ts.map +0 -1
|
@@ -0,0 +1,555 @@
|
|
|
1
|
+
import { a as RunRecord } from "./run-record-DdSa93_W.js";
|
|
2
|
+
import { R as Scenario, b as GenerationRecord, p as Gate, w as JudgeScore } from "./types-DYuNHo9R.js";
|
|
3
|
+
import { E as RiskDifferenceResult, _ as McNemarResult, d as EProcessState, v as PairedBootstrapOptions, y as PairedBootstrapResult } from "./statistics-D6Uebe_4.js";
|
|
4
|
+
import { a as PairedPromotionDecision, i as PairedMcNemarEvidence, r as PairedDecisionStatistic, t as PairedDecisionMethod } from "./paired-promotion-decision-B6zJ3gYM.js";
|
|
5
|
+
//#region src/paired-arms.d.ts
|
|
6
|
+
/** One arm observation of one work item. Structural on purpose: callers
|
|
7
|
+
* project their own record type (e.g. a `RunRecord`) into this shape. */
|
|
8
|
+
interface PairedArmRow {
|
|
9
|
+
/** Matching key — rows sharing a `pairKey` across both arms form pairs
|
|
10
|
+
* (typically the task/scenario/seed identity). */
|
|
11
|
+
pairKey: string;
|
|
12
|
+
/** Rep identity within a `pairKey` (e.g. a seed or rep number). Required on
|
|
13
|
+
* every row of a `pairKey` that has more than one rep in either arm; reps
|
|
14
|
+
* then pair only on exact (`pairKey`, `repKey`) match, never on outcome
|
|
15
|
+
* content. Optional when each arm has at most one rep of the item. */
|
|
16
|
+
repKey?: string;
|
|
17
|
+
/** Arm label this row was produced under. */
|
|
18
|
+
arm: string;
|
|
19
|
+
/** Binary outcome; omit when the comparison has no pass/fail notion. */
|
|
20
|
+
pass?: boolean;
|
|
21
|
+
/** Named numeric measurements (score, cost, latency, …). */
|
|
22
|
+
metrics?: Record<string, number>;
|
|
23
|
+
}
|
|
24
|
+
interface PairArmsOptions {
|
|
25
|
+
/** Arm treated as the control side of every pair. */
|
|
26
|
+
baselineArm: string;
|
|
27
|
+
/** Arm treated as the treatment side of every pair. */
|
|
28
|
+
treatmentArm: string;
|
|
29
|
+
}
|
|
30
|
+
/** One matched (baseline, treatment) observation of the same work item. */
|
|
31
|
+
interface MatchedPair {
|
|
32
|
+
pairKey: string;
|
|
33
|
+
/** 0-based position of this pair within its `pairKey`, ordered by sorted
|
|
34
|
+
* `repKey` (always 0 for a single-rep item). The rep identity itself is on
|
|
35
|
+
* the rows (`baseline.repKey` / `treatment.repKey`). */
|
|
36
|
+
repIndex: number;
|
|
37
|
+
baseline: PairedArmRow;
|
|
38
|
+
treatment: PairedArmRow;
|
|
39
|
+
}
|
|
40
|
+
interface PairArmsResult {
|
|
41
|
+
/** Matched pairs, ordered by (`pairKey`, `repIndex`). */
|
|
42
|
+
pairs: MatchedPair[];
|
|
43
|
+
/** Baseline rows left without a treatment counterpart — reported, never
|
|
44
|
+
* silently dropped. */
|
|
45
|
+
unpairedBaseline: PairedArmRow[];
|
|
46
|
+
/** Treatment rows left without a baseline counterpart. */
|
|
47
|
+
unpairedTreatment: PairedArmRow[];
|
|
48
|
+
}
|
|
49
|
+
/**
|
|
50
|
+
* Match rows across two arms into (baseline, treatment) pairs by `pairKey`.
|
|
51
|
+
*
|
|
52
|
+
* A `pairKey` with at most one row per arm pairs directly, no `repKey`
|
|
53
|
+
* needed. A `pairKey` with multiple reps in either arm requires `repKey` on
|
|
54
|
+
* every one of its rows, and reps pair only on exact (`pairKey`, `repKey`)
|
|
55
|
+
* match — pairing is keyed purely on row identity, never on outcome content
|
|
56
|
+
* (outcome-keyed matching deflates discordant counts and biases McNemar), and
|
|
57
|
+
* is therefore independent of input order. Reps whose `repKey` has no
|
|
58
|
+
* counterpart in the other arm, and items present in only one arm, land in
|
|
59
|
+
* the unpaired lists — reported, never truncated.
|
|
60
|
+
*
|
|
61
|
+
* Fail-loud: throws when either named arm has zero rows (an unknown arm
|
|
62
|
+
* name would otherwise read as "everything unpaired"), when the two arm
|
|
63
|
+
* names are equal, when a multi-rep `pairKey` has a row without `repKey`, or
|
|
64
|
+
* when a (`pairKey`, arm) group repeats a `repKey` (the match would be
|
|
65
|
+
* ambiguous).
|
|
66
|
+
*/
|
|
67
|
+
declare function pairArms(rows: readonly PairedArmRow[], opts: PairArmsOptions): PairArmsResult;
|
|
68
|
+
/** Paired pass/fail comparison over the pairs where BOTH sides carry `pass`. */
|
|
69
|
+
interface PairedCorrectness {
|
|
70
|
+
/** Discordant pairs where the treatment passed and the baseline failed. */
|
|
71
|
+
b10: number;
|
|
72
|
+
/** Discordant pairs where the baseline passed and the treatment failed. */
|
|
73
|
+
b01: number;
|
|
74
|
+
/** Exact McNemar significance over the paired outcomes (`b === b10`, `c === b01`). */
|
|
75
|
+
mcnemar: McNemarResult;
|
|
76
|
+
/** Paired effect size: p(treatment) − p(baseline) with a paired-variance CI. */
|
|
77
|
+
riskDifference: RiskDifferenceResult;
|
|
78
|
+
}
|
|
79
|
+
/** Paired delta summary for one named metric (delta = treatment − baseline). */
|
|
80
|
+
interface PairedMetricDelta {
|
|
81
|
+
name: string;
|
|
82
|
+
/** Pairs where BOTH sides carry a finite value for this metric. */
|
|
83
|
+
n: number;
|
|
84
|
+
/** Pairs where at least one side does not carry the metric. */
|
|
85
|
+
nMissing: number;
|
|
86
|
+
/** Median paired delta, or null when `n === 0`. */
|
|
87
|
+
medianDelta: number | null;
|
|
88
|
+
/** Mean paired delta, or null when `n === 0`. */
|
|
89
|
+
meanDelta: number | null;
|
|
90
|
+
/** Bootstrap CI on the paired delta (`pairedBootstrap`); null when
|
|
91
|
+
* `n === 0` — a zero-width [0, 0] interval on no data would read as a
|
|
92
|
+
* measured tight null. */
|
|
93
|
+
bootstrapCi: PairedBootstrapResult | null;
|
|
94
|
+
/** Wilcoxon signed-rank test on the paired deltas; null when `n === 0`. */
|
|
95
|
+
wilcoxon: {
|
|
96
|
+
w: number;
|
|
97
|
+
p: number;
|
|
98
|
+
} | null;
|
|
99
|
+
}
|
|
100
|
+
interface ComparePairedArmsOptions extends PairArmsOptions {
|
|
101
|
+
/** Metrics to compare. Default: every metric name observed on any matched
|
|
102
|
+
* pair, sorted. A name that appears on no pair is still reported (with
|
|
103
|
+
* `n = 0`) so a misspelled metric is visible instead of vanishing. */
|
|
104
|
+
metricNames?: string[];
|
|
105
|
+
/** Passed through to `pairedBootstrap` — set `seed` for reproducible CIs. */
|
|
106
|
+
bootstrap?: PairedBootstrapOptions;
|
|
107
|
+
}
|
|
108
|
+
interface PairedArmsComparison {
|
|
109
|
+
nPairs: number;
|
|
110
|
+
nUnpairedBaseline: number;
|
|
111
|
+
nUnpairedTreatment: number;
|
|
112
|
+
/** null when no matched pair carries `pass` on both sides — a pass/fail
|
|
113
|
+
* verdict over rows that never measured pass/fail would be fabricated. */
|
|
114
|
+
correctness: PairedCorrectness | null;
|
|
115
|
+
metricDeltas: PairedMetricDelta[];
|
|
116
|
+
}
|
|
117
|
+
/**
|
|
118
|
+
* Full matched-pair arm comparison: pair via {@link pairArms}, then compose
|
|
119
|
+
* the paired estimators from `statistics` over the matched pairs.
|
|
120
|
+
*
|
|
121
|
+
* Correctness uses only the pairs where both sides carry `pass` (`mcnemar.n`
|
|
122
|
+
* is that subset's size); each metric uses only the pairs where both sides
|
|
123
|
+
* carry a finite value for it, with the remainder counted in `nMissing`.
|
|
124
|
+
* Deltas are treatment − baseline throughout.
|
|
125
|
+
*
|
|
126
|
+
* Fail-loud: inherits {@link pairArms}'s unknown-arm throw, and throws on a
|
|
127
|
+
* non-finite metric value — silently treating corrupt telemetry as "metric
|
|
128
|
+
* absent" would misreport it as missing coverage.
|
|
129
|
+
*/
|
|
130
|
+
declare function comparePairedArms(rows: readonly PairedArmRow[], opts: ComparePairedArmsOptions): PairedArmsComparison;
|
|
131
|
+
interface MatchedRunRecordPair {
|
|
132
|
+
pairKey: string;
|
|
133
|
+
repKey: string;
|
|
134
|
+
baseline: RunRecord;
|
|
135
|
+
treatment: RunRecord;
|
|
136
|
+
}
|
|
137
|
+
interface PairRunRecordsResult {
|
|
138
|
+
pairs: MatchedRunRecordPair[];
|
|
139
|
+
unpairedBaseline: RunRecord[];
|
|
140
|
+
unpairedTreatment: RunRecord[];
|
|
141
|
+
}
|
|
142
|
+
/**
|
|
143
|
+
* Pair two RunRecord arms by the identity of the evaluated work:
|
|
144
|
+
* `(experimentId, scenarioId, seed)`.
|
|
145
|
+
*
|
|
146
|
+
* Falling back to array order, candidate id, or experiment id can compare
|
|
147
|
+
* different tasks and fabricate lift. Duplicate identities throw.
|
|
148
|
+
*/
|
|
149
|
+
declare function pairRunRecords(baselineRuns: readonly RunRecord[], treatmentRuns: readonly RunRecord[]): PairRunRecordsResult;
|
|
150
|
+
//#endregion
|
|
151
|
+
//#region src/pre-registration.d.ts
|
|
152
|
+
/**
|
|
153
|
+
* Pre-registered hypotheses — declare what you're testing BEFORE the
|
|
154
|
+
* run, check it AFTER. Prevents p-hacking, optional stopping, and the
|
|
155
|
+
* "we ran until it looked good" failure mode.
|
|
156
|
+
*
|
|
157
|
+
* Manifest is a plain JSON-friendly object. Sign it with a content hash
|
|
158
|
+
* + timestamp; the registered record becomes immutable. Post-run,
|
|
159
|
+
* evaluate the manifest against observed results — the library refuses
|
|
160
|
+
* to let you re-interpret a different metric as the declared one.
|
|
161
|
+
*/
|
|
162
|
+
interface HypothesisManifest {
|
|
163
|
+
id: string;
|
|
164
|
+
/** Human prose — goes into the audit trail. */
|
|
165
|
+
hypothesis: string;
|
|
166
|
+
/** Metric the hypothesis claims to move. */
|
|
167
|
+
metric: string;
|
|
168
|
+
/** 'increase' = candidate should score higher than baseline; 'decrease' = lower. */
|
|
169
|
+
direction: 'increase' | 'decrease';
|
|
170
|
+
/** Minimum effect size to count (same units as the metric). */
|
|
171
|
+
minEffect: number;
|
|
172
|
+
/** Alpha threshold. */
|
|
173
|
+
alpha: number;
|
|
174
|
+
/** Target statistical power at which sample size was pre-computed. */
|
|
175
|
+
power: number;
|
|
176
|
+
/** Declared N per arm before running. */
|
|
177
|
+
preRegisteredN: number;
|
|
178
|
+
/** ISO8601 timestamp the manifest was registered. */
|
|
179
|
+
registeredAt: string;
|
|
180
|
+
/** Optional identifiers to tie into the trace corpus. */
|
|
181
|
+
baselineLabel?: string;
|
|
182
|
+
candidateLabel?: string;
|
|
183
|
+
}
|
|
184
|
+
/**
|
|
185
|
+
* Identifier for the hashing scheme used to produce `contentHash`.
|
|
186
|
+
*
|
|
187
|
+
* `'sha256-content'` — sha256 hex over the canonicalized manifest with
|
|
188
|
+
* the `contentHash` and `algo` fields stripped. Held as a string union
|
|
189
|
+
* so future schemes can be added without breaking parsers; SignedManifest
|
|
190
|
+
* values without `algo` deserialize cleanly because the field is optional.
|
|
191
|
+
*/
|
|
192
|
+
type SignedManifestAlgo = 'sha256-content';
|
|
193
|
+
interface SignedManifest extends HypothesisManifest {
|
|
194
|
+
/** sha256 hex of canonicalized manifest (everything except contentHash and algo). */
|
|
195
|
+
contentHash: string;
|
|
196
|
+
/**
|
|
197
|
+
* Algorithm string describing how `contentHash` was produced.
|
|
198
|
+
*
|
|
199
|
+
* Optional on the type so serialized manifests without it still parse,
|
|
200
|
+
* but ALWAYS populated by {@link signManifest}. Consumers that want to
|
|
201
|
+
* enforce a known algorithm should reject manifests where this field
|
|
202
|
+
* is missing or unrecognized.
|
|
203
|
+
*/
|
|
204
|
+
algo?: SignedManifestAlgo;
|
|
205
|
+
}
|
|
206
|
+
interface HypothesisResult {
|
|
207
|
+
manifest: SignedManifest;
|
|
208
|
+
observedN: number;
|
|
209
|
+
observedEffect: number;
|
|
210
|
+
observedPValue: number;
|
|
211
|
+
/** True iff the observed effect hits the pre-declared direction with
|
|
212
|
+
* magnitude ≥ minEffect AND p < alpha. */
|
|
213
|
+
confirmed: boolean;
|
|
214
|
+
/** Enumerated reasons the hypothesis was rejected (each a machine-tag). */
|
|
215
|
+
rejectionReasons: Array<'wrong_direction' | 'effect_too_small' | 'not_significant' | 'undersampled'>;
|
|
216
|
+
notes?: string;
|
|
217
|
+
}
|
|
218
|
+
/**
|
|
219
|
+
* Deterministic JSON canonicalization — sort object keys recursively.
|
|
220
|
+
*
|
|
221
|
+
* Two semantically-equal objects produce byte-identical canonicalized output;
|
|
222
|
+
* this is what makes a content-hash stable across encoders, key insertion
|
|
223
|
+
* orders, and runtime versions. Exported for any consumer that needs the same
|
|
224
|
+
* canonicalization guarantee outside the manifest-signing path (e.g., signing
|
|
225
|
+
* an artifact bundle, hashing a dataset version, etc.).
|
|
226
|
+
*/
|
|
227
|
+
declare function canonicalize(v: unknown): unknown;
|
|
228
|
+
/**
|
|
229
|
+
* SHA-256 hex (full 64 chars) over the canonicalized JSON encoding of `obj`.
|
|
230
|
+
*
|
|
231
|
+
* The same primitive `signManifest` and `verifyManifest` are built on, exposed
|
|
232
|
+
* directly so consumers signing arbitrary structured content (artifact bundles,
|
|
233
|
+
* production packets, dataset manifests, etc.) don't have to re-derive
|
|
234
|
+
* canonicalize+sha256 from scratch.
|
|
235
|
+
*
|
|
236
|
+
* Stable across:
|
|
237
|
+
* - object key insertion order (canonicalization sorts keys recursively)
|
|
238
|
+
* - encoder choice (UTF-8 via TextEncoder, fixed)
|
|
239
|
+
* - runtime (uses the Web Crypto subtle digest, present in Node ≥18 and browsers)
|
|
240
|
+
*
|
|
241
|
+
* Named `hashJson` to disambiguate from `prompt-registry.ts`'s `hashContent`,
|
|
242
|
+
* which takes a string input and returns a truncated 12-char prompt id.
|
|
243
|
+
* Use `hashJson` when you mean "canonicalize then hash."
|
|
244
|
+
*
|
|
245
|
+
* @example
|
|
246
|
+
* const hash = await hashJson({ id: '1', kind: 'spec' })
|
|
247
|
+
* // 'a3f1...' (64 hex chars)
|
|
248
|
+
*/
|
|
249
|
+
declare function hashJson<T>(obj: T): Promise<string>;
|
|
250
|
+
/**
|
|
251
|
+
* Sign a manifest with a SHA-256 content hash.
|
|
252
|
+
*
|
|
253
|
+
* The hash covers the canonicalized manifest with the `contentHash`
|
|
254
|
+
* and `algo` fields stripped; this lets verifiers re-sign the rest and
|
|
255
|
+
* compare. Returned manifest always carries `algo: 'sha256-content'`
|
|
256
|
+
* so downstream consumers can identify the scheme; manifests without
|
|
257
|
+
* `algo` still verify because it is stripped before hashing on both sides.
|
|
258
|
+
*/
|
|
259
|
+
declare function signManifest(m: HypothesisManifest): Promise<SignedManifest>;
|
|
260
|
+
/**
|
|
261
|
+
* Verify that a signed manifest has not been tampered with.
|
|
262
|
+
*
|
|
263
|
+
* Strips `contentHash` and `algo` before re-signing so manifests without
|
|
264
|
+
* `algo` verify identically to ones that carry it.
|
|
265
|
+
*/
|
|
266
|
+
declare function verifyManifest(m: SignedManifest): Promise<boolean>;
|
|
267
|
+
/**
|
|
268
|
+
* Evaluate a pre-registered hypothesis against observed results.
|
|
269
|
+
* Mechanical — no re-interpretation permitted.
|
|
270
|
+
*/
|
|
271
|
+
declare function evaluateHypothesis(manifest: SignedManifest, observed: {
|
|
272
|
+
n: number;
|
|
273
|
+
effect: number;
|
|
274
|
+
pValue: number;
|
|
275
|
+
}): Promise<HypothesisResult>;
|
|
276
|
+
//#endregion
|
|
277
|
+
//#region src/campaign/gates/sequential.d.ts
|
|
278
|
+
type SequentialDecision = 'promote' | 'continue' | 'undecided-at-maxN';
|
|
279
|
+
interface SequentialObservation {
|
|
280
|
+
decision: SequentialDecision;
|
|
281
|
+
/** Current e-value (the betting wealth) against H0. */
|
|
282
|
+
eValue: number;
|
|
283
|
+
/** Paired deltas consumed so far. */
|
|
284
|
+
n: number;
|
|
285
|
+
/** Names the decision basis. For 'undecided-at-maxN' it states explicitly
|
|
286
|
+
* that exhausting the budget is NOT evidence of no effect. */
|
|
287
|
+
reason: string;
|
|
288
|
+
}
|
|
289
|
+
interface SequentialPairedGateOptions {
|
|
290
|
+
/** Type-I budget. With `preRegistration` bound this MUST match
|
|
291
|
+
* `manifest.alpha` (conflict throws). Default 0.05. */
|
|
292
|
+
alpha?: number;
|
|
293
|
+
/** Minimum paired deltas before a promote may fire. The stopping rule is
|
|
294
|
+
* "first n ≥ minN with e-value ≥ 1/alpha" — still a valid stopping time.
|
|
295
|
+
* Default 5. */
|
|
296
|
+
minN?: number;
|
|
297
|
+
/** Pre-registered observation budget. Required unless `preRegistration`
|
|
298
|
+
* supplies it via `preRegisteredN` (conflict throws). */
|
|
299
|
+
maxN?: number;
|
|
300
|
+
/** Bet truncation forwarded to `eProcess`. Default 0.5. */
|
|
301
|
+
maxBet?: number;
|
|
302
|
+
/** Bound on |delta| in the judge's native scale; deltas are mapped to
|
|
303
|
+
* x = (d/scale + 1)/2 ∈ [0,1]. A delta outside ±scale throws (use
|
|
304
|
+
* `detectScale` to pick 1 vs 100 BEFORE streaming). Default 1. */
|
|
305
|
+
scale?: number;
|
|
306
|
+
/** Seed for the data-independent shuffle of paired deltas in `decide(ctx)`
|
|
307
|
+
* (exchangeability guard). Default 1337. */
|
|
308
|
+
shuffleSeed?: number;
|
|
309
|
+
/** Bind the pre-registered hypothesis. Verified (content hash) at
|
|
310
|
+
* construction; alpha/maxN/direction/minEffect come FROM the manifest. */
|
|
311
|
+
preRegistration?: SignedManifest;
|
|
312
|
+
/** Override the gate name in reports. */
|
|
313
|
+
name?: string;
|
|
314
|
+
}
|
|
315
|
+
interface SequentialPairedGate<TArtifact = unknown, TScenario extends Scenario = Scenario> extends Gate<TArtifact, TScenario> {
|
|
316
|
+
/** Streaming entry point: feed one paired per-scenario delta
|
|
317
|
+
* (candidate − baseline, native scale). Each gate instance carries ONE
|
|
318
|
+
* observe-stream; `decide(ctx)` runs on its own fresh stream and never
|
|
319
|
+
* consumes or advances this one. 'promote' is sticky; observing past the
|
|
320
|
+
* pre-registered maxN throws (extending a finished stream after seeing
|
|
321
|
+
* the result reopens optional stopping — start a NEW pre-registered
|
|
322
|
+
* test). */
|
|
323
|
+
observe(delta: number): SequentialObservation;
|
|
324
|
+
/** Read-only snapshot of the observe-stream. */
|
|
325
|
+
state(): EProcessState & {
|
|
326
|
+
decision: SequentialDecision;
|
|
327
|
+
};
|
|
328
|
+
}
|
|
329
|
+
/**
|
|
330
|
+
* Anytime-valid sequential paired gate. Conforms to the existing `Gate`
|
|
331
|
+
* contract (`decide(ctx)` consumes candidate vs baseline judge scores via
|
|
332
|
+
* `pairHoldout` — same pairing granularity as the fixed-n gates: full cellId,
|
|
333
|
+
* never scenarioId) and adds a streaming `observe(delta)` entry for campaigns
|
|
334
|
+
* that score cells incrementally and want to stop mid-stream.
|
|
335
|
+
*
|
|
336
|
+
* Decision mapping onto the substrate's five-valued `GateDecision`:
|
|
337
|
+
* - 'promote' → 'ship'
|
|
338
|
+
* - 'continue' → 'need_more_work' (stream ended before maxN with
|
|
339
|
+
* the e-value undecided — more reps could decide)
|
|
340
|
+
* - 'undecided-at-maxN' → 'hold', with the reason stating it is NOT
|
|
341
|
+
* evidence of no effect (never a silent default)
|
|
342
|
+
*/
|
|
343
|
+
declare function sequentialPairedGate<TArtifact = unknown, TScenario extends Scenario = Scenario>(options: SequentialPairedGateOptions): SequentialPairedGate<TArtifact, TScenario>;
|
|
344
|
+
interface SequentialDecideOptions {
|
|
345
|
+
/** Type-I budget for the early-stop evidence. Default 0.05. */
|
|
346
|
+
alpha?: number;
|
|
347
|
+
/** Minimum paired deltas before a stop may fire. Default 5. */
|
|
348
|
+
minN?: number;
|
|
349
|
+
/** Bet truncation forwarded to `eProcess`. Default 0.5. */
|
|
350
|
+
maxBet?: number;
|
|
351
|
+
/** Bound on |per-scenario composite delta|. Default 1. */
|
|
352
|
+
scale?: number;
|
|
353
|
+
}
|
|
354
|
+
interface SequentialDecideFn {
|
|
355
|
+
(args: {
|
|
356
|
+
history: GenerationRecord[];
|
|
357
|
+
}): {
|
|
358
|
+
stop: boolean;
|
|
359
|
+
reason?: string;
|
|
360
|
+
};
|
|
361
|
+
/** Read-only snapshot of the accumulated e-process (observability + tests). */
|
|
362
|
+
state(): EProcessState;
|
|
363
|
+
}
|
|
364
|
+
/**
|
|
365
|
+
* `SurfaceProposer.decide` adapter — stops the optimization loop the moment
|
|
366
|
+
* the e-process decides the loop has produced a real improvement, instead of
|
|
367
|
+
* always running `maxGenerations`.
|
|
368
|
+
*
|
|
369
|
+
* Stream: for each generation g ≥ 1, the per-scenario composite deltas of
|
|
370
|
+
* generation g's top candidate vs the generation-0 top candidate (the
|
|
371
|
+
* incumbent the loop set out to beat), paired by scenarioId. H0: no proposed
|
|
372
|
+
* surface improves any scenario's expected composite over the incumbent —
|
|
373
|
+
* under it every delta has conditional mean ≤ 0 and the e-process is valid.
|
|
374
|
+
* Once wealth ≥ 1/alpha the loop stops and hands the winner to the promotion
|
|
375
|
+
* gate (which re-scores on HELD-OUT data — this adapter only spends the
|
|
376
|
+
* exploration budget, it never promotes).
|
|
377
|
+
*
|
|
378
|
+
* Honesty caveats: (1) the incumbent's scores are measured once and shared
|
|
379
|
+
* across all generations' deltas, so type-I control is exact only insofar as
|
|
380
|
+
* those scores approximate the incumbent's true per-scenario means (more reps
|
|
381
|
+
* → tighter); (2) an UNDECIDED process never stops the loop — absence of a
|
|
382
|
+
* crossing is NOT evidence of no effect, so the loop simply runs its normal
|
|
383
|
+
* course. Calling the adapter repeatedly with a growing history consumes each
|
|
384
|
+
* generation exactly once (re-feeding an already-seen record would double-count
|
|
385
|
+
* evidence).
|
|
386
|
+
*/
|
|
387
|
+
declare function sequentialDecide(options?: SequentialDecideOptions): SequentialDecideFn;
|
|
388
|
+
//#endregion
|
|
389
|
+
//#region src/campaign/gates/statistical-heldout.d.ts
|
|
390
|
+
interface PairedHoldout {
|
|
391
|
+
/** Baseline scalar per paired cell (same order as `after`/`cellIds`). */
|
|
392
|
+
before: number[];
|
|
393
|
+
/** Candidate scalar per paired cell. */
|
|
394
|
+
after: number[];
|
|
395
|
+
/** The full cellIds (`scenario:rep`) that paired, in order. */
|
|
396
|
+
cellIds: string[];
|
|
397
|
+
}
|
|
398
|
+
/**
|
|
399
|
+
* Pair candidate vs baseline holdout observations by FULL cellId. `select`
|
|
400
|
+
* pulls the scalar from a cell's judge reports (composite, or a named
|
|
401
|
+
* dimension); a cell contributes the mean of `select` across its judges. Cells
|
|
402
|
+
* whose scenario is not in `scenarioIds`, or where `select` is undefined for
|
|
403
|
+
* every judge on either side, are skipped on BOTH sides so the arrays stay
|
|
404
|
+
* paired. Throws when the two maps disagree on which holdout cells exist — a
|
|
405
|
+
* load-bearing invariant: the baseline + winner holdout campaigns run the same
|
|
406
|
+
* scenarios with the same seed base, so their cellIds MUST align; a mismatch
|
|
407
|
+
* means a silent pairing bug, not a soft fallback.
|
|
408
|
+
*/
|
|
409
|
+
declare function pairHoldout(candidate: Map<string, Record<string, JudgeScore>>, baseline: Map<string, Record<string, JudgeScore>>, scenarioIds: Set<string>, select: (s: JudgeScore) => number | undefined): PairedHoldout;
|
|
410
|
+
interface HeldoutSignificance {
|
|
411
|
+
paired: PairedHoldout;
|
|
412
|
+
/**
|
|
413
|
+
* The paired bootstrap on the requested statistic (MEAN by default — see the
|
|
414
|
+
* tie note on `heldoutSignificance`).
|
|
415
|
+
*
|
|
416
|
+
* DIAGNOSTIC, not necessarily the interval the verdict keyed on. On a
|
|
417
|
+
* two-point (pass/fail) outcome the decision routes to Tango's score interval
|
|
418
|
+
* instead, because a percentile bootstrap of the mean over a three-atom
|
|
419
|
+
* lattice is not a valid interval at a nonzero margin. Read
|
|
420
|
+
* `decision.low`/`decision.high` for the interval that actually decided, and
|
|
421
|
+
* `decisionStatistic` for which one it is.
|
|
422
|
+
*/
|
|
423
|
+
bootstrap: PairedBootstrapResult;
|
|
424
|
+
/** The MEDIAN paired-delta bootstrap, reported as a diagnostic. When many
|
|
425
|
+
* scenarios are tied (both sides solve them), the median is pinned near 0
|
|
426
|
+
* regardless of the mean lift — comparing the two exposes tie-domination. */
|
|
427
|
+
medianBootstrap: PairedBootstrapResult;
|
|
428
|
+
/**
|
|
429
|
+
* The full promotion decision: which estimator the outcome's shape admits,
|
|
430
|
+
* the interval it produced, McNemar's exact veto on the two-point path, and
|
|
431
|
+
* whether the interval was zero-width (no evidence in either direction). The
|
|
432
|
+
* single source of `significant`.
|
|
433
|
+
*/
|
|
434
|
+
decision: PairedPromotionDecision;
|
|
435
|
+
/** Which paired estimator the verdict was decided on. */
|
|
436
|
+
decisionStatistic: PairedDecisionStatistic;
|
|
437
|
+
/** McNemar's exact evidence on the two-point path; null otherwise. */
|
|
438
|
+
mcnemar: PairedMcNemarEvidence | null;
|
|
439
|
+
/** Fraction of paired observations that are exact ties (|delta| < 1e-9). A
|
|
440
|
+
* high tie fraction is WHY a median-based gate would have missed a real lift;
|
|
441
|
+
* it is the observability the tie fix adds. */
|
|
442
|
+
tieFraction: number;
|
|
443
|
+
/** n paired observations. */
|
|
444
|
+
n: number;
|
|
445
|
+
/** Effective minimum after applying the bootstrap's hard statistical floor. */
|
|
446
|
+
minimumRequired: number;
|
|
447
|
+
/** Statistical method that carried the decision. */
|
|
448
|
+
decisionMethod: PairedDecisionMethod;
|
|
449
|
+
/** Exact one-sided p-value on the small-sample path; otherwise null. */
|
|
450
|
+
pValue: number | null;
|
|
451
|
+
/** True iff n >= minimumRequired, the DECIDING interval has nonzero width,
|
|
452
|
+
* its lower bound clears the threshold, and McNemar's exact test does not
|
|
453
|
+
* veto at a non-negative threshold. */
|
|
454
|
+
significant: boolean;
|
|
455
|
+
/** Set when n < minimumRequired — too little evidence to claim significance. */
|
|
456
|
+
fewRuns: boolean;
|
|
457
|
+
}
|
|
458
|
+
interface HeldoutSignificanceOptions {
|
|
459
|
+
deltaThreshold?: number;
|
|
460
|
+
minProductiveRuns?: number;
|
|
461
|
+
confidence?: number;
|
|
462
|
+
resamples?: number;
|
|
463
|
+
/** Fixed by default for a deterministic, reproducible gate verdict. */
|
|
464
|
+
seed?: number;
|
|
465
|
+
statistic?: 'mean' | 'median';
|
|
466
|
+
}
|
|
467
|
+
/**
|
|
468
|
+
* Significance of the held-out composite lift: ship only when the lower bound
|
|
469
|
+
* of the interval the outcome's shape ADMITS exceeds `deltaThreshold` (default
|
|
470
|
+
* 0 ⇒ "confidently positive"). Interpret `deltaThreshold` in the judge's native
|
|
471
|
+
* scale.
|
|
472
|
+
*
|
|
473
|
+
* The decision is delegated whole to {@link decidePairedPromotion}, the one
|
|
474
|
+
* copy of the rule (`src/paired-promotion-decision.ts`), which `HeldOutGate`
|
|
475
|
+
* also calls. That module's header carries the measurements; the short version
|
|
476
|
+
* is three guards a bare `bootstrap.low > threshold` does not have:
|
|
477
|
+
*
|
|
478
|
+
* - a two-point (pass/fail) outcome decides on Tango's SCORE interval, the
|
|
479
|
+
* only paired-binary construction that stays valid at a nonzero margin;
|
|
480
|
+
* - McNemar's exact test VETOES at any non-negative threshold;
|
|
481
|
+
* - a ZERO-WIDTH interval is refused rather than promoted, in either
|
|
482
|
+
* direction — [0,0] clears every negative threshold and [g,g] clears every
|
|
483
|
+
* threshold below g, and both are an absence of evidence, not a result.
|
|
484
|
+
*
|
|
485
|
+
* Measured on this function before those guards landed, at a nominal 5 %:
|
|
486
|
+
* 14.60 % false promotion at n = 40 on a paired-binary noninferiority boundary,
|
|
487
|
+
* and 88.50 % at n = 6 under a bounded asymmetric null whose true mean paired
|
|
488
|
+
* delta is exactly 0.
|
|
489
|
+
*
|
|
490
|
+
* At small n, where the percentile bootstrap is descriptive only, a
|
|
491
|
+
* pre-registered exact sign test still carries the bootstrap path.
|
|
492
|
+
*/
|
|
493
|
+
declare function heldoutSignificance(paired: PairedHoldout, opts?: HeldoutSignificanceOptions): HeldoutSignificance;
|
|
494
|
+
interface DimensionRegression {
|
|
495
|
+
dimension: string;
|
|
496
|
+
/** Paired bootstrap on (candidate − baseline). DIAGNOSTIC on a pass/fail
|
|
497
|
+
* dimension, where `ci` carries the interval that decided instead. */
|
|
498
|
+
bootstrap: PairedBootstrapResult;
|
|
499
|
+
/** Which paired statistic `bootstrap.low` is the lower bound of. `'mean'`
|
|
500
|
+
* unless the caller asked for the median. `bootstrap.median` still carries
|
|
501
|
+
* the median point estimate either way. */
|
|
502
|
+
bootstrapStatistic: 'median' | 'mean';
|
|
503
|
+
/** The interval `regressed` was decided on, in the dimension's native units. */
|
|
504
|
+
ci: {
|
|
505
|
+
low: number;
|
|
506
|
+
high: number;
|
|
507
|
+
};
|
|
508
|
+
/** Which estimator produced `ci`. */
|
|
509
|
+
decisionStatistic: PairedDecisionStatistic;
|
|
510
|
+
/** McNemar's exact evidence on a pass/fail dimension; null otherwise. */
|
|
511
|
+
mcnemar: PairedMcNemarEvidence | null;
|
|
512
|
+
/** `ci` has zero width — no evidence in either direction. */
|
|
513
|
+
indeterminate: boolean;
|
|
514
|
+
/** True iff the candidate may have regressed this dimension by more than
|
|
515
|
+
* tolerance: the lower bound of the DECIDING interval on (candidate −
|
|
516
|
+
* baseline) is below −tolerance, OR the exact small-sample test proves a drop
|
|
517
|
+
* past tolerance. */
|
|
518
|
+
regressed: boolean;
|
|
519
|
+
tolerance: number;
|
|
520
|
+
n: number;
|
|
521
|
+
}
|
|
522
|
+
/** Detect the native scale of a set of scores: 0-100 when any magnitude clears
|
|
523
|
+
* 1.5, else [0,1]. Used to auto-scale the regression tolerance so a default
|
|
524
|
+
* expressed for [0,1] is not silently a no-op on a 0-100 dimension. */
|
|
525
|
+
declare function detectScale(values: number[]): 1 | 100;
|
|
526
|
+
/** Per-critical-dimension regression guard. For each dimension, pair the
|
|
527
|
+
* candidate vs baseline values by full cellId and bootstrap the paired delta;
|
|
528
|
+
* a dimension is "regressed" when the CI lower bound < −tolerance (conservative
|
|
529
|
+
* — blocks if the credible worst case exceeds tolerance, which is the right
|
|
530
|
+
* posture for safety dimensions like `hallucination_free`). When `tolerance`
|
|
531
|
+
* is omitted it auto-scales: 0.05 on [0,1], 5 on 0-100.
|
|
532
|
+
*
|
|
533
|
+
* The interval comes from {@link decidePairedPromotion}, so a pass/fail
|
|
534
|
+
* dimension is judged on Tango's score interval rather than a percentile
|
|
535
|
+
* bootstrap of the mean — `tolerance` is a NONZERO margin, and the bootstrap
|
|
536
|
+
* is not a valid interval at one. That matters most here because this guard
|
|
537
|
+
* fails OPEN by construction: `tolerance` is positive, so an interval pinned at
|
|
538
|
+
* [0,0] never satisfies `low < −tolerance` and a real regression on a safety
|
|
539
|
+
* dimension would be reported as `regressed: false`. On the median it fails the
|
|
540
|
+
* same way for the same reason — when most pairs tie, which is automatic for a
|
|
541
|
+
* pass/fail dimension on {0,1} and on the 0-100 encoding `detectScale` exists
|
|
542
|
+
* to support, the median CI collapses to [0,0]. Pass `statistic: 'median'` to
|
|
543
|
+
* restore the pre-0.134 behaviour. */
|
|
544
|
+
declare function dimensionRegressions(candidate: Map<string, Record<string, JudgeScore>>, baseline: Map<string, Record<string, JudgeScore>>, scenarioIds: Set<string>, criticalDimensions: string[], opts?: {
|
|
545
|
+
tolerance?: number;
|
|
546
|
+
confidence?: number;
|
|
547
|
+
resamples?: number;
|
|
548
|
+
seed?: number;
|
|
549
|
+
/** Paired statistic the CI is computed on. Default `'mean'` — see
|
|
550
|
+
* {@link DECISION_PAIRED_DELTA_STATISTIC} for why the median is not. */
|
|
551
|
+
statistic?: 'mean' | 'median';
|
|
552
|
+
}): DimensionRegression[];
|
|
553
|
+
//#endregion
|
|
554
|
+
export { PairArmsResult as A, hashJson as C, MatchedPair as D, ComparePairedArmsOptions as E, PairedMetricDelta as F, comparePairedArms as I, pairArms as L, PairedArmRow as M, PairedArmsComparison as N, MatchedRunRecordPair as O, PairedCorrectness as P, pairRunRecords as R, evaluateHypothesis as S, verifyManifest as T, HypothesisManifest as _, detectScale as a, SignedManifestAlgo as b, pairHoldout as c, SequentialDecision as d, SequentialObservation as f, sequentialPairedGate as g, sequentialDecide as h, PairedHoldout as i, PairRunRecordsResult as j, PairArmsOptions as k, SequentialDecideFn as l, SequentialPairedGateOptions as m, HeldoutSignificance as n, dimensionRegressions as o, SequentialPairedGate as p, HeldoutSignificanceOptions as r, heldoutSignificance as s, DimensionRegression as t, SequentialDecideOptions as u, HypothesisResult as v, signManifest as w, canonicalize as x, SignedManifest as y };
|
|
555
|
+
//# sourceMappingURL=statistical-heldout-Dn9ruizm.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"statistical-heldout-Dn9ruizm.d.ts","names":[],"sources":["../src/paired-arms.ts","../src/pre-registration.ts","../src/campaign/gates/sequential.ts","../src/campaign/gates/statistical-heldout.ts"],"mappings":";;;;;;;UAwCiB;;;EAGf;;;;;EAKA;;EAEA;;EAEA;;EAEA,UAAU;;UAGK;;EAEf;;EAEA;;;UAIe;EACf;;;;EAIA;EACA,UAAU;EACV,WAAW;;UAGI;;EAEf,OAAO;;;EAGP,kBAAkB;;EAElB,mBAAmB;;;;;;;;;;;;;;;;;;;;iBAqBL,SAAS,eAAe,gBAAgB,MAAM,kBAAkB;;UA8G/D;;EAEf;;EAEA;;EAEA,SAAS;;EAET,gBAAgB;;;UAID;EACf;;EAEA;;EAEA;;EAEA;;EAEA;;;;EAIA,aAAa;;EAEb;IAAY;IAAW;;;UAGR,iCAAiC;;;;EAIhD;;EAEA,YAAY;;UAGG;EACf;EACA;EACA;;;EAGA,aAAa;EACb,cAAc;;;;;;;;;;;;;;;iBAgBA,kBACd,eAAe,gBACf,MAAM,2BACL;UAmEc;EACf;EACA;EACA,UAAU;EACV,WAAW;;UAGI;EACf,OAAO;EACP,kBAAkB;EAClB,mBAAmB;;;;;;;;;iBAeL,eACd,uBAAuB,aACvB,wBAAwB,cACvB;;;;;;;;;;;;;UC1Wc;EACf;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;EACA;;;;;;;;;;KAWU;UAEK,uBAAuB;;EAEtC;;;;;;;;;EASA,OAAO;;UAGQ;EACf,UAAU;EACV;EACA;EACA;;;EAGA;;EAEA,kBAAkB;EAGlB;;;;;;;;;;;iBAYc,aAAa;;;;;;;;;;;;;;;;;;;;;;iBA8BP,SAAS,GAAG,KAAK,IAAI;;;;;;;;;;iBAkBrB,aAAa,GAAG,qBAAqB,QAAQ;;;;;;;iBAW7C,eAAe,GAAG,iBAAiB;;;;;iBAWnC,mBACpB,UAAU,gBACV;EAAY;EAAW;EAAgB;IACtC,QAAQ;;;KCpHC;UAEK;EACf,UAAU;;EAEV;;EAEA;;;EAGA;;UAGe;;;EAGf;;;;EAIA;;;EAGA;;EAEA;;;;EAIA;;;EAGA;;;EAGA,kBAAkB;;EAElB;;UAGe,qBAAqB,qBAAqB,kBAAkB,WAAW,kBAC9E,KAAK,WAAW;;;;;;;;EAQxB,QAAQ,gBAAgB;;EAExB,SAAS;IAAkB,UAAU;;;;;;;;;;;;;;;;;iBA0MvB,qBAAqB,qBAAqB,kBAAkB,WAAW,UACrF,SAAS,8BACR,qBAAqB,WAAW;UAsFlB;;EAEf;;EAEA;;EAEA;;EAEA;;UAGe;GACd;IAAQ,SAAS;;IAAyB;IAAe;;;EAE1D,SAAS;;;;;;;;;;;;;;;;;;;;;;;;;iBA0BK,iBAAiB,UAAS,0BAA+B;;;UCtXxD;;EAEf;;EAEA;;EAEA;;;;;;;;;;;;;iBAcc,YACd,WAAW,YAAY,eAAe,cACtC,UAAU,YAAY,eAAe,cACrC,aAAa,aACb,SAAS,GAAG,oCACX;UAsDc;EACf,QAAQ;;;;;;;;;;;;EAYR,WAAW;;;;EAIX,iBAAiB;;;;;;;EAOjB,UAAU;;EAEV,mBAAmB;;EAEnB,SAAS;;;;EAIT;;EAEA;;EAEA;;EAEA,gBAAgB;;EAEhB;;;;EAIA;;EAEA;;UAGe;EACf;EACA;EACA;EACA;;EAEA;EACA;;;;;;;;;;;;;;;;;;;;;;;;;;;;iBA6Bc,oBACd,QAAQ,eACR,OAAM,6BACL;UAkEc;EACf;;;EAGA,WAAW;;;;EAIX;;EAEA;IAAM;IAAa;;;EAEnB,mBAAmB;;EAEnB,SAAS;;EAET;;;;;EAKA;EACA;EACA;;;;;iBAMc,YAAY;;;;;;;;;;;;;;;;;;;iBAsBZ,qBACd,WAAW,YAAY,eAAe,cACtC,UAAU,YAAY,eAAe,cACrC,aAAa,aACb,8BACA;EACE;EACA;EACA;EACA;;;EAGA;IAED"}
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { g as JudgeScore } from "./types-
|
|
1
|
+
import { g as JudgeScore } from "./types-D216SgwM.js";
|
|
2
2
|
//#region src/judge-calibration.d.ts
|
|
3
3
|
/**
|
|
4
4
|
* Judge calibration — measure judge quality against human gold + bias.
|
|
@@ -965,4 +965,4 @@ declare function eProcess(opts?: EProcessOptions): EProcess;
|
|
|
965
965
|
declare function mulberry32(seed: number): () => number;
|
|
966
966
|
//#endregion
|
|
967
967
|
export { pairedCohensDz as $, WeightedCompositeInput as A, positionalBias as At, eProcess as B, RankTestMethod as C, GoldenItem as Ct, ScoreRiskDifferenceResult as D, calibrateJudge as Dt, RiskDifferenceResult as E, VerbosityBiasResult as Et, cliffsDelta as F, mannWhitneyU as G, interRaterReliability as H, cohensD as I, mcnemarRequiredN as J, mcnemar as K, confidenceInterval as L, WilcoxonSignedRankResult as M, verbosityBias as Mt, benjaminiHochberg as N, SignTestAlternative as O, calibrateJudgeContinuous as Ot, bonferroni as P, pairedBootstrap as Q, corpusInterRaterAgreement as R, ProportionInterval as S, ContinuousCalibrationResult as St, RankTestOptions as T, SelfPreferenceResult as Tt, interpretCliffs as U, holm as V, isBinaryOutcomeVector as W, normalizeScores as X, mulberry32 as Y, pairedBinaryScale as Z, McNemarResult as _, wilson as _t, CorpusAgreementReport as a, pairedSignTest as at, PairedSignTestResult as b, ContinuousAgreement as bt, DEFAULT_PERMUTATIONS as c, passAtK as ct, EProcessState as d, requiredPairedSampleSize as dt, pairedDeltaTieFraction as et, EProcessStep as f, requiredSampleSize as ft, MannWhitneyResult as g, wilcoxonSignedRank as gt, MANN_WHITNEY_EXACT_MAX_WORK as h, weightedMean as ht, CorpusAgreementPerDimension as i, pairedRiskDifferenceScore as it, WeightedCompositeResult as j, selfPreference as jt, WILCOXON_EXACT_MAX_N as k, continuousAgreement as kt, EProcess as l, pearsonR as lt, MANN_WHITNEY_EXACT_MAX_STATES as m, weightedComposite as mt, CliffsMagnitude as n, pairedRiskDifference as nt, CorpusScoreRecord as o, pairedTTest as ot, ExactRiskDifferenceResult as p, spearmanR as pt, mcnemarPower as q, CorpusAgreementOptions as r, pairedRiskDifferenceExact as rt, DECISION_PAIRED_DELTA_STATISTIC as s, partialCredit as st, BOOTSTRAP_GATE_MIN_N as t, pairedMde as tt, EProcessOptions as u, ranks as ut, PairedBootstrapOptions as v, CalibrationResult as vt, RankTestMethodRequest as w, PositionalBiasResult as wt, PairedTTestResult as x, ContinuousAgreementOptions as xt, PairedBootstrapResult as y, CandidateScore as yt, corpusInterRaterAgreementFromJudgeScores as z };
|
|
968
|
-
//# sourceMappingURL=statistics-
|
|
968
|
+
//# sourceMappingURL=statistics-D6Uebe_4.d.ts.map
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"statistics-
|
|
1
|
+
{"version":3,"file":"statistics-D6Uebe_4.d.ts","names":[],"sources":["../src/judge-calibration.ts","../src/statistics.ts"],"mappings":";;;;;;;;;;;;;;;;;;;;;;UAuBiB;EACf;EACA;;EAEA;;UAGe;EACf;EACA;;EAEA;;UAGe;EACf;EACA;;EAEA;;EAEA;;EAEA,YAAY;IAAQ;IAAgB;IAAe;IAAe;;;;;;iBAMpD,eACd,QAAQ,cACR,WAAW,mBACV;UA0Bc;;;;;EAKf;EACA;;;;;;iBAOc,eAAe,QAAQ,mBAAmB;UAgBzC;;EAEf;EACA;;iBAGc,cACd,SAAS;EAAQ;EAAmB;KACnC;UAYc;;EAEf;EACA;EACA;EACA;;;;;;;iBAQc,eACd,SAAS;EAAQ;EAAe;KAC/B;UA8Ec;;EAEf;;EAEA;;EAEA;;EAEA;;EAEA;IACE;IACA;;;EAGF;;EAEA;;UAGe;;EAEf;;EAEA;;EAEA;;EAEA;;;;;;;;;iBAUc,oBACd,oBACA,OAAM,6BACL;UAqEc,oCAAoC;;EAEnD;;EAEA;EACA;EACA;IACE;IACA;;;;;;;iBAQY,yBACd,QAAQ,cACR,WAAW,kBACX,OAAM,6BACL;;;;;cCnVU,kBAAe,QAAY,iBAAe;;iBAGvC,aAAa;EAAU;EAAe;;;;;;;;;;iBAoBtC,mBACd,kBACA,qBACA;EAAQ;EAAe;;EACpB;EAAc;EAAe;;;;;;;;;;;;;;;;;;;iBAiDlB,sBAAsB,aAAa;;KAkFvC;;;;;KAMA;UAEK;;EAEf,SAAS;;EAET;;;EAGA;;;cAIW;;cAEA;;cAEA;;cAEA;UAEI;;EAEf;;EAEA;;EAEA;;EAEA,QAAQ;;EAER;;;;;;;;;;;;;;iBAec,aACd,aACA,aACA,OAAM,kBACL;;iBAwFa,cAAc,iBAAiB;UAK9B;;EAEf;EACA;;EAEA;;;;;;;;;;;;;;;;iBAiBc,YAAY,kBAAkB,kBAAkB;UAyB/C;;;EAGf;;EAEA;;EAEA,QAAQ;;EAER;;EAEA;;;;;;;;;;;;;;;iBAgBc,mBACd,kBACA,iBACA,OAAM,kBACL;;;;;;;;;;;;;iBAmFa,QAAQ,aAAa;;;;;;;;;iBAqBrB,eAAe,kBAAkB;KAoBrC;;;;;;;;;;;iBAYI,YAAY,kBAAkB;;;;;;iBAkB9B,gBAAgB,gBAAgB;;;;;iBAuBhC,MAAM;;;;;;iBAmBN,SAAS,aAAa;;;;;iBAuBtB,UAAU,aAAa;UAKtB;;EAEf,MAAM;;;;;EAKN,SAAS;;EAET;;UAGe;EACf;EACA;;;;;;;;;;;iBAYc,kBAAkB,OAAO,yBAAyB;UA4CjD;;EAEf;;EAEA;;EAEA;;EAEA;;UAGe,oCAAoC;EACnD;;EAEA;;EAEA;;UAGe;;EAEf,cAAc;;EAEd;;EAEA;;EAEA;;EAEA;;UAGe,+BAA+B;;;;;;EAM9C;;;;;;EAMA;;;;;;;;;;;;;;;;;;iBAmBc,0BACd,SAAS,qBACT,OAAM,yBACL;;;;;;;;;iBA0Ha,yCACd,aAAa;EAAQ;EAAgB,QAAQ;IAC7C,OAAM,yBACL;;;;;;;iBA8Ba,mBAAmB;EACjC;EACA;EACA;EACA;;;;;;;;;;;iBAsBc,yBAAyB;EACvC;EACA;EACA;EACA;;;;;;;;iBAkBc,UAAU;EACxB;EACA;EACA;EACA;;;;;;;;;;;;;;;;iBAyBc,iBAAiB;EAC/B;EACA;EACA;EACA;EACA;;;;;;;iBAyBc,aAAa;EAC3B;EACA;EACA;EACA;EACA;;;;;;;;;iBAwBc,WACd,4BACA;EACG;EAAoB;;;;;;;;;;iBAgBT,KACd,4BACA;EACG;EAAoB;;;;;;;;;iBA4BT,kBACd,4BACA;EACG;EAAmB;;UAuCP;;EAEf;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;;;;;;;;;;;cAaW;UAEI;;EAEf;;EAEA;;EAEA;;;EAGA;;;;;;;;;;;;iBAac,gBACd,kBACA,iBACA,OAAM,yBACL;;KA0DS;;UAGK;;EAEf;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA,aAAa;;EAEb;;;;;;;;;;;;;;;iBAgBc,eACd,gCACA,aAAa,sBACZ;;UAgDc;;EAEf;;EAEA;;EAEA;;;;;;;;;iBAUc,OAAO,mBAAmB,WAAW,sBAAoB;;;;;;;;;;;;;;;;;;;;;;;;;;iBA2CzD,sBAAsB,QAAQ;;UAU7B;;EAEf;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;;;;;;;;;;;;;iBAec,QACd,SAAS,6BACT,WAAW,8BACV;;UAmBc;;EAEf;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;;;;;;;;;;;;;;;;iBAkBc,qBACd,SAAS,6BACT,WAAW,6BACX,sBACC;;;;UAkCc;;EAEf;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;iBA+Bc,0BACd,SAAS,6BACT,WAAW,6BACX,sBACC;;;UAyEc;;EAEf;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;;;;;;;;;;;;;;;;;;;;;;;;;;iBA8Dc,0BACd,SAAS,6BACT,WAAW,6BACX,sBACC;;;;;;;;;;;;;;;;;;;;iBA6Ea,kBACd,QAAQ,mBACR,OAAO;;;iBAkBO,uBACd,QAAQ,mBACR,OAAO;;;;;;;;;;;;;;;;;;;;;;;;;;;;;cA4CI;;;;;;;;;;;iBAYG,QAAQ,WAAW,WAAW;UAgE7B;;;EAGf;;;;EAIA;;;;EAIA;;UAGe;;EAEf;;EAEA;;EAEA;;UAGe,sBAAsB;EACrC;EACA;EACA;;EAEA;;EAEA;;UAGe;;;EAGf,OAAO,YAAY;EACnB,SAAS;;;;;;;;;;;;;;;;;;;;;;;;;;;;;iBA8BK,SAAS,OAAM,kBAAuB;;;;;iBAsatC,WAAW"}
|
|
@@ -0,0 +1,51 @@
|
|
|
1
|
+
//#region src/trajectory-replay/steps.d.ts
|
|
2
|
+
/**
|
|
3
|
+
* Recorded shell-trajectory steps and the observation grammar they carry.
|
|
4
|
+
*
|
|
5
|
+
* A recorded trajectory is the action/observation sequence an agent actually
|
|
6
|
+
* ran. Scaffolds that execute one shell command per step (mini-SWE and the
|
|
7
|
+
* CodeTracer-normalized corpora built from it) tag each observation with the
|
|
8
|
+
* command's returncode and its combined output:
|
|
9
|
+
*
|
|
10
|
+
* <returncode>2</returncode>
|
|
11
|
+
* <output>
|
|
12
|
+
* …command output…
|
|
13
|
+
* </output>
|
|
14
|
+
*
|
|
15
|
+
* The parsers here are the only place that grammar is decoded. Everything
|
|
16
|
+
* downstream — replay verdicts, corpus enumeration, fix prompts — reads the
|
|
17
|
+
* returncode, the output, and the failure signature through these functions.
|
|
18
|
+
*/
|
|
19
|
+
/**
|
|
20
|
+
* One step of a recorded shell trajectory. Structural: any richer step record
|
|
21
|
+
* (file refs, thinking text, tool type) satisfies it.
|
|
22
|
+
*/
|
|
23
|
+
interface RecordedTrajectoryStep {
|
|
24
|
+
/** 1-based position in the trajectory. */
|
|
25
|
+
readonly step_id: number;
|
|
26
|
+
readonly action: string;
|
|
27
|
+
/** Null when the step recorded no observation (terminal submit steps). */
|
|
28
|
+
readonly observation: string | null;
|
|
29
|
+
}
|
|
30
|
+
/** Recorded returncode of a step, or null when the observation carries none. */
|
|
31
|
+
declare function parseRecordedReturncode(observation: string | null): number | null;
|
|
32
|
+
/** Text between the observation's <output> tags, or the raw observation when
|
|
33
|
+
* the tags are absent. */
|
|
34
|
+
declare function parseObservationOutput(observation: string | null): string;
|
|
35
|
+
/**
|
|
36
|
+
* Stable failure-signature candidate: the first line of the recorded output
|
|
37
|
+
* that contains the word "error". Null when no such line exists — a verdict
|
|
38
|
+
* then falls back to returncode-only matching and says so.
|
|
39
|
+
* Pass an explicit signature to override (compiler quote glyphs vary with
|
|
40
|
+
* locale, so a hand-picked ASCII substring is often more robust).
|
|
41
|
+
*/
|
|
42
|
+
declare function deriveFailureSignature(observation: string | null): string | null;
|
|
43
|
+
/** mini-SWE's end-of-run submit convention: the agent echoes this sentinel
|
|
44
|
+
* and dumps the diff. A label on this step marks a bad SUBMIT DECISION, not a
|
|
45
|
+
* failed command — there is no executable failure to reproduce, so it is
|
|
46
|
+
* never a counterfactual replay target. */
|
|
47
|
+
declare const SUBMIT_ACTION_SIGNATURE = "COMPLETE_TASK_AND_SUBMIT_FINAL_OUTPUT";
|
|
48
|
+
declare function isSubmitAction(action: string): boolean;
|
|
49
|
+
//#endregion
|
|
50
|
+
export { parseObservationOutput as a, isSubmitAction as i, SUBMIT_ACTION_SIGNATURE as n, parseRecordedReturncode as o, deriveFailureSignature as r, RecordedTrajectoryStep as t };
|
|
51
|
+
//# sourceMappingURL=steps-BArUxhna.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"steps-BArUxhna.d.ts","names":[],"sources":["../src/trajectory-replay/steps.ts"],"mappings":";;;;;;;;;;;;;;;;;;;;;;UAsBiB;;WAEN;WACA;;WAEA;;;iBAIK,wBAAwB;;;iBAQxB,uBAAuB;;;;;;;;iBAavB,uBAAuB;;;;;cAW1B;iBAEG,eAAe"}
|