@tangle-network/agent-eval 0.133.2 → 0.133.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (103) hide show
  1. package/CHANGELOG.md +156 -0
  2. package/dist/{analyze-runs-BmX-h_yn.d.ts → analyze-runs-BClW9OSe.d.ts} +3 -3
  3. package/dist/{analyze-runs-BmX-h_yn.d.ts.map → analyze-runs-BClW9OSe.d.ts.map} +1 -1
  4. package/dist/{analyze-runs-B-afTpCv.js → analyze-runs-qk8op0tN.js} +63 -42
  5. package/dist/analyze-runs-qk8op0tN.js.map +1 -0
  6. package/dist/baseline-BaPxoROc.js +149 -0
  7. package/dist/baseline-BaPxoROc.js.map +1 -0
  8. package/dist/{baseline-hG3K85h4.d.ts → baseline-D_fT6277.d.ts} +43 -11
  9. package/dist/baseline-D_fT6277.d.ts.map +1 -0
  10. package/dist/benchmarks/index.d.ts +1 -1
  11. package/dist/benchmarks/index.js +1 -1
  12. package/dist/{benchmarks-CJr1H1_a.js → benchmarks-BP9sgMia.js} +3 -3
  13. package/dist/{benchmarks-CJr1H1_a.js.map → benchmarks-BP9sgMia.js.map} +1 -1
  14. package/dist/builder-eval/index.js +1 -1
  15. package/dist/campaign/index.d.ts +2 -2
  16. package/dist/campaign/index.js +2 -2
  17. package/dist/{campaign-BJjn1rhw.js → campaign--V4ffEKR.js} +12 -6
  18. package/dist/{campaign-BJjn1rhw.js.map → campaign--V4ffEKR.js.map} +1 -1
  19. package/dist/{client-COvaLoQG.d.ts → client-Du7B81wW.d.ts} +28 -14
  20. package/dist/client-Du7B81wW.d.ts.map +1 -0
  21. package/dist/{client-CYzbdJOZ.js → client-LIuo-KPv.js} +19 -7
  22. package/dist/client-LIuo-KPv.js.map +1 -0
  23. package/dist/contract/index.d.ts +3 -3
  24. package/dist/contract/index.d.ts.map +1 -1
  25. package/dist/contract/index.js +9 -8
  26. package/dist/contract/index.js.map +1 -1
  27. package/dist/{eval-campaign-DXhpZghy.js → eval-campaign-CvPcvqXC.js} +2 -2
  28. package/dist/{eval-campaign-DXhpZghy.js.map → eval-campaign-CvPcvqXC.js.map} +1 -1
  29. package/dist/hosted/index.d.ts +1 -1
  30. package/dist/hosted/index.d.ts.map +1 -1
  31. package/dist/hosted/index.js +1 -1
  32. package/dist/{index-BREtv3ZZ.d.ts → index-B5MNN1f1.d.ts} +3 -3
  33. package/dist/{index-BREtv3ZZ.d.ts.map → index-B5MNN1f1.d.ts.map} +1 -1
  34. package/dist/{index-C7Wue8R6.d.ts → index-DOqvIJ8I.d.ts} +27 -10
  35. package/dist/index-DOqvIJ8I.d.ts.map +1 -0
  36. package/dist/{index-DSC51roc.d.ts → index-DSC51roc2.d.ts} +1 -1
  37. package/dist/index-DSC51roc2.d.ts.map +1 -0
  38. package/dist/{index-nhIYz9hn.d.ts → index-DuhJaaiH.d.ts} +68 -7
  39. package/dist/index-DuhJaaiH.d.ts.map +1 -0
  40. package/dist/index.d.ts +56 -10
  41. package/dist/index.d.ts.map +1 -1
  42. package/dist/index.js +29 -20
  43. package/dist/index.js.map +1 -1
  44. package/dist/ledger-core/index.d.ts +2 -2
  45. package/dist/ledger-core/index.js +2 -2
  46. package/dist/{ledger-core-CPZfcrC2.js → ledger-core-DAKFKRzi.js} +136 -18
  47. package/dist/ledger-core-DAKFKRzi.js.map +1 -0
  48. package/dist/matrix/index.d.ts +1 -1
  49. package/dist/meta-eval/index.d.ts +1 -1
  50. package/dist/meta-eval/index.js +2 -2
  51. package/dist/multishot/index.d.ts +1 -1
  52. package/dist/openapi.json +1 -1
  53. package/dist/{paired-arms-6XItKzd1.js → paired-arms-CA_8pN01.js} +2 -2
  54. package/dist/{paired-arms-6XItKzd1.js.map → paired-arms-CA_8pN01.js.map} +1 -1
  55. package/dist/pipelines/index.d.ts +1 -1
  56. package/dist/pipelines/index.js +3 -2
  57. package/dist/pipelines/index.js.map +1 -1
  58. package/dist/{release-report-wuilQkvK.js → release-report-BVZBmRZp.js} +2 -2
  59. package/dist/{release-report-wuilQkvK.js.map → release-report-BVZBmRZp.js.map} +1 -1
  60. package/dist/{release-report-CjHWa8Ia.d.ts → release-report-DKBtegGt.d.ts} +2 -2
  61. package/dist/{release-report-CjHWa8Ia.d.ts.map → release-report-DKBtegGt.d.ts.map} +1 -1
  62. package/dist/reporting.d.ts +3 -3
  63. package/dist/reporting.js +4 -4
  64. package/dist/{researcher-CbSKhK8z.d.ts → researcher-BtD5U1Up.d.ts} +2 -2
  65. package/dist/{researcher-CbSKhK8z.d.ts.map → researcher-BtD5U1Up.d.ts.map} +1 -1
  66. package/dist/{reward-hacking-Dl2UBzej.js → reward-hacking-DCdRK9TY.js} +2 -2
  67. package/dist/{reward-hacking-Dl2UBzej.js.map → reward-hacking-DCdRK9TY.js.map} +1 -1
  68. package/dist/rl.d.ts +1 -1
  69. package/dist/rl.js +4 -4
  70. package/dist/{rubric-predictive-validity-QG7ydk0s.js → rubric-predictive-validity-D6Q6n9oq.js} +2 -2
  71. package/dist/{rubric-predictive-validity-QG7ydk0s.js.map → rubric-predictive-validity-D6Q6n9oq.js.map} +1 -1
  72. package/dist/{skillopt-optimization-method-wHF5xsUv.d.ts → skillopt-optimization-method-Dxr8pdZd.d.ts} +12 -7
  73. package/dist/{skillopt-optimization-method-wHF5xsUv.d.ts.map → skillopt-optimization-method-Dxr8pdZd.d.ts.map} +1 -1
  74. package/dist/{skillopt-optimization-method-CF6a327Q.js → skillopt-optimization-method-vvJ4bMNI.js} +119 -24
  75. package/dist/skillopt-optimization-method-vvJ4bMNI.js.map +1 -0
  76. package/dist/{statistics-DbvkkDPa.d.ts → statistics-D_4Snl-5.d.ts} +158 -30
  77. package/dist/statistics-D_4Snl-5.d.ts.map +1 -0
  78. package/dist/{statistics-DWM_AyLe.js → statistics-RwRNu2__.js} +546 -98
  79. package/dist/statistics-RwRNu2__.js.map +1 -0
  80. package/dist/{summary-report-Ci17nIdU.js → summary-report-BxtossFi.js} +3 -3
  81. package/dist/{summary-report-Ci17nIdU.js.map → summary-report-BxtossFi.js.map} +1 -1
  82. package/dist/{summary-report-CFnQgNfg.d.ts → summary-report-DyOhItws.d.ts} +3 -2
  83. package/dist/summary-report-DyOhItws.d.ts.map +1 -0
  84. package/dist/{baseline-DcX5hQDv.js → tool-use-metrics-DEGMKycK.js} +2 -114
  85. package/dist/tool-use-metrics-DEGMKycK.js.map +1 -0
  86. package/docs/design/statistics-decisions.md +271 -0
  87. package/docs/design.md +1 -0
  88. package/docs/insight-report.md +1 -1
  89. package/docs/research-report-methodology.md +4 -1
  90. package/package.json +2 -1
  91. package/dist/analyze-runs-B-afTpCv.js.map +0 -1
  92. package/dist/baseline-DcX5hQDv.js.map +0 -1
  93. package/dist/baseline-hG3K85h4.d.ts.map +0 -1
  94. package/dist/client-COvaLoQG.d.ts.map +0 -1
  95. package/dist/client-CYzbdJOZ.js.map +0 -1
  96. package/dist/index-C7Wue8R6.d.ts.map +0 -1
  97. package/dist/index-DSC51roc.d.ts.map +0 -1
  98. package/dist/index-nhIYz9hn.d.ts.map +0 -1
  99. package/dist/ledger-core-CPZfcrC2.js.map +0 -1
  100. package/dist/skillopt-optimization-method-CF6a327Q.js.map +0 -1
  101. package/dist/statistics-DWM_AyLe.js.map +0 -1
  102. package/dist/statistics-DbvkkDPa.d.ts.map +0 -1
  103. package/dist/summary-report-CFnQgNfg.d.ts.map +0 -1
@@ -3,10 +3,10 @@ import { c as costForTokenPricing, i as CostLedger, t as CostAccountingIncomplet
3
3
  import { f as maximumChargeForLlmRequest, l as costReceiptFromLlm, m as stripFencedJson, u as costReceiptFromLlmError } from "./llm-client-ClPW-dWB.js";
4
4
  import { i as combineAbortSignals, r as clamp01 } from "./run-score-iEEAWiBY.js";
5
5
  import { n as mapConcurrent } from "./concurrency-MUjT7VjM.js";
6
- import { a as confidenceInterval, j as weightedComposite, v as pairedBootstrap } from "./statistics-DWM_AyLe.js";
6
+ import { C as pairedBootstrap, D as pairedSignTest, I as weightedComposite, u as confidenceInterval } from "./statistics-RwRNu2__.js";
7
7
  import { t as DEFAULT_REDACTION_RULES } from "./redact-7Aq1ukl-.js";
8
- import { a as campaignCellExecutionEvidence, l as projectCampaignCellQuality, t as detectRewardHacking } from "./reward-hacking-Dl2UBzej.js";
9
- import { a as tryWithLedgerFileLock, i as appendLedgerLine, m as tryAcquireAtomicFileLock, p as probeAtomicFileLock } from "./ledger-core-CPZfcrC2.js";
8
+ import { a as campaignCellExecutionEvidence, l as projectCampaignCellQuality, t as detectRewardHacking } from "./reward-hacking-DCdRK9TY.js";
9
+ import { a as appendLedgerLine, h as tryAcquireAtomicFileLock, m as probeAtomicFileLock, o as tryWithLedgerFileLock } from "./ledger-core-DAKFKRzi.js";
10
10
  import { createRequire } from "node:module";
11
11
  import { z } from "zod";
12
12
  import { appendFileSync, existsSync, readFileSync, writeFileSync } from "node:fs";
@@ -155,6 +155,51 @@ function assertBackendReport(report, opts) {
155
155
  return report;
156
156
  }
157
157
  //#endregion
158
+ //#region src/paired-delta-test.ts
159
+ /** Smallest all-positive sample that can clear a one-sided exact sign test. */
160
+ function minimumPairsForPairedDeltaTest(confidence = .95) {
161
+ if (!Number.isFinite(confidence) || confidence <= 0 || confidence >= 1) throw new Error(`minimumPairsForPairedDeltaTest: confidence must be in (0,1), got ${confidence}`);
162
+ const oneSidedAlpha = (1 - confidence) / 2;
163
+ return Math.ceil(Math.log2(1 / oneSidedAlpha));
164
+ }
165
+ /**
166
+ * Tests whether a paired candidate-minus-baseline delta clears a threshold.
167
+ *
168
+ * At 20 or more pairs, the percentile bootstrap lower bound carries the
169
+ * decision. Below that point the interval is descriptive only, so the function
170
+ * switches to a pre-registered one-sided exact sign test. The exact path is
171
+ * deliberately conservative: it requires both a point estimate above the
172
+ * threshold and enough consistently positive paired differences.
173
+ */
174
+ function pairedDeltaTest(before, after, options = {}) {
175
+ const threshold = options.threshold ?? 0;
176
+ if (!Number.isFinite(threshold)) throw new Error(`pairedDeltaTest: threshold must be finite, got ${threshold}`);
177
+ const exactMinimum = minimumPairsForPairedDeltaTest(options.confidence ?? .95);
178
+ const requestedMinimum = options.minPairs ?? exactMinimum;
179
+ if (!Number.isInteger(requestedMinimum) || requestedMinimum < 1) throw new Error(`pairedDeltaTest: minPairs must be a positive integer, got ${requestedMinimum}`);
180
+ const minimumPairs = Math.max(requestedMinimum, exactMinimum);
181
+ const bootstrap = pairedBootstrap(before, after, options);
182
+ const sufficient = bootstrap.n >= minimumPairs;
183
+ if (bootstrap.gateEligible) return {
184
+ bootstrap,
185
+ method: "bootstrap-ci",
186
+ pValue: null,
187
+ minimumPairs,
188
+ sufficient,
189
+ significant: sufficient && bootstrap.low > threshold
190
+ };
191
+ const exact = pairedSignTest(before.map((value, index) => after[index] - value - threshold), "greater");
192
+ const estimate = options.statistic === "mean" ? bootstrap.mean : bootstrap.median;
193
+ return {
194
+ bootstrap,
195
+ method: "exact-sign",
196
+ pValue: exact.pValue,
197
+ minimumPairs,
198
+ sufficient,
199
+ significant: sufficient && estimate > threshold && exact.pValue <= (1 - bootstrap.confidence) / 2
200
+ };
201
+ }
202
+ //#endregion
158
203
  //#region src/json-recovery.ts
159
204
  /**
160
205
  * Truncation-tolerant JSON recovery — shared by every parser that reads JSON
@@ -5251,23 +5296,24 @@ function pairHoldout(candidate, baseline, scenarioIds, select) {
5251
5296
  }
5252
5297
  /** Significance of the held-out composite lift: ship only when the paired
5253
5298
  * bootstrap CI lower bound on (candidate − baseline) exceeds `deltaThreshold`
5254
- * (default 0 ⇒ "confidently positive"). Below `minProductiveRuns` paired
5255
- * observations there is not enough evidence to claim significance not
5256
- * significant (`fewRuns`). Interpret `deltaThreshold` in the judge's native
5257
- * composite scale. */
5299
+ * (default 0 ⇒ "confidently positive"). At small n, where the percentile
5300
+ * bootstrap is descriptive only, a pre-registered exact sign test carries
5301
+ * the decision. Interpret `deltaThreshold` in the judge's native scale. */
5258
5302
  function heldoutSignificance(paired, opts = {}) {
5259
5303
  const deltaThreshold = opts.deltaThreshold ?? 0;
5260
- const minProductiveRuns = opts.minProductiveRuns ?? 3;
5261
5304
  const confidence = opts.confidence ?? .95;
5262
5305
  const resamples = opts.resamples ?? 2e3;
5263
5306
  const seed = opts.seed ?? 1337;
5264
5307
  const statistic = opts.statistic ?? "mean";
5265
- const bootstrap = pairedBootstrap(paired.before, paired.after, {
5308
+ const decision = pairedDeltaTest(paired.before, paired.after, {
5266
5309
  confidence,
5267
5310
  resamples,
5268
5311
  statistic,
5269
- seed
5312
+ seed,
5313
+ threshold: deltaThreshold,
5314
+ minPairs: opts.minProductiveRuns
5270
5315
  });
5316
+ const bootstrap = decision.bootstrap;
5271
5317
  const medianBootstrap = statistic === "median" ? bootstrap : pairedBootstrap(paired.before, paired.after, {
5272
5318
  confidence,
5273
5319
  resamples,
@@ -5282,14 +5328,18 @@ function heldoutSignificance(paired, opts = {}) {
5282
5328
  if (Math.abs(after - before) < 1e-9) ties += 1;
5283
5329
  }
5284
5330
  const tieFraction = n === 0 ? 0 : ties / n;
5285
- const fewRuns = n < minProductiveRuns;
5331
+ const fewRuns = !decision.sufficient;
5332
+ const significant = decision.significant;
5286
5333
  return {
5287
5334
  paired,
5288
5335
  bootstrap,
5289
5336
  medianBootstrap,
5290
5337
  tieFraction,
5291
5338
  n,
5292
- significant: !fewRuns && bootstrap.low > deltaThreshold,
5339
+ minimumRequired: decision.minimumPairs,
5340
+ decisionMethod: decision.method,
5341
+ pValue: decision.pValue,
5342
+ significant,
5293
5343
  fewRuns
5294
5344
  };
5295
5345
  }
@@ -5317,10 +5367,17 @@ function dimensionRegressions(candidate, baseline, scenarioIds, criticalDimensio
5317
5367
  statistic: "median",
5318
5368
  seed: opts.seed ?? 1337
5319
5369
  });
5370
+ const regression = pairedDeltaTest(paired.after, paired.before, {
5371
+ confidence: opts.confidence ?? .95,
5372
+ resamples: opts.resamples ?? 2e3,
5373
+ statistic: "median",
5374
+ seed: opts.seed ?? 1337,
5375
+ threshold: tolerance
5376
+ });
5320
5377
  out.push({
5321
5378
  dimension: dim,
5322
5379
  bootstrap,
5323
- regressed: bootstrap.low < -tolerance,
5380
+ regressed: regression.significant,
5324
5381
  tolerance,
5325
5382
  n: paired.before.length
5326
5383
  });
@@ -5395,7 +5452,7 @@ function defaultProductionGate(options) {
5395
5452
  if (sig.fewRuns) requiredUnavailable.add("heldout-significance");
5396
5453
  if (!heldoutPass) {
5397
5454
  const tieNote = sig.tieFraction >= .4 ? `; ${(sig.tieFraction * 100).toFixed(0)}% tied scenarios` : "";
5398
- reasons.push(sig.fewRuns ? `held-out: only ${sig.n} paired runs (< ${minProductiveRuns}) — too few to claim significance` : `held-out CI.low ${sig.bootstrap.low.toFixed(3)} ≤ threshold ${deltaThreshold} (${heldoutStatistic} Δ ${delta.toFixed(3)}, ${(sig.bootstrap.confidence * 100).toFixed(0)}% CI [${sig.bootstrap.low.toFixed(3)}, ${sig.bootstrap.high.toFixed(3)}]${tieNote})`);
5455
+ reasons.push(sig.fewRuns ? `held-out: only ${sig.n} paired runs (< ${sig.minimumRequired}) — too few to claim significance` : `held-out CI.low ${sig.bootstrap.low.toFixed(3)} ≤ threshold ${deltaThreshold} (${heldoutStatistic} Δ ${delta.toFixed(3)}, ${(sig.bootstrap.confidence * 100).toFixed(0)}% CI [${sig.bootstrap.low.toFixed(3)}, ${sig.bootstrap.high.toFixed(3)}]${tieNote})`);
5399
5456
  }
5400
5457
  }
5401
5458
  const dimensionsProvided = options.criticalDimensions !== void 0;
@@ -5617,7 +5674,7 @@ function heldOutGate(options) {
5617
5674
  const ci = `${(sig.bootstrap.confidence * 100).toFixed(0)}% CI [${sig.bootstrap.low.toFixed(3)}, ${sig.bootstrap.high.toFixed(3)}]`;
5618
5675
  return {
5619
5676
  decision: passed ? "ship" : "hold",
5620
- reasons: passed ? [`held-out mean Δ ${delta.toFixed(3)}, CI.low ${sig.bootstrap.low.toFixed(3)} > ${deltaThreshold} (${ci}, n=${sig.n}${tieNote})`] : [sig.fewRuns ? `held-out: only ${sig.n} paired runs — too few to claim significance` : `held-out mean Δ ${delta.toFixed(3)}, CI.low ${sig.bootstrap.low.toFixed(3)} ≤ ${deltaThreshold} (${ci}, n=${sig.n}${tieNote})`],
5677
+ reasons: passed ? [`held-out mean Δ ${delta.toFixed(3)}, CI.low ${sig.bootstrap.low.toFixed(3)} > ${deltaThreshold} (${ci}, n=${sig.n}${tieNote})`] : [sig.fewRuns ? `held-out: only ${sig.n} paired runs; ${sig.minimumRequired} required — too few to claim significance` : `held-out mean Δ ${delta.toFixed(3)}, CI.low ${sig.bootstrap.low.toFixed(3)} ≤ ${deltaThreshold} (${ci}, n=${sig.n}${tieNote})`],
5621
5678
  contributingGates: [{
5622
5679
  name: "heldOutGate",
5623
5680
  status,
@@ -5686,6 +5743,30 @@ function powerPreflight(opts) {
5686
5743
  //#endregion
5687
5744
  //#region src/campaign/gates/promotion-policy.ts
5688
5745
  /**
5746
+ * Promotion policy over the evidence VECTOR — the substrate's answer to "never
5747
+ * collapse the multi-objective promotion decision into one scalar." A
5748
+ * `defaultProductionGate` is one opinionated composition; this module factors
5749
+ * the decision into two reusable pieces so MANY policies can compete over the
5750
+ * SAME evidence (the quant-desk pattern: one evidence bus, plural strategies):
5751
+ *
5752
+ * buildEvidenceVector(ctx, objectives, opts) -> EvidenceVector // the bus
5753
+ * PromotionPolicy = (ev: EvidenceVector) => GateResult // a strategy
5754
+ * paretoPolicy(ev) // the default strategy
5755
+ * paretoSignificanceGate(options): Gate // bus + policy as a Gate
5756
+ *
5757
+ * The Pareto policy is SYMMETRIC multi-objective: every objective is BOTH a
5758
+ * potential gain source AND a safety floor (unlike `defaultProductionGate`,
5759
+ * where only `composite` can win and `criticalDimensions` are pure floors). A
5760
+ * candidate ships iff it weakly DOMINATES the baseline at the confidence level —
5761
+ * no objective credibly worse (CI floor breach) AND at least one objective
5762
+ * credibly better (CI gain). Insufficient evidence on ANY axis -> need_more_work
5763
+ * (NOT folded into hold: "gather more reps" and "reject" are different actions).
5764
+ *
5765
+ * Cost/latency are NOT CI axes here — `GateContext` carries only an aggregate
5766
+ * per-side cost, no per-cell observation vector to bootstrap. Treat them as hard
5767
+ * constraints (compose with a budget gate via `composeGate`), not faked CIs.
5768
+ */
5769
+ /**
5689
5770
  * The Evidence Bus. For each objective, pair candidate vs baseline by full
5690
5771
  * cellId and bootstrap a CI on the good-direction paired delta. Reuses the
5691
5772
  * exact `pairHoldout` + `pairedBootstrap` machinery the held-out gate uses, so
@@ -5693,7 +5774,6 @@ function powerPreflight(opts) {
5693
5774
  */
5694
5775
  function buildEvidenceVector(ctx, objectives, opts = {}) {
5695
5776
  if (objectives.length === 0) throw new Error("buildEvidenceVector: at least 1 objective required");
5696
- const minProductiveRuns = opts.minProductiveRuns ?? 3;
5697
5777
  const confidence = opts.confidence ?? .95;
5698
5778
  const resamples = opts.resamples ?? 2e3;
5699
5779
  const seed = opts.seed ?? 1337;
@@ -5708,22 +5788,37 @@ function buildEvidenceVector(ctx, objectives, opts = {}) {
5708
5788
  select = (s) => s.dimensions[dim];
5709
5789
  }
5710
5790
  const paired = pairHoldout(ctx.judgeScores, baseline, scenarioIds, select);
5711
- const bootstrap = pairedBootstrap(obj.direction === "maximize" ? paired.before : paired.after, obj.direction === "maximize" ? paired.after : paired.before, {
5791
+ const before = obj.direction === "maximize" ? paired.before : paired.after;
5792
+ const after = obj.direction === "maximize" ? paired.after : paired.before;
5793
+ const n = paired.before.length;
5794
+ const floorTolerance = obj.floorTolerance ?? .05 * detectScale([...paired.before, ...paired.after]);
5795
+ const gainThreshold = obj.gainThreshold ?? 0;
5796
+ const improvement = pairedDeltaTest(before, after, {
5712
5797
  confidence,
5713
5798
  resamples,
5714
5799
  statistic: "median",
5715
- seed
5800
+ seed,
5801
+ threshold: gainThreshold,
5802
+ minPairs: opts.minProductiveRuns
5716
5803
  });
5717
- const n = paired.before.length;
5718
- const floorTolerance = obj.floorTolerance ?? .05 * detectScale([...paired.before, ...paired.after]);
5719
- const gainThreshold = obj.gainThreshold ?? 0;
5720
- const verdict = n < minProductiveRuns ? "few_runs" : bootstrap.low < -floorTolerance ? "regressed" : bootstrap.low > gainThreshold ? "improved" : "flat";
5804
+ const regression = pairedDeltaTest(after, before, {
5805
+ confidence,
5806
+ resamples,
5807
+ statistic: "median",
5808
+ seed,
5809
+ threshold: floorTolerance,
5810
+ minPairs: opts.minProductiveRuns
5811
+ });
5812
+ const bootstrap = improvement.bootstrap;
5813
+ const verdict = !improvement.sufficient ? "few_runs" : regression.significant ? "regressed" : improvement.significant ? "improved" : "flat";
5721
5814
  axes.push({
5722
5815
  name: obj.name,
5723
5816
  source: obj.source,
5724
5817
  direction: obj.direction,
5725
5818
  bootstrap,
5726
5819
  n,
5820
+ minimumRequired: improvement.minimumPairs,
5821
+ decisionMethod: improvement.method,
5727
5822
  gainThreshold,
5728
5823
  floorTolerance,
5729
5824
  verdict
@@ -7717,6 +7812,6 @@ function skillOptOptimizationMethod(config) {
7717
7812
  };
7718
7813
  }
7719
7814
  //#endregion
7720
- export { SEARCH_LEDGER_FILE_CONTEXT as $, acquireSingleRunLock as A, dominates as At, surfaceContentHash as B, JudgeParseError as Bt, detectScale as C, llmJudge as Ct, runCanaries as D, fileVerdictCache as Dt, pairHoldout as E, contentHash as Et, assertCodeSurfaceIdentity as F, BackendIntegrityError as Ft, DEFAULT_MUTATION_PRIMITIVES as G, campaignBreakdown as H, assertComponentSurface as I, assertRealAgentReceipts as It, planCampaignRun as J, buildReflectionPrompt as K, codeSurfaceIdentityMaterial as L, assertRealBackend as Lt, compareOptimizationMethods as M, paretoFrontierWithCrowding as Mt, costFromLedgerSummary as N, scalarScore as Nt, composeGate as O, inMemoryVerdictCache as Ot, optimizationTokenUsageFromSummary as P, recoverTruncatedJson as Pt, inMemoryCampaignStorage as Q, componentSurfaceIdentityMaterial as R, summarizeAgentReceiptIntegrity as Rt, defaultProductionGate as S, hashScenarios as St, heldoutSignificance as T, canonicalJson as Tt, campaignMeanComposite as U, surfaceHash as V, compareRankKeys as W, createRunCostLedger as X, runCampaign as Y, fsCampaignStorage as Z, buildEvidenceVector as _, redTeamReport as _t, emitLoopProvenance as a, assertCampaignDesign as at, powerPreflight as b, Dataset as bt, provenanceRecordPath as c, campaignSplitDigest as ct, runImprovementLoop as d, REFERENCE_EQUIVALENCE_INPUT_LIMITS as dt, SearchLedgerConflictError as et, runOptimization as f, REFERENCE_EQUIVALENCE_JUDGE_VERSION as ft, gepaOptimizationMethod as g, redTeamDataset as gt, labelTrustRank as h, DEFAULT_RED_TEAM_CORPUS as ht, canonicalDigest as i, tangleTracesRoot as it, assertOptimizationResult as j, paretoFrontier as jt, externalTextOptimizationMethod as k, crowdingDistance as kt, provenanceSpansPath as l, campaignSplitDigestFromIdentities as lt, isProposedCandidate as m, runReferenceEquivalenceJudge as mt, buildLoopProvenanceRecord as n, SearchLedgerIntegrityError as nt, loopProvenanceArgsFromResult as o, assertCampaignSplitIdentity as ot, runEval as p, createReferenceEquivalenceJudge as pt, parseReflectionResponse as q, campaignMeasurementDigest as r, resolveRunDir as rt, loopProvenanceSpans as s, campaignScenarioIdentity as st, skillOptOptimizationMethod as t, SearchLedgerError as tt, verifyLoopProvenanceRecord as u, openAutoPr as ut, paretoPolicy as v, scoreRedTeamOutput as vt, dimensionRegressions as w, cachedJudge as wt, heldOutGate as x, HoldoutLockedError as xt, paretoSignificanceGate as y, toolNamesForRun as yt, renderSurfaceDiff as z, summarizeBackendIntegrity as zt };
7815
+ export { SEARCH_LEDGER_FILE_CONTEXT as $, acquireSingleRunLock as A, dominates as At, surfaceContentHash as B, summarizeAgentReceiptIntegrity as Bt, detectScale as C, llmJudge as Ct, runCanaries as D, fileVerdictCache as Dt, pairHoldout as E, contentHash as Et, assertCodeSurfaceIdentity as F, minimumPairsForPairedDeltaTest as Ft, DEFAULT_MUTATION_PRIMITIVES as G, campaignBreakdown as H, JudgeParseError as Ht, assertComponentSurface as I, pairedDeltaTest as It, planCampaignRun as J, buildReflectionPrompt as K, codeSurfaceIdentityMaterial as L, BackendIntegrityError as Lt, compareOptimizationMethods as M, paretoFrontierWithCrowding as Mt, costFromLedgerSummary as N, scalarScore as Nt, composeGate as O, inMemoryVerdictCache as Ot, optimizationTokenUsageFromSummary as P, recoverTruncatedJson as Pt, inMemoryCampaignStorage as Q, componentSurfaceIdentityMaterial as R, assertRealAgentReceipts as Rt, defaultProductionGate as S, hashScenarios as St, heldoutSignificance as T, canonicalJson as Tt, campaignMeanComposite as U, surfaceHash as V, summarizeBackendIntegrity as Vt, compareRankKeys as W, createRunCostLedger as X, runCampaign as Y, fsCampaignStorage as Z, buildEvidenceVector as _, redTeamReport as _t, emitLoopProvenance as a, assertCampaignDesign as at, powerPreflight as b, Dataset as bt, provenanceRecordPath as c, campaignSplitDigest as ct, runImprovementLoop as d, REFERENCE_EQUIVALENCE_INPUT_LIMITS as dt, SearchLedgerConflictError as et, runOptimization as f, REFERENCE_EQUIVALENCE_JUDGE_VERSION as ft, gepaOptimizationMethod as g, redTeamDataset as gt, labelTrustRank as h, DEFAULT_RED_TEAM_CORPUS as ht, canonicalDigest as i, tangleTracesRoot as it, assertOptimizationResult as j, paretoFrontier as jt, externalTextOptimizationMethod as k, crowdingDistance as kt, provenanceSpansPath as l, campaignSplitDigestFromIdentities as lt, isProposedCandidate as m, runReferenceEquivalenceJudge as mt, buildLoopProvenanceRecord as n, SearchLedgerIntegrityError as nt, loopProvenanceArgsFromResult as o, assertCampaignSplitIdentity as ot, runEval as p, createReferenceEquivalenceJudge as pt, parseReflectionResponse as q, campaignMeasurementDigest as r, resolveRunDir as rt, loopProvenanceSpans as s, campaignScenarioIdentity as st, skillOptOptimizationMethod as t, SearchLedgerError as tt, verifyLoopProvenanceRecord as u, openAutoPr as ut, paretoPolicy as v, scoreRedTeamOutput as vt, dimensionRegressions as w, cachedJudge as wt, heldOutGate as x, HoldoutLockedError as xt, paretoSignificanceGate as y, toolNamesForRun as yt, renderSurfaceDiff as z, assertRealBackend as zt };
7721
7816
 
7722
- //# sourceMappingURL=skillopt-optimization-method-CF6a327Q.js.map
7817
+ //# sourceMappingURL=skillopt-optimization-method-vvJ4bMNI.js.map