@tangle-network/agent-eval 0.135.1 → 0.135.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (85) hide show
  1. package/CHANGELOG.md +17 -0
  2. package/dist/{analyze-runs-DMo3Lb_y.d.ts → analyze-runs-Cda5Xkj1.d.ts} +3 -3
  3. package/dist/{analyze-runs-DMo3Lb_y.d.ts.map → analyze-runs-Cda5Xkj1.d.ts.map} +1 -1
  4. package/dist/{analyze-runs-qk8op0tN.js → analyze-runs-jjCmF8pU.js} +14 -7
  5. package/dist/analyze-runs-jjCmF8pU.js.map +1 -0
  6. package/dist/{baseline-BaPxoROc.js → baseline-BUeFcgrn.js} +2 -2
  7. package/dist/{baseline-BaPxoROc.js.map → baseline-BUeFcgrn.js.map} +1 -1
  8. package/dist/benchmarks/index.d.ts +1 -1
  9. package/dist/benchmarks/index.js +1 -1
  10. package/dist/{benchmarks-DwFrCrCS.js → benchmarks-Mtu251Jz.js} +3 -3
  11. package/dist/{benchmarks-DwFrCrCS.js.map → benchmarks-Mtu251Jz.js.map} +1 -1
  12. package/dist/builder-eval/index.js +1 -1
  13. package/dist/campaign/index.d.ts +2 -2
  14. package/dist/campaign/index.js +2 -2
  15. package/dist/{campaign-B0oyngTs.js → campaign-RVIqtJh0.js} +5 -5
  16. package/dist/{campaign-B0oyngTs.js.map → campaign-RVIqtJh0.js.map} +1 -1
  17. package/dist/{client-BIyh1RCr.d.ts → client-DcvgkaZi.d.ts} +13 -4
  18. package/dist/client-DcvgkaZi.d.ts.map +1 -0
  19. package/dist/contract/index.d.ts +3 -3
  20. package/dist/contract/index.d.ts.map +1 -1
  21. package/dist/contract/index.js +17 -5
  22. package/dist/contract/index.js.map +1 -1
  23. package/dist/{eval-campaign-aMG73hvX.js → eval-campaign-Cc8WZJ6b.js} +2 -2
  24. package/dist/{eval-campaign-aMG73hvX.js.map → eval-campaign-Cc8WZJ6b.js.map} +1 -1
  25. package/dist/hosted/index.d.ts +1 -1
  26. package/dist/{index-BoJNQR6n.d.ts → index-B4Fjfo5U.d.ts} +93 -15
  27. package/dist/index-B4Fjfo5U.d.ts.map +1 -0
  28. package/dist/{index-C21xKtxu.d.ts → index-CQsJcqch.d.ts} +3 -3
  29. package/dist/{index-C21xKtxu.d.ts.map → index-CQsJcqch.d.ts.map} +1 -1
  30. package/dist/{index-DSC51roc2.d.ts → index-DSC51roc.d.ts} +1 -1
  31. package/dist/index-DSC51roc.d.ts.map +1 -0
  32. package/dist/index.d.ts +41 -10
  33. package/dist/index.d.ts.map +1 -1
  34. package/dist/index.js +131 -53
  35. package/dist/index.js.map +1 -1
  36. package/dist/matrix/index.d.ts +1 -1
  37. package/dist/meta-eval/index.d.ts +1 -2
  38. package/dist/meta-eval/index.d.ts.map +1 -1
  39. package/dist/meta-eval/index.js +2 -2
  40. package/dist/multishot/index.d.ts +1 -1
  41. package/dist/openapi.json +1 -1
  42. package/dist/{paired-arms-CA_8pN01.js → paired-arms-BbFKrAU-.js} +2 -2
  43. package/dist/{paired-arms-CA_8pN01.js.map → paired-arms-BbFKrAU-.js.map} +1 -1
  44. package/dist/pipelines/index.js +2 -2
  45. package/dist/{release-report-BVZBmRZp.js → release-report-DooPguBc.js} +4 -3
  46. package/dist/{release-report-BVZBmRZp.js.map → release-report-DooPguBc.js.map} +1 -1
  47. package/dist/{release-report-CuULWKyk.d.ts → release-report-DpBxGGI1.d.ts} +2 -2
  48. package/dist/{release-report-CuULWKyk.d.ts.map → release-report-DpBxGGI1.d.ts.map} +1 -1
  49. package/dist/reporting.d.ts +3 -3
  50. package/dist/reporting.js +4 -4
  51. package/dist/{researcher-DVtruQ9U.d.ts → researcher-Doo95b50.d.ts} +2 -2
  52. package/dist/{researcher-DVtruQ9U.d.ts.map → researcher-Doo95b50.d.ts.map} +1 -1
  53. package/dist/{reward-hacking-DCdRK9TY.js → reward-hacking-a-kYs0-i.js} +2 -2
  54. package/dist/{reward-hacking-DCdRK9TY.js.map → reward-hacking-a-kYs0-i.js.map} +1 -1
  55. package/dist/rl.d.ts +45 -3
  56. package/dist/rl.d.ts.map +1 -1
  57. package/dist/rl.js +108 -22
  58. package/dist/rl.js.map +1 -1
  59. package/dist/{rubric-predictive-validity-D6Q6n9oq.js → rubric-predictive-validity-BJf-8ejY.js} +2 -2
  60. package/dist/{rubric-predictive-validity-D6Q6n9oq.js.map → rubric-predictive-validity-BJf-8ejY.js.map} +1 -1
  61. package/dist/{skillopt-optimization-method-C9yGg-4C.js → skillopt-optimization-method-0UmPD6aP.js} +340 -54
  62. package/dist/skillopt-optimization-method-0UmPD6aP.js.map +1 -0
  63. package/dist/{skillopt-optimization-method-DJ3l4w8W.d.ts → skillopt-optimization-method-CwSYkv35.d.ts} +39 -9
  64. package/dist/skillopt-optimization-method-CwSYkv35.d.ts.map +1 -0
  65. package/dist/{statistics-D_4Snl-5.d.ts → statistics-CKOqre5S.d.ts} +329 -3
  66. package/dist/statistics-CKOqre5S.d.ts.map +1 -0
  67. package/dist/{statistics-RwRNu2__.js → statistics-CnGCLLqc.js} +315 -2
  68. package/dist/statistics-CnGCLLqc.js.map +1 -0
  69. package/dist/{summary-report-BxtossFi.js → summary-report-BEk8OFLs.js} +11 -6
  70. package/dist/summary-report-BEk8OFLs.js.map +1 -0
  71. package/dist/{summary-report-DGp0-_XO.d.ts → summary-report-CPMINBqs.d.ts} +182 -7
  72. package/dist/summary-report-CPMINBqs.d.ts.map +1 -0
  73. package/package.json +1 -1
  74. package/dist/analyze-runs-qk8op0tN.js.map +0 -1
  75. package/dist/client-BIyh1RCr.d.ts.map +0 -1
  76. package/dist/index-BoJNQR6n.d.ts.map +0 -1
  77. package/dist/index-DSC51roc2.d.ts.map +0 -1
  78. package/dist/judge-calibration-DFtEMlde.d.ts +0 -146
  79. package/dist/judge-calibration-DFtEMlde.d.ts.map +0 -1
  80. package/dist/skillopt-optimization-method-C9yGg-4C.js.map +0 -1
  81. package/dist/skillopt-optimization-method-DJ3l4w8W.d.ts.map +0 -1
  82. package/dist/statistics-D_4Snl-5.d.ts.map +0 -1
  83. package/dist/statistics-RwRNu2__.js.map +0 -1
  84. package/dist/summary-report-BxtossFi.js.map +0 -1
  85. package/dist/summary-report-DGp0-_XO.d.ts.map +0 -1
@@ -3,9 +3,9 @@ import { c as costForTokenPricing, i as CostLedger, t as CostAccountingIncomplet
3
3
  import { f as maximumChargeForLlmRequest, l as costReceiptFromLlm, m as stripFencedJson, u as costReceiptFromLlmError } from "./llm-client-DHx8pzyJ.js";
4
4
  import { a as clamp01, o as combineAbortSignals, t as assertProposalFindings } from "./proposal-findings-DCawte-y.js";
5
5
  import { n as mapConcurrent } from "./concurrency-MUjT7VjM.js";
6
- import { C as pairedBootstrap, D as pairedSignTest, I as weightedComposite, u as confidenceInterval } from "./statistics-RwRNu2__.js";
6
+ import { E as pairedBootstrap, H as weightedComposite, M as pairedRiskDifferenceScore, N as pairedSignTest, O as pairedDeltaTieFraction, T as pairedBinaryScale, d as confidenceInterval, j as pairedRiskDifferenceExact } from "./statistics-CnGCLLqc.js";
7
7
  import { t as DEFAULT_REDACTION_RULES } from "./redact-7Aq1ukl-.js";
8
- import { a as campaignCellExecutionEvidence, l as projectCampaignCellQuality, t as detectRewardHacking } from "./reward-hacking-DCdRK9TY.js";
8
+ import { a as campaignCellExecutionEvidence, l as projectCampaignCellQuality, t as detectRewardHacking } from "./reward-hacking-a-kYs0-i.js";
9
9
  import { a as appendLedgerLine, h as tryAcquireAtomicFileLock, m as probeAtomicFileLock, o as tryWithLedgerFileLock } from "./ledger-core-DAKFKRzi.js";
10
10
  import { createRequire } from "node:module";
11
11
  import { z } from "zod";
@@ -170,6 +170,32 @@ function minimumPairsForPairedDeltaTest(confidence = .95) {
170
170
  * switches to a pre-registered one-sided exact sign test. The exact path is
171
171
  * deliberately conservative: it requires both a point estimate above the
172
172
  * threshold and enough consistently positive paired differences.
173
+ *
174
+ * ## A zero-width interval is never significant
175
+ *
176
+ * When every paired delta is identical the resample distribution is a point
177
+ * mass and the interval collapses: `[0, 0]` when all pairs tie, `[g, g]` on n
178
+ * identical deltas of g. Neither says the effect is certain — both say the
179
+ * sample carries no information about how far the estimate could be wrong, and
180
+ * `low > threshold` then answers on the point estimate alone. It fails in both
181
+ * directions: `[0, 0]` clears every NEGATIVE threshold, which is how a
182
+ * tie-dominated pass/fail comparison laundered a regression into a
183
+ * noninferiority pass, and `[g, g]` clears every threshold below g with no
184
+ * spread behind it. Under a bounded asymmetric null whose true mean paired
185
+ * delta is exactly 0 — 2 % of pairs dropping by 1.0, the rest gaining 0.0204 —
186
+ * every sample that misses the drop is exactly that shape, and deciding on
187
+ * `low > 0` promoted 65.65 % of samples at n = 20 against a nominal 5 %.
188
+ *
189
+ * So `indeterminate` is reported and `significant` is false whenever the
190
+ * interval has zero width, on BOTH paths: at small n the exact sign test is a
191
+ * test of the MEDIAN and a zero-spread sample is precisely where it stops
192
+ * saying anything about the mean the caller is thresholding.
193
+ *
194
+ * `threshold` may be negative — that is a noninferiority margin, and it is the
195
+ * regime the zero-width hole is worst in. For a two-point (pass/fail) outcome
196
+ * the percentile bootstrap is not a valid interval at a nonzero margin at all;
197
+ * use {@link decidePairedPromotion}, which routes those to Tango's score
198
+ * interval, rather than thresholding this function's bootstrap directly.
173
199
  */
174
200
  function pairedDeltaTest(before, after, options = {}) {
175
201
  const threshold = options.threshold ?? 0;
@@ -180,13 +206,15 @@ function pairedDeltaTest(before, after, options = {}) {
180
206
  const minimumPairs = Math.max(requestedMinimum, exactMinimum);
181
207
  const bootstrap = pairedBootstrap(before, after, options);
182
208
  const sufficient = bootstrap.n >= minimumPairs;
209
+ const indeterminate = bootstrap.n > 0 && (!Number.isFinite(bootstrap.low) || !Number.isFinite(bootstrap.high) || bootstrap.low === bootstrap.high);
183
210
  if (bootstrap.gateEligible) return {
184
211
  bootstrap,
185
212
  method: "bootstrap-ci",
186
213
  pValue: null,
187
214
  minimumPairs,
188
215
  sufficient,
189
- significant: sufficient && bootstrap.low > threshold
216
+ indeterminate,
217
+ significant: sufficient && !indeterminate && bootstrap.low > threshold
190
218
  };
191
219
  const exact = pairedSignTest(before.map((value, index) => after[index] - value - threshold), "greater");
192
220
  const estimate = options.statistic === "mean" ? bootstrap.mean : bootstrap.median;
@@ -196,10 +224,174 @@ function pairedDeltaTest(before, after, options = {}) {
196
224
  pValue: exact.pValue,
197
225
  minimumPairs,
198
226
  sufficient,
199
- significant: sufficient && estimate > threshold && exact.pValue <= (1 - bootstrap.confidence) / 2
227
+ indeterminate,
228
+ significant: sufficient && !indeterminate && estimate > threshold && exact.pValue <= (1 - bootstrap.confidence) / 2
200
229
  };
201
230
  }
202
231
  //#endregion
232
+ //#region src/paired-promotion-decision.ts
233
+ /**
234
+ * @module
235
+ * ONE rule for "does this paired interval clear a promotion threshold".
236
+ *
237
+ * The rule below was derived on `HeldOutGate` (#479) after the same estimator
238
+ * bug shipped twice. It then turned out that a SECOND gate — the composable
239
+ * `heldOutGate`, plus everything else routed through `heldoutSignificance` —
240
+ * still carried the original defect, because the rule had been written into one
241
+ * gate's method body rather than into a shared function. Two copies of a
242
+ * statistical rule is how a defect survives in one of them, so there is now
243
+ * exactly one copy and both gates call it.
244
+ *
245
+ * Three things the rule does that a bare `pairedBootstrap(...).low > threshold`
246
+ * does not:
247
+ *
248
+ * 1. **Two-point (pass/fail) outcomes decide on Tango's SCORE interval.** On a
249
+ * pass/fail eval the paired delta vector is dominated by ties, so the
250
+ * bootstrap of the mean is a resample of a lattice with three atoms and its
251
+ * percentile interval is not valid at a nonzero margin. The score interval
252
+ * (`pairedRiskDifferenceScore`) re-estimates the nuisance loss rate under
253
+ * each hypothesised margin instead of fixing it at the observed value, which
254
+ * is the only construction that stays a confidence interval as the margin
255
+ * moves off zero — the regime every noninferiority threshold lives in.
256
+ * Measured on the composable gate before this change, at a true risk
257
+ * difference sitting exactly on the production caller's -0.05 margin and a
258
+ * nominal 5 %: 14.60 % false promotion at n = 40 and 10.10 % at n = 76.
259
+ * 2. **McNemar's exact test holds a VETO at every non-negative threshold.**
260
+ * Redundant with the interval by construction and kept anyway, so that
261
+ * swapping the estimator for one without that duality cannot silently
262
+ * reintroduce "promotes what the exact test refuses". Witness: n = 6, b = 5,
263
+ * c = 0 — no exact argument reaches alpha = 0.05 with 5 discordant pairs
264
+ * (two-sided floor 2/2^5 = 0.0625), whatever an interval says. A NEGATIVE
265
+ * threshold is a noninferiority question, which McNemar's test of "no
266
+ * difference" is not the right test for, so the veto does not apply there.
267
+ * 3. **A ZERO-WIDTH interval is refused, wherever it sits.** At [0, 0] it
268
+ * cannot tell a gain from a regression and clears every negative threshold.
269
+ * Away from zero it fails the opposite way: n identical positive deltas give
270
+ * [g, g], which clears threshold 0 on no spread at all. Both are an absence
271
+ * of evidence. Measured on the composable gate before this change, under a
272
+ * bounded asymmetric null whose true mean paired delta is exactly 0: 88.50 %
273
+ * false promotion at n = 6 and 65.65 % at n = 20 against a nominal 5 %.
274
+ *
275
+ * Orthogonal to the small-sample switch inside {@link pairedDeltaTest}: that
276
+ * picks the TEST from the sample size (bootstrap CI at n >= 20, pre-registered
277
+ * exact sign test below it), this picks the ESTIMATOR from the outcome's shape.
278
+ * Both are needed — an exact sign test applied to a tie-pinned median is still
279
+ * blind, and a mean bootstrap CI at n = 6 is still not a valid test.
280
+ */
281
+ /**
282
+ * Which estimator {@link decidePairedPromotion} would use on this data, and the
283
+ * shape facts behind it — for callers that must report the shape on a path
284
+ * where no interval is computed at all (an early rejection, or zero pairs).
285
+ * Cheap: no bootstrap, no interval.
286
+ */
287
+ function pairedDecisionShape(before, after, statistic = "mean") {
288
+ const tieFraction = before.length === 0 ? null : pairedDeltaTieFraction(before, after);
289
+ if (statistic === "median") return {
290
+ statistic: "median_bootstrap",
291
+ binaryScale: null,
292
+ tieFraction
293
+ };
294
+ const binaryScale = pairedBinaryScale(before, after);
295
+ if (binaryScale !== null) return {
296
+ statistic: "paired_risk_difference",
297
+ binaryScale,
298
+ tieFraction
299
+ };
300
+ return {
301
+ statistic: "mean_bootstrap",
302
+ binaryScale: null,
303
+ tieFraction
304
+ };
305
+ }
306
+ /**
307
+ * Decide whether a paired candidate-minus-baseline delta clears a promotion
308
+ * threshold. `before` is the baseline arm, `after` the candidate arm, paired by
309
+ * position. Throws on unequal lengths.
310
+ */
311
+ function decidePairedPromotion(before, after, options = {}) {
312
+ if (before.length !== after.length) throw new Error(`decidePairedPromotion: unequal sample sizes (${before.length} vs ${after.length})`);
313
+ const threshold = options.threshold ?? 0;
314
+ if (!Number.isFinite(threshold)) throw new Error(`decidePairedPromotion: threshold must be finite, got ${threshold}`);
315
+ const confidence = options.confidence ?? .95;
316
+ const exactMinimum = minimumPairsForPairedDeltaTest(confidence);
317
+ const requestedMinimum = options.minPairs ?? exactMinimum;
318
+ if (!Number.isInteger(requestedMinimum) || requestedMinimum < 1) throw new Error(`decidePairedPromotion: minPairs must be a positive integer, got ${requestedMinimum}`);
319
+ const minimumPairs = Math.max(requestedMinimum, exactMinimum);
320
+ const n = before.length;
321
+ const sufficient = n >= minimumPairs;
322
+ const { binaryScale, tieFraction } = pairedDecisionShape(before, after, options.statistic);
323
+ let core;
324
+ if (binaryScale !== null) {
325
+ const unitControl = before.map((v) => v / binaryScale);
326
+ const unitTreatment = after.map((v) => v / binaryScale);
327
+ const exact = pairedRiskDifferenceExact(unitControl, unitTreatment, confidence);
328
+ const score = pairedRiskDifferenceScore(unitControl, unitTreatment, confidence);
329
+ const low = score.lower * binaryScale;
330
+ core = {
331
+ statistic: "paired_risk_difference",
332
+ method: "score-interval",
333
+ delta: score.riskDifference * binaryScale,
334
+ low,
335
+ high: score.upper * binaryScale,
336
+ bootstrap: null,
337
+ mcnemar: {
338
+ b: exact.b,
339
+ c: exact.c,
340
+ nDiscordant: exact.nDiscordant,
341
+ pValue: exact.pValue
342
+ },
343
+ pValue: null,
344
+ clearsThreshold: low > threshold,
345
+ label: "success-rate",
346
+ methodDetail: ""
347
+ };
348
+ } else {
349
+ const bootstrapStatistic = options.statistic === "median" ? "median" : "mean";
350
+ const test = pairedDeltaTest(before, after, {
351
+ confidence,
352
+ resamples: options.resamples,
353
+ statistic: bootstrapStatistic,
354
+ seed: options.seed,
355
+ threshold,
356
+ minPairs: options.minPairs
357
+ });
358
+ const ci = test.bootstrap;
359
+ core = {
360
+ statistic: bootstrapStatistic === "mean" ? "mean_bootstrap" : "median_bootstrap",
361
+ method: test.method,
362
+ delta: bootstrapStatistic === "mean" ? ci.mean : ci.median,
363
+ low: ci.low,
364
+ high: ci.high,
365
+ bootstrap: ci,
366
+ mcnemar: null,
367
+ pValue: test.pValue,
368
+ clearsThreshold: test.significant,
369
+ label: bootstrapStatistic,
370
+ methodDetail: test.method === "exact-sign" ? ` Below ${test.minimumPairs} pairs the interval is descriptive only; the decision is the exact one-sided sign test, p=${fmt(test.pValue ?? 1)}.` : ""
371
+ };
372
+ }
373
+ const indeterminate = !Number.isFinite(core.low) || !Number.isFinite(core.high) || core.low === core.high;
374
+ const indeterminateCause = !indeterminate ? "" : tieFraction === 1 ? "every paired delta is an exact tie" : core.mcnemar !== null && core.mcnemar.nDiscordant === 0 ? "every pair is concordant (0 discordant pairs)" : `the ${core.label} CI collapsed to a point at ${fmt(core.low)}`;
375
+ const exactTestVetoes = core.mcnemar !== null && threshold >= 0 && !(core.mcnemar.pValue < 1 - confidence);
376
+ return {
377
+ n,
378
+ threshold,
379
+ confidence,
380
+ binaryScale,
381
+ tieFraction,
382
+ minimumPairs,
383
+ sufficient,
384
+ indeterminate,
385
+ indeterminateCause,
386
+ exactTestVetoes,
387
+ promote: sufficient && !indeterminate && core.clearsThreshold && !exactTestVetoes,
388
+ ...core
389
+ };
390
+ }
391
+ function fmt(x) {
392
+ return x.toFixed(4);
393
+ }
394
+ //#endregion
203
395
  //#region src/json-recovery.ts
204
396
  /**
205
397
  * Truncation-tolerant JSON recovery — shared by every parser that reads JSON
@@ -4178,7 +4370,7 @@ async function compareOptimizationMethods(opts) {
4178
4370
  confidence: intervalConfidence,
4179
4371
  statistic: "mean"
4180
4372
  });
4181
- const favored = boot.low > 0 ? best.name : boot.high < 0 ? other.name : "tie";
4373
+ const favored = !Number.isFinite(boot.low) || !Number.isFinite(boot.high) || boot.low === boot.high ? "tie" : boot.low > 0 ? best.name : boot.high < 0 ? other.name : "tie";
4182
4374
  return {
4183
4375
  a: best.name,
4184
4376
  b: other.name,
@@ -5294,18 +5486,39 @@ function pairHoldout(candidate, baseline, scenarioIds, select) {
5294
5486
  cellIds
5295
5487
  };
5296
5488
  }
5297
- /** Significance of the held-out composite lift: ship only when the paired
5298
- * bootstrap CI lower bound on (candidate baseline) exceeds `deltaThreshold`
5299
- * (default 0 "confidently positive"). At small n, where the percentile
5300
- * bootstrap is descriptive only, a pre-registered exact sign test carries
5301
- * the decision. Interpret `deltaThreshold` in the judge's native scale. */
5489
+ /**
5490
+ * Significance of the held-out composite lift: ship only when the lower bound
5491
+ * of the interval the outcome's shape ADMITS exceeds `deltaThreshold` (default
5492
+ * 0 "confidently positive"). Interpret `deltaThreshold` in the judge's native
5493
+ * scale.
5494
+ *
5495
+ * The decision is delegated whole to {@link decidePairedPromotion}, the one
5496
+ * copy of the rule (`src/paired-promotion-decision.ts`), which `HeldOutGate`
5497
+ * also calls. That module's header carries the measurements; the short version
5498
+ * is three guards a bare `bootstrap.low > threshold` does not have:
5499
+ *
5500
+ * - a two-point (pass/fail) outcome decides on Tango's SCORE interval, the
5501
+ * only paired-binary construction that stays valid at a nonzero margin;
5502
+ * - McNemar's exact test VETOES at any non-negative threshold;
5503
+ * - a ZERO-WIDTH interval is refused rather than promoted, in either
5504
+ * direction — [0,0] clears every negative threshold and [g,g] clears every
5505
+ * threshold below g, and both are an absence of evidence, not a result.
5506
+ *
5507
+ * Measured on this function before those guards landed, at a nominal 5 %:
5508
+ * 14.60 % false promotion at n = 40 on a paired-binary noninferiority boundary,
5509
+ * and 88.50 % at n = 6 under a bounded asymmetric null whose true mean paired
5510
+ * delta is exactly 0.
5511
+ *
5512
+ * At small n, where the percentile bootstrap is descriptive only, a
5513
+ * pre-registered exact sign test still carries the bootstrap path.
5514
+ */
5302
5515
  function heldoutSignificance(paired, opts = {}) {
5303
5516
  const deltaThreshold = opts.deltaThreshold ?? 0;
5304
5517
  const confidence = opts.confidence ?? .95;
5305
5518
  const resamples = opts.resamples ?? 2e3;
5306
5519
  const seed = opts.seed ?? 1337;
5307
5520
  const statistic = opts.statistic ?? "mean";
5308
- const decision = pairedDeltaTest(paired.before, paired.after, {
5521
+ const decision = decidePairedPromotion(paired.before, paired.after, {
5309
5522
  confidence,
5310
5523
  resamples,
5311
5524
  statistic,
@@ -5313,7 +5526,12 @@ function heldoutSignificance(paired, opts = {}) {
5313
5526
  threshold: deltaThreshold,
5314
5527
  minPairs: opts.minProductiveRuns
5315
5528
  });
5316
- const bootstrap = decision.bootstrap;
5529
+ const bootstrap = decision.bootstrap ?? pairedBootstrap(paired.before, paired.after, {
5530
+ confidence,
5531
+ resamples,
5532
+ statistic,
5533
+ seed
5534
+ });
5317
5535
  const medianBootstrap = statistic === "median" ? bootstrap : pairedBootstrap(paired.before, paired.after, {
5318
5536
  confidence,
5319
5537
  resamples,
@@ -5328,19 +5546,20 @@ function heldoutSignificance(paired, opts = {}) {
5328
5546
  if (Math.abs(after - before) < 1e-9) ties += 1;
5329
5547
  }
5330
5548
  const tieFraction = n === 0 ? 0 : ties / n;
5331
- const fewRuns = !decision.sufficient;
5332
- const significant = decision.significant;
5333
5549
  return {
5334
5550
  paired,
5335
5551
  bootstrap,
5336
5552
  medianBootstrap,
5553
+ decision,
5554
+ decisionStatistic: decision.statistic,
5555
+ mcnemar: decision.mcnemar,
5337
5556
  tieFraction,
5338
5557
  n,
5339
5558
  minimumRequired: decision.minimumPairs,
5340
5559
  decisionMethod: decision.method,
5341
5560
  pValue: decision.pValue,
5342
- significant,
5343
- fewRuns
5561
+ significant: decision.promote,
5562
+ fewRuns: !decision.sufficient
5344
5563
  };
5345
5564
  }
5346
5565
  /** Detect the native scale of a set of scores: 0-100 when any magnitude clears
@@ -5354,30 +5573,50 @@ function detectScale(values) {
5354
5573
  * a dimension is "regressed" when the CI lower bound < −tolerance (conservative
5355
5574
  * — blocks if the credible worst case exceeds tolerance, which is the right
5356
5575
  * posture for safety dimensions like `hallucination_free`). When `tolerance`
5357
- * is omitted it auto-scales: 0.05 on [0,1], 5 on 0-100. */
5576
+ * is omitted it auto-scales: 0.05 on [0,1], 5 on 0-100.
5577
+ *
5578
+ * The interval comes from {@link decidePairedPromotion}, so a pass/fail
5579
+ * dimension is judged on Tango's score interval rather than a percentile
5580
+ * bootstrap of the mean — `tolerance` is a NONZERO margin, and the bootstrap
5581
+ * is not a valid interval at one. That matters most here because this guard
5582
+ * fails OPEN by construction: `tolerance` is positive, so an interval pinned at
5583
+ * [0,0] never satisfies `low < −tolerance` and a real regression on a safety
5584
+ * dimension would be reported as `regressed: false`. On the median it fails the
5585
+ * same way for the same reason — when most pairs tie, which is automatic for a
5586
+ * pass/fail dimension on {0,1} and on the 0-100 encoding `detectScale` exists
5587
+ * to support, the median CI collapses to [0,0]. Pass `statistic: 'median'` to
5588
+ * restore the pre-0.134 behaviour. */
5358
5589
  function dimensionRegressions(candidate, baseline, scenarioIds, criticalDimensions, opts = {}) {
5359
5590
  const out = [];
5360
5591
  for (const dim of criticalDimensions) {
5361
5592
  const paired = pairHoldout(candidate, baseline, scenarioIds, (s) => s.dimensions[dim]);
5362
5593
  if (paired.before.length === 0) continue;
5363
5594
  const tolerance = opts.tolerance ?? .05 * detectScale([...paired.before, ...paired.after]);
5364
- const bootstrap = pairedBootstrap(paired.before, paired.after, {
5595
+ const bootstrapStatistic = opts.statistic ?? "mean";
5596
+ const shared = {
5365
5597
  confidence: opts.confidence ?? .95,
5366
5598
  resamples: opts.resamples ?? 2e3,
5367
- statistic: "median",
5599
+ statistic: bootstrapStatistic,
5368
5600
  seed: opts.seed ?? 1337
5369
- });
5370
- const regression = pairedDeltaTest(paired.after, paired.before, {
5371
- confidence: opts.confidence ?? .95,
5372
- resamples: opts.resamples ?? 2e3,
5373
- statistic: "median",
5374
- seed: opts.seed ?? 1337,
5601
+ };
5602
+ const guard = decidePairedPromotion(paired.before, paired.after, shared);
5603
+ const regression = decidePairedPromotion(paired.after, paired.before, {
5604
+ ...shared,
5375
5605
  threshold: tolerance
5376
5606
  });
5607
+ const bootstrap = guard.bootstrap ?? pairedBootstrap(paired.before, paired.after, shared);
5377
5608
  out.push({
5378
5609
  dimension: dim,
5379
5610
  bootstrap,
5380
- regressed: regression.significant,
5611
+ bootstrapStatistic,
5612
+ ci: {
5613
+ low: guard.low,
5614
+ high: guard.high
5615
+ },
5616
+ decisionStatistic: guard.statistic,
5617
+ mcnemar: guard.mcnemar,
5618
+ indeterminate: guard.indeterminate,
5619
+ regressed: bootstrap.low < -tolerance || regression.promote,
5381
5620
  tolerance,
5382
5621
  n: paired.before.length
5383
5622
  });
@@ -5430,7 +5669,8 @@ function defaultProductionGate(options) {
5430
5669
  seed,
5431
5670
  statistic: heldoutStatistic
5432
5671
  });
5433
- delta = heldoutStatistic === "median" ? sig.bootstrap.median : sig.bootstrap.mean;
5672
+ const dec = sig.decision;
5673
+ delta = dec.delta;
5434
5674
  const heldoutPass = sig.significant;
5435
5675
  contributing.push({
5436
5676
  name: "heldout-significance",
@@ -5438,13 +5678,20 @@ function defaultProductionGate(options) {
5438
5678
  detail: {
5439
5679
  n: sig.n,
5440
5680
  delta,
5681
+ decisionStatistic: sig.decisionStatistic,
5682
+ decisionMethod: sig.decisionMethod,
5683
+ binaryScale: dec.binaryScale,
5684
+ mcnemar: dec.mcnemar,
5685
+ indeterminate: dec.indeterminate,
5441
5686
  deltaMean: sig.bootstrap.mean,
5442
5687
  deltaMedianDiagnostic: sig.medianBootstrap.median,
5443
5688
  deltaMedian: sig.medianBootstrap.median,
5444
5689
  tieFraction: sig.tieFraction,
5445
- ciLow: sig.bootstrap.low,
5446
- ciHigh: sig.bootstrap.high,
5447
- confidence: sig.bootstrap.confidence,
5690
+ ciLow: dec.low,
5691
+ ciHigh: dec.high,
5692
+ bootstrapCiLow: sig.bootstrap.low,
5693
+ bootstrapCiHigh: sig.bootstrap.high,
5694
+ confidence: dec.confidence,
5448
5695
  deltaThreshold,
5449
5696
  fewRuns: sig.fewRuns
5450
5697
  }
@@ -5452,7 +5699,8 @@ function defaultProductionGate(options) {
5452
5699
  if (sig.fewRuns) requiredUnavailable.add("heldout-significance");
5453
5700
  if (!heldoutPass) {
5454
5701
  const tieNote = sig.tieFraction >= .4 ? `; ${(sig.tieFraction * 100).toFixed(0)}% tied scenarios` : "";
5455
- reasons.push(sig.fewRuns ? `held-out: only ${sig.n} paired runs (< ${sig.minimumRequired}) — too few to claim significance` : `held-out CI.low ${sig.bootstrap.low.toFixed(3)} ≤ threshold ${deltaThreshold} (${heldoutStatistic} Δ ${delta.toFixed(3)}, ${(sig.bootstrap.confidence * 100).toFixed(0)}% CI [${sig.bootstrap.low.toFixed(3)}, ${sig.bootstrap.high.toFixed(3)}]${tieNote})`);
5702
+ const ci = `${(dec.confidence * 100).toFixed(0)}% CI [${dec.low.toFixed(3)}, ${dec.high.toFixed(3)}]`;
5703
+ reasons.push(sig.fewRuns ? `held-out: only ${sig.n} paired runs (< ${sig.minimumRequired}) — too few to claim significance` : dec.indeterminate ? `held-out: ${dec.indeterminateCause}, so the paired CI is ${ci} and carries no direction — it cannot clear threshold ${deltaThreshold} on evidence${tieNote}` : dec.exactTestVetoes ? `held-out: McNemar exact p=${dec.mcnemar?.pValue.toExponential(2)} does not reject at α=${(1 - dec.confidence).toFixed(4)} (${dec.label} Δ ${delta.toFixed(3)}, ${ci}${tieNote})` : `held-out CI.low ${dec.low.toFixed(3)} ≤ threshold ${deltaThreshold} (${dec.label} Δ ${delta.toFixed(3)}, ${ci}${tieNote})`);
5456
5704
  }
5457
5705
  }
5458
5706
  const dimensionsProvided = options.criticalDimensions !== void 0;
@@ -5480,7 +5728,11 @@ function defaultProductionGate(options) {
5480
5728
  missingDimensions,
5481
5729
  regressions: dimRegs.map((result) => ({
5482
5730
  dimension: result.dimension,
5483
- ciLow: result.bootstrap.low,
5731
+ ciLow: result.ci.low,
5732
+ ciHigh: result.ci.high,
5733
+ decisionStatistic: result.decisionStatistic,
5734
+ indeterminate: result.indeterminate,
5735
+ bootstrapCiLow: result.bootstrap.low,
5484
5736
  median: result.bootstrap.median,
5485
5737
  tolerance: result.tolerance,
5486
5738
  n: result.n,
@@ -5492,7 +5744,7 @@ function defaultProductionGate(options) {
5492
5744
  requiredUnavailable.add("dimension-regression");
5493
5745
  reasons.push(`critical dimension(s) were not scored: ${missingDimensions.join(", ")}`);
5494
5746
  }
5495
- if (regressed.length > 0) reasons.push(`critical dimension(s) regressed: ${regressed.map((result) => `${result.dimension} CI.low ${result.bootstrap.low.toFixed(3)} < -${result.tolerance}`).join("; ")}`);
5747
+ if (regressed.length > 0) reasons.push(`critical dimension(s) regressed: ${regressed.map((result) => `${result.dimension} CI.low ${result.ci.low.toFixed(3)} < -${result.tolerance}`).join("; ")}`);
5496
5748
  }
5497
5749
  const budgetUsd = options.budgetUsd;
5498
5750
  const budgetConfigured = budgetUsd !== void 0;
@@ -5646,8 +5898,10 @@ function extractText(artifact) {
5646
5898
  //#endregion
5647
5899
  //#region src/campaign/gates/heldout-gate.ts
5648
5900
  /**
5649
- * Composable held-out gate: ships only when the PAIRED bootstrap CI lower bound
5650
- * of the candidate-minus-baseline composite delta clears `deltaThreshold`.
5901
+ * Composable held-out gate: ships only when the lower bound of the DECIDING
5902
+ * paired interval on the candidate-minus-baseline composite delta clears
5903
+ * `deltaThreshold` — Tango's score interval on a pass/fail holdout, the mean
5904
+ * bootstrap otherwise. See {@link decidePairedPromotion}.
5651
5905
  */
5652
5906
  function heldOutGate(options) {
5653
5907
  const deltaThreshold = options.deltaThreshold ?? .5;
@@ -5667,24 +5921,35 @@ function heldOutGate(options) {
5667
5921
  resamples,
5668
5922
  seed
5669
5923
  });
5670
- const delta = sig.bootstrap.mean;
5924
+ const dec = sig.decision;
5925
+ const delta = dec.delta;
5671
5926
  const passed = sig.significant;
5672
5927
  const status = sig.fewRuns ? "not_evaluated" : passed ? "pass" : "fail";
5673
5928
  const tieNote = sig.tieFraction >= .4 ? `, ${(sig.tieFraction * 100).toFixed(0)}% tied` : "";
5674
- const ci = `${(sig.bootstrap.confidence * 100).toFixed(0)}% CI [${sig.bootstrap.low.toFixed(3)}, ${sig.bootstrap.high.toFixed(3)}]`;
5929
+ const ci = `${(dec.confidence * 100).toFixed(0)}% CI [${dec.low.toFixed(3)}, ${dec.high.toFixed(3)}]`;
5930
+ const held = `held-out ${dec.label} Δ ${delta.toFixed(3)}`;
5931
+ const holdReason = sig.fewRuns ? `held-out: only ${sig.n} paired runs; ${sig.minimumRequired} required — too few to claim significance` : dec.indeterminate ? `held-out: ${dec.indeterminateCause}, so the paired CI is ${ci} and carries no direction — it cannot clear ${deltaThreshold} on evidence (n=${sig.n}${tieNote})` : dec.exactTestVetoes ? `${held}, McNemar exact p=${dec.mcnemar?.pValue.toExponential(2)} does not reject at α=${(1 - dec.confidence).toFixed(4)} (${ci}, n=${sig.n}${tieNote})` : `${held}, CI.low ${dec.low.toFixed(3)} ≤ ${deltaThreshold} (${ci}, n=${sig.n}${tieNote})`;
5675
5932
  return {
5676
5933
  decision: passed ? "ship" : "hold",
5677
- reasons: passed ? [`held-out mean Δ ${delta.toFixed(3)}, CI.low ${sig.bootstrap.low.toFixed(3)} > ${deltaThreshold} (${ci}, n=${sig.n}${tieNote})`] : [sig.fewRuns ? `held-out: only ${sig.n} paired runs; ${sig.minimumRequired} required — too few to claim significance` : `held-out mean Δ ${delta.toFixed(3)}, CI.low ${sig.bootstrap.low.toFixed(3)} ≤ ${deltaThreshold} (${ci}, n=${sig.n}${tieNote})`],
5934
+ reasons: passed ? [`${held}, CI.low ${dec.low.toFixed(3)} > ${deltaThreshold} (${ci}, n=${sig.n}${tieNote})`] : [holdReason],
5678
5935
  contributingGates: [{
5679
5936
  name: "heldOutGate",
5680
5937
  status,
5681
5938
  detail: {
5682
- deltaMean: delta,
5939
+ deltaMean: sig.bootstrap.mean,
5940
+ decidingDelta: delta,
5941
+ decisionStatistic: sig.decisionStatistic,
5942
+ decisionMethod: sig.decisionMethod,
5943
+ binaryScale: dec.binaryScale,
5944
+ mcnemar: dec.mcnemar,
5945
+ indeterminate: dec.indeterminate,
5683
5946
  deltaMedianDiagnostic: sig.medianBootstrap.median,
5684
5947
  tieFraction: sig.tieFraction,
5685
- ciLow: sig.bootstrap.low,
5686
- ciHigh: sig.bootstrap.high,
5687
- confidence: sig.bootstrap.confidence,
5948
+ ciLow: dec.low,
5949
+ ciHigh: dec.high,
5950
+ bootstrapCiLow: sig.bootstrap.low,
5951
+ bootstrapCiHigh: sig.bootstrap.high,
5952
+ confidence: dec.confidence,
5688
5953
  n: sig.n,
5689
5954
  deltaThreshold,
5690
5955
  fewRuns: sig.fewRuns,
@@ -5793,29 +6058,44 @@ function buildEvidenceVector(ctx, objectives, opts = {}) {
5793
6058
  const n = paired.before.length;
5794
6059
  const floorTolerance = obj.floorTolerance ?? .05 * detectScale([...paired.before, ...paired.after]);
5795
6060
  const gainThreshold = obj.gainThreshold ?? 0;
5796
- const improvement = pairedDeltaTest(before, after, {
6061
+ const bootstrapStatistic = opts.statistic ?? "mean";
6062
+ const improvement = decidePairedPromotion(before, after, {
5797
6063
  confidence,
5798
6064
  resamples,
5799
- statistic: "median",
6065
+ statistic: bootstrapStatistic,
5800
6066
  seed,
5801
6067
  threshold: gainThreshold,
5802
6068
  minPairs: opts.minProductiveRuns
5803
6069
  });
5804
- const regression = pairedDeltaTest(after, before, {
6070
+ const regression = decidePairedPromotion(after, before, {
5805
6071
  confidence,
5806
6072
  resamples,
5807
- statistic: "median",
6073
+ statistic: bootstrapStatistic,
5808
6074
  seed,
5809
6075
  threshold: floorTolerance,
5810
6076
  minPairs: opts.minProductiveRuns
5811
6077
  });
5812
- const bootstrap = improvement.bootstrap;
5813
- const verdict = !improvement.sufficient ? "few_runs" : regression.significant ? "regressed" : improvement.significant ? "improved" : "flat";
6078
+ const bootstrap = improvement.bootstrap ?? pairedBootstrap(before, after, {
6079
+ confidence,
6080
+ resamples,
6081
+ statistic: bootstrapStatistic,
6082
+ seed
6083
+ });
6084
+ const floorBreached = bootstrap.low < -floorTolerance || regression.promote;
6085
+ const verdict = !improvement.sufficient ? "few_runs" : floorBreached ? "regressed" : improvement.promote ? "improved" : "flat";
5814
6086
  axes.push({
5815
6087
  name: obj.name,
5816
6088
  source: obj.source,
5817
6089
  direction: obj.direction,
5818
6090
  bootstrap,
6091
+ bootstrapStatistic,
6092
+ ci: {
6093
+ low: improvement.low,
6094
+ high: improvement.high
6095
+ },
6096
+ decisionStatistic: improvement.statistic,
6097
+ mcnemar: improvement.mcnemar,
6098
+ indeterminate: improvement.indeterminate,
5819
6099
  n,
5820
6100
  minimumRequired: improvement.minimumPairs,
5821
6101
  decisionMethod: improvement.method,
@@ -5851,8 +6131,14 @@ const paretoPolicy = (ev) => {
5851
6131
  verdict: ax.verdict,
5852
6132
  n: ax.n,
5853
6133
  deltaMedian: ax.bootstrap.median,
5854
- ciLow: ax.bootstrap.low,
5855
- ciHigh: ax.bootstrap.high,
6134
+ ciLow: ax.ci.low,
6135
+ ciHigh: ax.ci.high,
6136
+ decisionStatistic: ax.decisionStatistic,
6137
+ decisionMethod: ax.decisionMethod,
6138
+ mcnemar: ax.mcnemar,
6139
+ indeterminate: ax.indeterminate,
6140
+ bootstrapCiLow: ax.bootstrap.low,
6141
+ bootstrapCiHigh: ax.bootstrap.high,
5856
6142
  confidence: ax.bootstrap.confidence,
5857
6143
  gainThreshold: ax.gainThreshold,
5858
6144
  floorTolerance: ax.floorTolerance
@@ -5865,13 +6151,13 @@ const paretoPolicy = (ev) => {
5865
6151
  const reasons = [];
5866
6152
  if (regressed.length > 0) {
5867
6153
  decision = "hold";
5868
- for (const a of regressed) reasons.push(`objective '${a.name}' regressed: good-direction CI.low ${a.bootstrap.low.toFixed(3)} < -${a.floorTolerance} (n=${a.n})`);
6154
+ for (const a of regressed) reasons.push(`objective '${a.name}' regressed: good-direction CI.low ${a.ci.low.toFixed(3)} < -${a.floorTolerance} (n=${a.n})`);
5869
6155
  } else if (fewRuns.length > 0) {
5870
6156
  decision = "need_more_work";
5871
6157
  for (const a of fewRuns) reasons.push(`objective '${a.name}' has only n=${a.n} paired runs — insufficient evidence to claim significance`);
5872
6158
  } else if (improved.length > 0) {
5873
6159
  decision = "ship";
5874
- reasons.push(`Pareto improvement at the confidence level: ${improved.map((a) => `'${a.name}' +${a.bootstrap.median.toFixed(3)} (CI.low ${a.bootstrap.low.toFixed(3)})`).join(", ")}; no objective regressed`);
6160
+ reasons.push(`Pareto improvement at the confidence level: ${improved.map((a) => `'${a.name}' +${a.ci.low > 0 ? a.ci.low.toFixed(3) : a.bootstrap.mean.toFixed(3)} (CI.low ${a.ci.low.toFixed(3)})`).join(", ")}; no objective regressed`);
5875
6161
  } else {
5876
6162
  decision = "hold";
5877
6163
  reasons.push("no Pareto improvement: candidate statistically equivalent to baseline on every objective");
@@ -7833,6 +8119,6 @@ function skillOptOptimizationMethod(config) {
7833
8119
  };
7834
8120
  }
7835
8121
  //#endregion
7836
- export { SEARCH_LEDGER_FILE_CONTEXT as $, acquireSingleRunLock as A, dominates as At, surfaceContentHash as B, summarizeAgentReceiptIntegrity as Bt, detectScale as C, llmJudge as Ct, runCanaries as D, fileVerdictCache as Dt, pairHoldout as E, contentHash as Et, assertCodeSurfaceIdentity as F, minimumPairsForPairedDeltaTest as Ft, DEFAULT_MUTATION_PRIMITIVES as G, campaignBreakdown as H, JudgeParseError as Ht, assertComponentSurface as I, pairedDeltaTest as It, planCampaignRun as J, buildReflectionPrompt as K, codeSurfaceIdentityMaterial as L, BackendIntegrityError as Lt, compareOptimizationMethods as M, paretoFrontierWithCrowding as Mt, costFromLedgerSummary as N, scalarScore as Nt, composeGate as O, inMemoryVerdictCache as Ot, optimizationTokenUsageFromSummary as P, recoverTruncatedJson as Pt, inMemoryCampaignStorage as Q, componentSurfaceIdentityMaterial as R, assertRealAgentReceipts as Rt, defaultProductionGate as S, hashScenarios as St, heldoutSignificance as T, canonicalJson as Tt, campaignMeanComposite as U, surfaceHash as V, summarizeBackendIntegrity as Vt, compareRankKeys as W, createRunCostLedger as X, runCampaign as Y, fsCampaignStorage as Z, buildEvidenceVector as _, redTeamReport as _t, emitLoopProvenance as a, assertCampaignDesign as at, powerPreflight as b, Dataset as bt, provenanceRecordPath as c, campaignSplitDigest as ct, runImprovementLoop as d, REFERENCE_EQUIVALENCE_INPUT_LIMITS as dt, SearchLedgerConflictError as et, runOptimization as f, REFERENCE_EQUIVALENCE_JUDGE_VERSION as ft, gepaOptimizationMethod as g, redTeamDataset as gt, labelTrustRank as h, DEFAULT_RED_TEAM_CORPUS as ht, canonicalDigest as i, tangleTracesRoot as it, assertOptimizationResult as j, paretoFrontier as jt, externalTextOptimizationMethod as k, crowdingDistance as kt, provenanceSpansPath as l, campaignSplitDigestFromIdentities as lt, isProposedCandidate as m, runReferenceEquivalenceJudge as mt, buildLoopProvenanceRecord as n, SearchLedgerIntegrityError as nt, loopProvenanceArgsFromResult as o, assertCampaignSplitIdentity as ot, runEval as p, createReferenceEquivalenceJudge as pt, parseReflectionResponse as q, campaignMeasurementDigest as r, resolveRunDir as rt, loopProvenanceSpans as s, campaignScenarioIdentity as st, skillOptOptimizationMethod as t, SearchLedgerError as tt, verifyLoopProvenanceRecord as u, openAutoPr as ut, paretoPolicy as v, scoreRedTeamOutput as vt, dimensionRegressions as w, cachedJudge as wt, heldOutGate as x, HoldoutLockedError as xt, paretoSignificanceGate as y, toolNamesForRun as yt, renderSurfaceDiff as z, assertRealBackend as zt };
8122
+ export { SEARCH_LEDGER_FILE_CONTEXT as $, acquireSingleRunLock as A, dominates as At, surfaceContentHash as B, assertRealAgentReceipts as Bt, detectScale as C, llmJudge as Ct, runCanaries as D, fileVerdictCache as Dt, pairHoldout as E, contentHash as Et, assertCodeSurfaceIdentity as F, decidePairedPromotion as Ft, DEFAULT_MUTATION_PRIMITIVES as G, campaignBreakdown as H, summarizeAgentReceiptIntegrity as Ht, assertComponentSurface as I, pairedDecisionShape as It, planCampaignRun as J, buildReflectionPrompt as K, codeSurfaceIdentityMaterial as L, minimumPairsForPairedDeltaTest as Lt, compareOptimizationMethods as M, paretoFrontierWithCrowding as Mt, costFromLedgerSummary as N, scalarScore as Nt, composeGate as O, inMemoryVerdictCache as Ot, optimizationTokenUsageFromSummary as P, recoverTruncatedJson as Pt, inMemoryCampaignStorage as Q, componentSurfaceIdentityMaterial as R, pairedDeltaTest as Rt, defaultProductionGate as S, hashScenarios as St, heldoutSignificance as T, canonicalJson as Tt, campaignMeanComposite as U, summarizeBackendIntegrity as Ut, surfaceHash as V, assertRealBackend as Vt, compareRankKeys as W, JudgeParseError as Wt, createRunCostLedger as X, runCampaign as Y, fsCampaignStorage as Z, buildEvidenceVector as _, redTeamReport as _t, emitLoopProvenance as a, assertCampaignDesign as at, powerPreflight as b, Dataset as bt, provenanceRecordPath as c, campaignSplitDigest as ct, runImprovementLoop as d, REFERENCE_EQUIVALENCE_INPUT_LIMITS as dt, SearchLedgerConflictError as et, runOptimization as f, REFERENCE_EQUIVALENCE_JUDGE_VERSION as ft, gepaOptimizationMethod as g, redTeamDataset as gt, labelTrustRank as h, DEFAULT_RED_TEAM_CORPUS as ht, canonicalDigest as i, tangleTracesRoot as it, assertOptimizationResult as j, paretoFrontier as jt, externalTextOptimizationMethod as k, crowdingDistance as kt, provenanceSpansPath as l, campaignSplitDigestFromIdentities as lt, isProposedCandidate as m, runReferenceEquivalenceJudge as mt, buildLoopProvenanceRecord as n, SearchLedgerIntegrityError as nt, loopProvenanceArgsFromResult as o, assertCampaignSplitIdentity as ot, runEval as p, createReferenceEquivalenceJudge as pt, parseReflectionResponse as q, campaignMeasurementDigest as r, resolveRunDir as rt, loopProvenanceSpans as s, campaignScenarioIdentity as st, skillOptOptimizationMethod as t, SearchLedgerError as tt, verifyLoopProvenanceRecord as u, openAutoPr as ut, paretoPolicy as v, scoreRedTeamOutput as vt, dimensionRegressions as w, cachedJudge as wt, heldOutGate as x, HoldoutLockedError as xt, paretoSignificanceGate as y, toolNamesForRun as yt, renderSurfaceDiff as z, BackendIntegrityError as zt };
7837
8123
 
7838
- //# sourceMappingURL=skillopt-optimization-method-C9yGg-4C.js.map
8124
+ //# sourceMappingURL=skillopt-optimization-method-0UmPD6aP.js.map