@tangle-network/agent-eval 0.133.2 → 0.134.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (131) hide show
  1. package/CHANGELOG.md +218 -0
  2. package/dist/analyst/index.d.ts +11 -35
  3. package/dist/analyst/index.d.ts.map +1 -1
  4. package/dist/analyst/index.js +4 -53
  5. package/dist/analyst/index.js.map +1 -1
  6. package/dist/{analyze-runs-BmX-h_yn.d.ts → analyze-runs-DMo3Lb_y.d.ts} +4 -4
  7. package/dist/{analyze-runs-BmX-h_yn.d.ts.map → analyze-runs-DMo3Lb_y.d.ts.map} +1 -1
  8. package/dist/{analyze-runs-B-afTpCv.js → analyze-runs-qk8op0tN.js} +63 -42
  9. package/dist/analyze-runs-qk8op0tN.js.map +1 -0
  10. package/dist/baseline-BaPxoROc.js +149 -0
  11. package/dist/baseline-BaPxoROc.js.map +1 -0
  12. package/dist/{baseline-hG3K85h4.d.ts → baseline-D_fT6277.d.ts} +43 -11
  13. package/dist/baseline-D_fT6277.d.ts.map +1 -0
  14. package/dist/benchmarks/index.d.ts +1 -1
  15. package/dist/benchmarks/index.js +1 -1
  16. package/dist/{benchmarks-CJr1H1_a.js → benchmarks-v5piCeDl.js} +3 -3
  17. package/dist/{benchmarks-CJr1H1_a.js.map → benchmarks-v5piCeDl.js.map} +1 -1
  18. package/dist/builder-eval/index.js +1 -1
  19. package/dist/campaign/index.d.ts +5 -4
  20. package/dist/campaign/index.js +4 -3
  21. package/dist/{campaign-BJjn1rhw.js → campaign-DEC_7DLn.js} +12 -6
  22. package/dist/{campaign-BJjn1rhw.js.map → campaign-DEC_7DLn.js.map} +1 -1
  23. package/dist/{client-COvaLoQG.d.ts → client-BIyh1RCr.d.ts} +29 -15
  24. package/dist/client-BIyh1RCr.d.ts.map +1 -0
  25. package/dist/{client-CYzbdJOZ.js → client-LIuo-KPv.js} +19 -7
  26. package/dist/client-LIuo-KPv.js.map +1 -0
  27. package/dist/contract/index.d.ts +12 -11
  28. package/dist/contract/index.d.ts.map +1 -1
  29. package/dist/contract/index.js +12 -18
  30. package/dist/contract/index.js.map +1 -1
  31. package/dist/{default-registry-Cl3pHo4n.d.ts → default-registry-Brxr728w.d.ts} +4 -268
  32. package/dist/default-registry-Brxr728w.d.ts.map +1 -0
  33. package/dist/{default-registry-D3T9XbuY.js → default-registry-IjYs7T8l.js} +4 -61
  34. package/dist/default-registry-IjYs7T8l.js.map +1 -0
  35. package/dist/{eval-campaign-DXhpZghy.js → eval-campaign-CvPcvqXC.js} +2 -2
  36. package/dist/{eval-campaign-DXhpZghy.js.map → eval-campaign-CvPcvqXC.js.map} +1 -1
  37. package/dist/hosted/index.d.ts +2 -2
  38. package/dist/hosted/index.d.ts.map +1 -1
  39. package/dist/hosted/index.js +1 -1
  40. package/dist/{index-C7Wue8R6.d.ts → index-BoJNQR6n.d.ts} +29 -11
  41. package/dist/index-BoJNQR6n.d.ts.map +1 -0
  42. package/dist/{index-BREtv3ZZ.d.ts → index-C21xKtxu.d.ts} +4 -4
  43. package/dist/{index-BREtv3ZZ.d.ts.map → index-C21xKtxu.d.ts.map} +1 -1
  44. package/dist/{index-DSC51roc.d.ts → index-DSC51roc2.d.ts} +1 -1
  45. package/dist/index-DSC51roc2.d.ts.map +1 -0
  46. package/dist/{index-nhIYz9hn.d.ts → index-DuhJaaiH.d.ts} +68 -7
  47. package/dist/index-DuhJaaiH.d.ts.map +1 -0
  48. package/dist/index.d.ts +60 -13
  49. package/dist/index.d.ts.map +1 -1
  50. package/dist/index.js +134 -27
  51. package/dist/index.js.map +1 -1
  52. package/dist/ledger-core/index.d.ts +2 -2
  53. package/dist/ledger-core/index.js +2 -2
  54. package/dist/{ledger-core-CPZfcrC2.js → ledger-core-DAKFKRzi.js} +136 -18
  55. package/dist/ledger-core-DAKFKRzi.js.map +1 -0
  56. package/dist/matrix/index.d.ts +1 -1
  57. package/dist/meta-eval/index.d.ts +1 -1
  58. package/dist/meta-eval/index.js +2 -2
  59. package/dist/multishot/index.d.ts +2 -2
  60. package/dist/openapi.json +1 -1
  61. package/dist/{paired-arms-6XItKzd1.js → paired-arms-CA_8pN01.js} +2 -2
  62. package/dist/{paired-arms-6XItKzd1.js.map → paired-arms-CA_8pN01.js.map} +1 -1
  63. package/dist/pipelines/index.d.ts +1 -1
  64. package/dist/pipelines/index.js +3 -2
  65. package/dist/pipelines/index.js.map +1 -1
  66. package/dist/proposal-findings-DCawte-y.js +164 -0
  67. package/dist/proposal-findings-DCawte-y.js.map +1 -0
  68. package/dist/{release-report-wuilQkvK.js → release-report-BVZBmRZp.js} +2 -2
  69. package/dist/{release-report-wuilQkvK.js.map → release-report-BVZBmRZp.js.map} +1 -1
  70. package/dist/{release-report-CjHWa8Ia.d.ts → release-report-CuULWKyk.d.ts} +2 -2
  71. package/dist/{release-report-CjHWa8Ia.d.ts.map → release-report-CuULWKyk.d.ts.map} +1 -1
  72. package/dist/reporting.d.ts +3 -3
  73. package/dist/reporting.js +4 -4
  74. package/dist/{researcher-CbSKhK8z.d.ts → researcher-DVtruQ9U.d.ts} +2 -2
  75. package/dist/{researcher-CbSKhK8z.d.ts.map → researcher-DVtruQ9U.d.ts.map} +1 -1
  76. package/dist/{reward-hacking-Dl2UBzej.js → reward-hacking-DCdRK9TY.js} +2 -2
  77. package/dist/{reward-hacking-Dl2UBzej.js.map → reward-hacking-DCdRK9TY.js.map} +1 -1
  78. package/dist/rl.d.ts +2 -2
  79. package/dist/rl.d.ts.map +1 -1
  80. package/dist/rl.js +18 -5
  81. package/dist/rl.js.map +1 -1
  82. package/dist/{rubric-predictive-validity-QG7ydk0s.js → rubric-predictive-validity-D6Q6n9oq.js} +2 -2
  83. package/dist/{rubric-predictive-validity-QG7ydk0s.js.map → rubric-predictive-validity-D6Q6n9oq.js.map} +1 -1
  84. package/dist/{semantic-concept-judge-BypLt6Fw.js → semantic-concept-judge-C0P1VTXD.js} +2 -3
  85. package/dist/{semantic-concept-judge-BypLt6Fw.js.map → semantic-concept-judge-C0P1VTXD.js.map} +1 -1
  86. package/dist/{skill-usage-BaaxFSJR.d.ts → skill-usage-BDQVPIG1.d.ts} +3 -2
  87. package/dist/skill-usage-BDQVPIG1.d.ts.map +1 -0
  88. package/dist/{skillopt-optimization-method-CF6a327Q.js → skillopt-optimization-method-BY6vKLJB.js} +169 -53
  89. package/dist/skillopt-optimization-method-BY6vKLJB.js.map +1 -0
  90. package/dist/{skillopt-optimization-method-wHF5xsUv.d.ts → skillopt-optimization-method-DJ3l4w8W.d.ts} +20 -18
  91. package/dist/skillopt-optimization-method-DJ3l4w8W.d.ts.map +1 -0
  92. package/dist/{statistics-DbvkkDPa.d.ts → statistics-D_4Snl-5.d.ts} +158 -30
  93. package/dist/statistics-D_4Snl-5.d.ts.map +1 -0
  94. package/dist/{statistics-DWM_AyLe.js → statistics-RwRNu2__.js} +546 -98
  95. package/dist/statistics-RwRNu2__.js.map +1 -0
  96. package/dist/{summary-report-Ci17nIdU.js → summary-report-BxtossFi.js} +3 -3
  97. package/dist/{summary-report-Ci17nIdU.js.map → summary-report-BxtossFi.js.map} +1 -1
  98. package/dist/{summary-report-CFnQgNfg.d.ts → summary-report-DGp0-_XO.d.ts} +61 -4
  99. package/dist/summary-report-DGp0-_XO.d.ts.map +1 -0
  100. package/dist/{baseline-DcX5hQDv.js → tool-use-metrics-DEGMKycK.js} +2 -114
  101. package/dist/tool-use-metrics-DEGMKycK.js.map +1 -0
  102. package/dist/types-DVjczBM9.d.ts +276 -0
  103. package/dist/types-DVjczBM9.d.ts.map +1 -0
  104. package/dist/{types-BokuXvOG.d.ts → types-DiWLru6Z.d.ts} +20 -37
  105. package/dist/types-DiWLru6Z.d.ts.map +1 -0
  106. package/docs/campaign-proposers.md +5 -0
  107. package/docs/design/statistics-decisions.md +271 -0
  108. package/docs/design.md +1 -0
  109. package/docs/insight-report.md +1 -1
  110. package/docs/research-report-methodology.md +4 -1
  111. package/package.json +2 -1
  112. package/dist/analyze-runs-B-afTpCv.js.map +0 -1
  113. package/dist/baseline-DcX5hQDv.js.map +0 -1
  114. package/dist/baseline-hG3K85h4.d.ts.map +0 -1
  115. package/dist/client-COvaLoQG.d.ts.map +0 -1
  116. package/dist/client-CYzbdJOZ.js.map +0 -1
  117. package/dist/default-registry-Cl3pHo4n.d.ts.map +0 -1
  118. package/dist/default-registry-D3T9XbuY.js.map +0 -1
  119. package/dist/index-C7Wue8R6.d.ts.map +0 -1
  120. package/dist/index-DSC51roc.d.ts.map +0 -1
  121. package/dist/index-nhIYz9hn.d.ts.map +0 -1
  122. package/dist/ledger-core-CPZfcrC2.js.map +0 -1
  123. package/dist/run-score-iEEAWiBY.js +0 -41
  124. package/dist/run-score-iEEAWiBY.js.map +0 -1
  125. package/dist/skill-usage-BaaxFSJR.d.ts.map +0 -1
  126. package/dist/skillopt-optimization-method-CF6a327Q.js.map +0 -1
  127. package/dist/skillopt-optimization-method-wHF5xsUv.d.ts.map +0 -1
  128. package/dist/statistics-DWM_AyLe.js.map +0 -1
  129. package/dist/statistics-DbvkkDPa.d.ts.map +0 -1
  130. package/dist/summary-report-CFnQgNfg.d.ts.map +0 -1
  131. package/dist/types-BokuXvOG.d.ts.map +0 -1
@@ -1,12 +1,12 @@
1
1
  import { i as JudgeError, s as ValidationError, t as AgentEvalError } from "./errors-8YnH8WlF.js";
2
2
  import { c as costForTokenPricing, i as CostLedger, t as CostAccountingIncompleteError } from "./cost-ledger-BrJxbrMy.js";
3
3
  import { f as maximumChargeForLlmRequest, l as costReceiptFromLlm, m as stripFencedJson, u as costReceiptFromLlmError } from "./llm-client-ClPW-dWB.js";
4
- import { i as combineAbortSignals, r as clamp01 } from "./run-score-iEEAWiBY.js";
4
+ import { a as clamp01, o as combineAbortSignals, t as assertProposalFindings } from "./proposal-findings-DCawte-y.js";
5
5
  import { n as mapConcurrent } from "./concurrency-MUjT7VjM.js";
6
- import { a as confidenceInterval, j as weightedComposite, v as pairedBootstrap } from "./statistics-DWM_AyLe.js";
6
+ import { C as pairedBootstrap, D as pairedSignTest, I as weightedComposite, u as confidenceInterval } from "./statistics-RwRNu2__.js";
7
7
  import { t as DEFAULT_REDACTION_RULES } from "./redact-7Aq1ukl-.js";
8
- import { a as campaignCellExecutionEvidence, l as projectCampaignCellQuality, t as detectRewardHacking } from "./reward-hacking-Dl2UBzej.js";
9
- import { a as tryWithLedgerFileLock, i as appendLedgerLine, m as tryAcquireAtomicFileLock, p as probeAtomicFileLock } from "./ledger-core-CPZfcrC2.js";
8
+ import { a as campaignCellExecutionEvidence, l as projectCampaignCellQuality, t as detectRewardHacking } from "./reward-hacking-DCdRK9TY.js";
9
+ import { a as appendLedgerLine, h as tryAcquireAtomicFileLock, m as probeAtomicFileLock, o as tryWithLedgerFileLock } from "./ledger-core-DAKFKRzi.js";
10
10
  import { createRequire } from "node:module";
11
11
  import { z } from "zod";
12
12
  import { appendFileSync, existsSync, readFileSync, writeFileSync } from "node:fs";
@@ -155,6 +155,51 @@ function assertBackendReport(report, opts) {
155
155
  return report;
156
156
  }
157
157
  //#endregion
158
+ //#region src/paired-delta-test.ts
159
+ /** Smallest all-positive sample that can clear a one-sided exact sign test. */
160
+ function minimumPairsForPairedDeltaTest(confidence = .95) {
161
+ if (!Number.isFinite(confidence) || confidence <= 0 || confidence >= 1) throw new Error(`minimumPairsForPairedDeltaTest: confidence must be in (0,1), got ${confidence}`);
162
+ const oneSidedAlpha = (1 - confidence) / 2;
163
+ return Math.ceil(Math.log2(1 / oneSidedAlpha));
164
+ }
165
+ /**
166
+ * Tests whether a paired candidate-minus-baseline delta clears a threshold.
167
+ *
168
+ * At 20 or more pairs, the percentile bootstrap lower bound carries the
169
+ * decision. Below that point the interval is descriptive only, so the function
170
+ * switches to a pre-registered one-sided exact sign test. The exact path is
171
+ * deliberately conservative: it requires both a point estimate above the
172
+ * threshold and enough consistently positive paired differences.
173
+ */
174
+ function pairedDeltaTest(before, after, options = {}) {
175
+ const threshold = options.threshold ?? 0;
176
+ if (!Number.isFinite(threshold)) throw new Error(`pairedDeltaTest: threshold must be finite, got ${threshold}`);
177
+ const exactMinimum = minimumPairsForPairedDeltaTest(options.confidence ?? .95);
178
+ const requestedMinimum = options.minPairs ?? exactMinimum;
179
+ if (!Number.isInteger(requestedMinimum) || requestedMinimum < 1) throw new Error(`pairedDeltaTest: minPairs must be a positive integer, got ${requestedMinimum}`);
180
+ const minimumPairs = Math.max(requestedMinimum, exactMinimum);
181
+ const bootstrap = pairedBootstrap(before, after, options);
182
+ const sufficient = bootstrap.n >= minimumPairs;
183
+ if (bootstrap.gateEligible) return {
184
+ bootstrap,
185
+ method: "bootstrap-ci",
186
+ pValue: null,
187
+ minimumPairs,
188
+ sufficient,
189
+ significant: sufficient && bootstrap.low > threshold
190
+ };
191
+ const exact = pairedSignTest(before.map((value, index) => after[index] - value - threshold), "greater");
192
+ const estimate = options.statistic === "mean" ? bootstrap.mean : bootstrap.median;
193
+ return {
194
+ bootstrap,
195
+ method: "exact-sign",
196
+ pValue: exact.pValue,
197
+ minimumPairs,
198
+ sufficient,
199
+ significant: sufficient && estimate > threshold && exact.pValue <= (1 - bootstrap.confidence) / 2
200
+ };
201
+ }
202
+ //#endregion
158
203
  //#region src/json-recovery.ts
159
204
  /**
160
205
  * Truncation-tolerant JSON recovery — shared by every parser that reads JSON
@@ -3835,7 +3880,7 @@ function describeExternalScenario(scenario, label, maxChars, describe) {
3835
3880
  assertJsonValue(data, `${label} scenario '${scenario.id}'`);
3836
3881
  const serializedChars = JSON.stringify(data).length;
3837
3882
  if (serializedChars > maxChars) throw new Error(`${label} scenario '${scenario.id}' exceeds maxEvidenceChars (${serializedChars} > ${maxChars})`);
3838
- return deepFreeze({
3883
+ return deepFreeze$1({
3839
3884
  id: scenario.id,
3840
3885
  data
3841
3886
  });
@@ -3886,9 +3931,9 @@ async function scoreOneScenario(args) {
3886
3931
  function cloneExternalTextCandidate$1(candidate) {
3887
3932
  return typeof candidate === "string" ? candidate : { ...candidate };
3888
3933
  }
3889
- function deepFreeze(value) {
3934
+ function deepFreeze$1(value) {
3890
3935
  if (value && typeof value === "object") {
3891
- for (const child of Object.values(value)) deepFreeze(child);
3936
+ for (const child of Object.values(value)) deepFreeze$1(child);
3892
3937
  Object.freeze(value);
3893
3938
  }
3894
3939
  return value;
@@ -5251,23 +5296,24 @@ function pairHoldout(candidate, baseline, scenarioIds, select) {
5251
5296
  }
5252
5297
  /** Significance of the held-out composite lift: ship only when the paired
5253
5298
  * bootstrap CI lower bound on (candidate − baseline) exceeds `deltaThreshold`
5254
- * (default 0 ⇒ "confidently positive"). Below `minProductiveRuns` paired
5255
- * observations there is not enough evidence to claim significance not
5256
- * significant (`fewRuns`). Interpret `deltaThreshold` in the judge's native
5257
- * composite scale. */
5299
+ * (default 0 ⇒ "confidently positive"). At small n, where the percentile
5300
+ * bootstrap is descriptive only, a pre-registered exact sign test carries
5301
+ * the decision. Interpret `deltaThreshold` in the judge's native scale. */
5258
5302
  function heldoutSignificance(paired, opts = {}) {
5259
5303
  const deltaThreshold = opts.deltaThreshold ?? 0;
5260
- const minProductiveRuns = opts.minProductiveRuns ?? 3;
5261
5304
  const confidence = opts.confidence ?? .95;
5262
5305
  const resamples = opts.resamples ?? 2e3;
5263
5306
  const seed = opts.seed ?? 1337;
5264
5307
  const statistic = opts.statistic ?? "mean";
5265
- const bootstrap = pairedBootstrap(paired.before, paired.after, {
5308
+ const decision = pairedDeltaTest(paired.before, paired.after, {
5266
5309
  confidence,
5267
5310
  resamples,
5268
5311
  statistic,
5269
- seed
5312
+ seed,
5313
+ threshold: deltaThreshold,
5314
+ minPairs: opts.minProductiveRuns
5270
5315
  });
5316
+ const bootstrap = decision.bootstrap;
5271
5317
  const medianBootstrap = statistic === "median" ? bootstrap : pairedBootstrap(paired.before, paired.after, {
5272
5318
  confidence,
5273
5319
  resamples,
@@ -5282,14 +5328,18 @@ function heldoutSignificance(paired, opts = {}) {
5282
5328
  if (Math.abs(after - before) < 1e-9) ties += 1;
5283
5329
  }
5284
5330
  const tieFraction = n === 0 ? 0 : ties / n;
5285
- const fewRuns = n < minProductiveRuns;
5331
+ const fewRuns = !decision.sufficient;
5332
+ const significant = decision.significant;
5286
5333
  return {
5287
5334
  paired,
5288
5335
  bootstrap,
5289
5336
  medianBootstrap,
5290
5337
  tieFraction,
5291
5338
  n,
5292
- significant: !fewRuns && bootstrap.low > deltaThreshold,
5339
+ minimumRequired: decision.minimumPairs,
5340
+ decisionMethod: decision.method,
5341
+ pValue: decision.pValue,
5342
+ significant,
5293
5343
  fewRuns
5294
5344
  };
5295
5345
  }
@@ -5317,10 +5367,17 @@ function dimensionRegressions(candidate, baseline, scenarioIds, criticalDimensio
5317
5367
  statistic: "median",
5318
5368
  seed: opts.seed ?? 1337
5319
5369
  });
5370
+ const regression = pairedDeltaTest(paired.after, paired.before, {
5371
+ confidence: opts.confidence ?? .95,
5372
+ resamples: opts.resamples ?? 2e3,
5373
+ statistic: "median",
5374
+ seed: opts.seed ?? 1337,
5375
+ threshold: tolerance
5376
+ });
5320
5377
  out.push({
5321
5378
  dimension: dim,
5322
5379
  bootstrap,
5323
- regressed: bootstrap.low < -tolerance,
5380
+ regressed: regression.significant,
5324
5381
  tolerance,
5325
5382
  n: paired.before.length
5326
5383
  });
@@ -5395,7 +5452,7 @@ function defaultProductionGate(options) {
5395
5452
  if (sig.fewRuns) requiredUnavailable.add("heldout-significance");
5396
5453
  if (!heldoutPass) {
5397
5454
  const tieNote = sig.tieFraction >= .4 ? `; ${(sig.tieFraction * 100).toFixed(0)}% tied scenarios` : "";
5398
- reasons.push(sig.fewRuns ? `held-out: only ${sig.n} paired runs (< ${minProductiveRuns}) — too few to claim significance` : `held-out CI.low ${sig.bootstrap.low.toFixed(3)} ≤ threshold ${deltaThreshold} (${heldoutStatistic} Δ ${delta.toFixed(3)}, ${(sig.bootstrap.confidence * 100).toFixed(0)}% CI [${sig.bootstrap.low.toFixed(3)}, ${sig.bootstrap.high.toFixed(3)}]${tieNote})`);
5455
+ reasons.push(sig.fewRuns ? `held-out: only ${sig.n} paired runs (< ${sig.minimumRequired}) — too few to claim significance` : `held-out CI.low ${sig.bootstrap.low.toFixed(3)} ≤ threshold ${deltaThreshold} (${heldoutStatistic} Δ ${delta.toFixed(3)}, ${(sig.bootstrap.confidence * 100).toFixed(0)}% CI [${sig.bootstrap.low.toFixed(3)}, ${sig.bootstrap.high.toFixed(3)}]${tieNote})`);
5399
5456
  }
5400
5457
  }
5401
5458
  const dimensionsProvided = options.criticalDimensions !== void 0;
@@ -5617,7 +5674,7 @@ function heldOutGate(options) {
5617
5674
  const ci = `${(sig.bootstrap.confidence * 100).toFixed(0)}% CI [${sig.bootstrap.low.toFixed(3)}, ${sig.bootstrap.high.toFixed(3)}]`;
5618
5675
  return {
5619
5676
  decision: passed ? "ship" : "hold",
5620
- reasons: passed ? [`held-out mean Δ ${delta.toFixed(3)}, CI.low ${sig.bootstrap.low.toFixed(3)} > ${deltaThreshold} (${ci}, n=${sig.n}${tieNote})`] : [sig.fewRuns ? `held-out: only ${sig.n} paired runs — too few to claim significance` : `held-out mean Δ ${delta.toFixed(3)}, CI.low ${sig.bootstrap.low.toFixed(3)} ≤ ${deltaThreshold} (${ci}, n=${sig.n}${tieNote})`],
5677
+ reasons: passed ? [`held-out mean Δ ${delta.toFixed(3)}, CI.low ${sig.bootstrap.low.toFixed(3)} > ${deltaThreshold} (${ci}, n=${sig.n}${tieNote})`] : [sig.fewRuns ? `held-out: only ${sig.n} paired runs; ${sig.minimumRequired} required — too few to claim significance` : `held-out mean Δ ${delta.toFixed(3)}, CI.low ${sig.bootstrap.low.toFixed(3)} ≤ ${deltaThreshold} (${ci}, n=${sig.n}${tieNote})`],
5621
5678
  contributingGates: [{
5622
5679
  name: "heldOutGate",
5623
5680
  status,
@@ -5686,6 +5743,30 @@ function powerPreflight(opts) {
5686
5743
  //#endregion
5687
5744
  //#region src/campaign/gates/promotion-policy.ts
5688
5745
  /**
5746
+ * Promotion policy over the evidence VECTOR — the substrate's answer to "never
5747
+ * collapse the multi-objective promotion decision into one scalar." A
5748
+ * `defaultProductionGate` is one opinionated composition; this module factors
5749
+ * the decision into two reusable pieces so MANY policies can compete over the
5750
+ * SAME evidence (the quant-desk pattern: one evidence bus, plural strategies):
5751
+ *
5752
+ * buildEvidenceVector(ctx, objectives, opts) -> EvidenceVector // the bus
5753
+ * PromotionPolicy = (ev: EvidenceVector) => GateResult // a strategy
5754
+ * paretoPolicy(ev) // the default strategy
5755
+ * paretoSignificanceGate(options): Gate // bus + policy as a Gate
5756
+ *
5757
+ * The Pareto policy is SYMMETRIC multi-objective: every objective is BOTH a
5758
+ * potential gain source AND a safety floor (unlike `defaultProductionGate`,
5759
+ * where only `composite` can win and `criticalDimensions` are pure floors). A
5760
+ * candidate ships iff it weakly DOMINATES the baseline at the confidence level —
5761
+ * no objective credibly worse (CI floor breach) AND at least one objective
5762
+ * credibly better (CI gain). Insufficient evidence on ANY axis -> need_more_work
5763
+ * (NOT folded into hold: "gather more reps" and "reject" are different actions).
5764
+ *
5765
+ * Cost/latency are NOT CI axes here — `GateContext` carries only an aggregate
5766
+ * per-side cost, no per-cell observation vector to bootstrap. Treat them as hard
5767
+ * constraints (compose with a budget gate via `composeGate`), not faked CIs.
5768
+ */
5769
+ /**
5689
5770
  * The Evidence Bus. For each objective, pair candidate vs baseline by full
5690
5771
  * cellId and bootstrap a CI on the good-direction paired delta. Reuses the
5691
5772
  * exact `pairHoldout` + `pairedBootstrap` machinery the held-out gate uses, so
@@ -5693,7 +5774,6 @@ function powerPreflight(opts) {
5693
5774
  */
5694
5775
  function buildEvidenceVector(ctx, objectives, opts = {}) {
5695
5776
  if (objectives.length === 0) throw new Error("buildEvidenceVector: at least 1 objective required");
5696
- const minProductiveRuns = opts.minProductiveRuns ?? 3;
5697
5777
  const confidence = opts.confidence ?? .95;
5698
5778
  const resamples = opts.resamples ?? 2e3;
5699
5779
  const seed = opts.seed ?? 1337;
@@ -5708,22 +5788,37 @@ function buildEvidenceVector(ctx, objectives, opts = {}) {
5708
5788
  select = (s) => s.dimensions[dim];
5709
5789
  }
5710
5790
  const paired = pairHoldout(ctx.judgeScores, baseline, scenarioIds, select);
5711
- const bootstrap = pairedBootstrap(obj.direction === "maximize" ? paired.before : paired.after, obj.direction === "maximize" ? paired.after : paired.before, {
5791
+ const before = obj.direction === "maximize" ? paired.before : paired.after;
5792
+ const after = obj.direction === "maximize" ? paired.after : paired.before;
5793
+ const n = paired.before.length;
5794
+ const floorTolerance = obj.floorTolerance ?? .05 * detectScale([...paired.before, ...paired.after]);
5795
+ const gainThreshold = obj.gainThreshold ?? 0;
5796
+ const improvement = pairedDeltaTest(before, after, {
5712
5797
  confidence,
5713
5798
  resamples,
5714
5799
  statistic: "median",
5715
- seed
5800
+ seed,
5801
+ threshold: gainThreshold,
5802
+ minPairs: opts.minProductiveRuns
5716
5803
  });
5717
- const n = paired.before.length;
5718
- const floorTolerance = obj.floorTolerance ?? .05 * detectScale([...paired.before, ...paired.after]);
5719
- const gainThreshold = obj.gainThreshold ?? 0;
5720
- const verdict = n < minProductiveRuns ? "few_runs" : bootstrap.low < -floorTolerance ? "regressed" : bootstrap.low > gainThreshold ? "improved" : "flat";
5804
+ const regression = pairedDeltaTest(after, before, {
5805
+ confidence,
5806
+ resamples,
5807
+ statistic: "median",
5808
+ seed,
5809
+ threshold: floorTolerance,
5810
+ minPairs: opts.minProductiveRuns
5811
+ });
5812
+ const bootstrap = improvement.bootstrap;
5813
+ const verdict = !improvement.sufficient ? "few_runs" : regression.significant ? "regressed" : improvement.significant ? "improved" : "flat";
5721
5814
  axes.push({
5722
5815
  name: obj.name,
5723
5816
  source: obj.source,
5724
5817
  direction: obj.direction,
5725
5818
  bootstrap,
5726
5819
  n,
5820
+ minimumRequired: improvement.minimumPairs,
5821
+ decisionMethod: improvement.method,
5727
5822
  gainThreshold,
5728
5823
  floorTolerance,
5729
5824
  verdict
@@ -6413,6 +6508,8 @@ async function runOptimization(opts) {
6413
6508
  const candidateConcurrency = opts.candidateConcurrency ?? 1;
6414
6509
  if (typeof opts.runDir !== "string" || opts.runDir.trim().length === 0) throw new Error("runOptimization: runDir is required and must be a non-empty string");
6415
6510
  if (!Number.isInteger(candidateConcurrency) || candidateConcurrency < 1) throw new Error("runOptimization: candidateConcurrency must be a positive integer");
6511
+ const initialFindings = immutableProposalSnapshot(assertProposalFindings(opts.findings ?? [], "runOptimization initial proposal findings"), "initial findings");
6512
+ const baselineSurface = immutableProposalSnapshot(opts.baselineSurface, "baseline surface");
6416
6513
  opts.runDir = resolveRunDir(opts.runDir, opts.repo);
6417
6514
  const storage = opts.storage ?? fsCampaignStorage();
6418
6515
  const costLedger = opts.costLedger ?? createRunCostLedger({
@@ -6425,7 +6522,7 @@ async function runOptimization(opts) {
6425
6522
  const premeasuredBaseline = opts.premeasuredBaseline;
6426
6523
  const baselineCampaign = premeasuredBaseline ? validatedPremeasuredBaseline({
6427
6524
  input: premeasuredBaseline,
6428
- baselineSurface: opts.baselineSurface,
6525
+ baselineSurface,
6429
6526
  scenarios: opts.scenarios,
6430
6527
  reps,
6431
6528
  seed: opts.seed ?? 42
@@ -6433,7 +6530,7 @@ async function runOptimization(opts) {
6433
6530
  ...opts,
6434
6531
  costLedger,
6435
6532
  costPhase: "search.baseline",
6436
- dispatch: (scenario, ctx) => opts.dispatchWithSurface(opts.baselineSurface, scenario, ctx),
6533
+ dispatch: (scenario, ctx) => opts.dispatchWithSurface(baselineSurface, scenario, ctx),
6437
6534
  runDir: `${opts.runDir}/baseline`
6438
6535
  });
6439
6536
  const baselineCoverage = campaignCoverage(baselineCampaign.cells, opts.scenarios, reps, requireJudgeScore);
@@ -6443,10 +6540,10 @@ async function runOptimization(opts) {
6443
6540
  }
6444
6541
  const generations = [];
6445
6542
  const history = [];
6446
- let currentFindings = opts.findings ?? [];
6543
+ let currentFindings = initialFindings;
6447
6544
  const selectionRankKey = opts.selectionRankKey ?? ((campaign) => [campaignMeanComposite(campaign)]);
6448
- let winnerSurface = opts.baselineSurface;
6449
- let winnerSurfaceHash = surfaceHash(opts.baselineSurface);
6545
+ let winnerSurface = baselineSurface;
6546
+ let winnerSurfaceHash = surfaceHash(baselineSurface);
6450
6547
  let winnerComposite = campaignMeanComposite(baselineCampaign);
6451
6548
  let winnerRankKey = selectionRankKey(baselineCampaign);
6452
6549
  assertFiniteRankKey(winnerRankKey, "selectionRankKey for baseline");
@@ -6454,7 +6551,7 @@ async function runOptimization(opts) {
6454
6551
  let winnerOutcome = baselineOutcome;
6455
6552
  let winnerLabel;
6456
6553
  let winnerRationale;
6457
- const scored = [toParetoParent(opts.baselineSurface, winnerSurfaceHash, baselineCampaign, -1)];
6554
+ const scored = [toParetoParent(baselineSurface, winnerSurfaceHash, baselineCampaign, -1)];
6458
6555
  if (opts.analyzeGeneration && opts.maxGenerations > 0 && baselineCampaign.cells.length > 0) {
6459
6556
  const fresh = await opts.analyzeGeneration({
6460
6557
  generation: -1,
@@ -6468,31 +6565,34 @@ async function runOptimization(opts) {
6468
6565
  costLedger,
6469
6566
  costPhase: "analysis.baseline"
6470
6567
  });
6471
- if (Array.isArray(fresh)) currentFindings = fresh;
6568
+ if (!Array.isArray(fresh)) throw new TypeError("runOptimization: analyzeGeneration must return an array");
6569
+ currentFindings = immutableProposalSnapshot(assertProposalFindings(fresh, "runOptimization baseline analysis findings"), "baseline analysis findings");
6472
6570
  }
6473
6571
  for (let gen = 0; gen < opts.maxGenerations; gen++) {
6474
- if (proposer.decide?.({ history }).stop) break;
6572
+ const proposalHistory = immutableProposalSnapshot(history, "history");
6573
+ if (proposer.decide?.({ history: proposalHistory }).stop) break;
6475
6574
  const paretoParents = computeParetoFrontier(scored);
6476
6575
  const parentSurfaceHash = winnerSurfaceHash;
6477
6576
  const parentComposite = winnerComposite;
6478
- const proposed = await proposer.propose({
6479
- currentSurface: winnerSurface,
6480
- history,
6481
- findings: currentFindings,
6577
+ const proposalContext = Object.freeze({
6578
+ currentSurface: immutableProposalSnapshot(winnerSurface, "current surface"),
6579
+ history: proposalHistory,
6580
+ findings: immutableProposalSnapshot(assertProposalFindings(currentFindings, "runOptimization proposal findings"), "findings"),
6482
6581
  populationSize: opts.populationSize,
6483
6582
  generation: gen,
6484
- signal: new AbortController().signal,
6485
- baselineOutcome,
6486
- incumbentOutcome: winnerOutcome,
6487
- report: opts.report,
6488
- dataset: opts.labeledStore && opts.labeledStore !== "off" ? opts.labeledStore : void 0,
6583
+ signal: opts.signal ?? new AbortController().signal,
6584
+ baselineOutcome: immutableProposalSnapshot(baselineOutcome, "baseline outcome"),
6585
+ incumbentOutcome: immutableProposalSnapshot(winnerOutcome, "incumbent outcome"),
6489
6586
  maxImprovementShots: opts.maxImprovementShots,
6490
- paretoParents,
6587
+ paretoParents: immutableProposalSnapshot(paretoParents, "Pareto parents"),
6491
6588
  costLedger,
6492
6589
  costPhase: "search.proposal"
6493
6590
  });
6494
- if (proposed.length === 0) break;
6495
- const surfaceResults = await mapConcurrent(proposed.map((p) => isProposedCandidate(p) ? p : {
6591
+ const proposed = await proposer.propose(proposalContext);
6592
+ if (!Array.isArray(proposed)) throw new TypeError("runOptimization: proposer must return an array");
6593
+ const proposalSnapshot = immutableProposalSnapshot(proposed, "candidate outputs");
6594
+ if (proposalSnapshot.length === 0) break;
6595
+ const surfaceResults = await mapConcurrent(proposalSnapshot.map((p) => isProposedCandidate(p) ? p : {
6496
6596
  surface: p,
6497
6597
  label: "",
6498
6598
  rationale: ""
@@ -6589,11 +6689,13 @@ async function runOptimization(opts) {
6589
6689
  costLedger,
6590
6690
  costPhase: "analysis.generation"
6591
6691
  });
6592
- if (Array.isArray(fresh)) currentFindings = fresh;
6692
+ if (!Array.isArray(fresh)) throw new TypeError("runOptimization: analyzeGeneration must return an array");
6693
+ currentFindings = immutableProposalSnapshot(assertProposalFindings(fresh, "runOptimization generation analysis findings"), "generation analysis findings");
6593
6694
  }
6594
6695
  }
6595
6696
  return {
6596
6697
  generations,
6698
+ baselineSurface,
6597
6699
  winnerSurface,
6598
6700
  winnerSurfaceHash,
6599
6701
  winnerLabel,
@@ -6603,6 +6705,19 @@ async function runOptimization(opts) {
6603
6705
  cost: costLedger.summary()
6604
6706
  };
6605
6707
  }
6708
+ function immutableProposalSnapshot(value, label) {
6709
+ try {
6710
+ return deepFreeze(structuredClone(value));
6711
+ } catch (cause) {
6712
+ throw new TypeError(`runOptimization: proposal ${label} must contain snapshot-safe data`, { cause });
6713
+ }
6714
+ }
6715
+ function deepFreeze(value, seen = /* @__PURE__ */ new WeakSet()) {
6716
+ if (typeof value !== "object" || value === null || seen.has(value)) return value;
6717
+ seen.add(value);
6718
+ for (const descriptor of Object.values(Object.getOwnPropertyDescriptors(value))) if ("value" in descriptor) deepFreeze(descriptor.value, seen);
6719
+ return Object.freeze(value);
6720
+ }
6606
6721
  function validatedPremeasuredBaseline(args) {
6607
6722
  const { input } = args;
6608
6723
  if (input.surfaceHash !== surfaceHash(args.baselineSurface)) throw new Error("runOptimization: premeasured baseline surface hash does not match baselineSurface");
@@ -6712,7 +6827,8 @@ async function runImprovementLoop(opts) {
6712
6827
  dispatchTimeoutMs,
6713
6828
  costLedger
6714
6829
  });
6715
- const winnerIsBaseline = optimization.winnerSurfaceHash === surfaceHash(opts.baselineSurface);
6830
+ const baselineSurface = optimization.baselineSurface;
6831
+ const winnerIsBaseline = optimization.winnerSurfaceHash === surfaceHash(baselineSurface);
6716
6832
  const holdoutDeferred = (opts.holdout ?? "measured") === "deferred";
6717
6833
  const baselineOnHoldout = holdoutDeferred ? await runCampaign({
6718
6834
  ...opts,
@@ -6732,7 +6848,7 @@ async function runImprovementLoop(opts) {
6732
6848
  costPhase: "holdout.baseline",
6733
6849
  dispatchTimeoutMs,
6734
6850
  scenarios: opts.holdoutScenarios,
6735
- dispatch: (scenario, ctx) => opts.dispatchWithSurface(opts.baselineSurface, scenario, ctx),
6851
+ dispatch: (scenario, ctx) => opts.dispatchWithSurface(baselineSurface, scenario, ctx),
6736
6852
  runDir: `${opts.runDir}/holdout-baseline`
6737
6853
  });
6738
6854
  const winnerOnHoldout = winnerIsBaseline || holdoutDeferred ? baselineOnHoldout : await runCampaign({
@@ -6772,7 +6888,7 @@ async function runImprovementLoop(opts) {
6772
6888
  let neutralizedOnHoldout;
6773
6889
  let neutralizedSurface;
6774
6890
  if (opts.neutralize && !winnerIsBaseline && !holdoutDeferred) {
6775
- const surface = opts.neutralize(optimization.winnerSurface, opts.baselineSurface);
6891
+ const surface = opts.neutralize(optimization.winnerSurface, baselineSurface);
6776
6892
  neutralizedSurface = surface;
6777
6893
  neutralizedOnHoldout = await runCampaign({
6778
6894
  ...opts,
@@ -6825,7 +6941,7 @@ async function runImprovementLoop(opts) {
6825
6941
  costPhase: "promotion.gate",
6826
6942
  signal: new AbortController().signal
6827
6943
  });
6828
- const promotedDiff = optimization.winnerSurfaceHash === surfaceHash(opts.baselineSurface) ? "" : renderSurfaceDiff(optimization.winnerSurface, opts.baselineSurface);
6944
+ const promotedDiff = optimization.winnerSurfaceHash === surfaceHash(baselineSurface) ? "" : renderSurfaceDiff(optimization.winnerSurface, baselineSurface);
6829
6945
  let prResult;
6830
6946
  if (opts.autoOnPromote === "pr" && gateResult.decision === "ship") prResult = openAutoPr({
6831
6947
  result: winnerOnHoldout,
@@ -7717,6 +7833,6 @@ function skillOptOptimizationMethod(config) {
7717
7833
  };
7718
7834
  }
7719
7835
  //#endregion
7720
- export { SEARCH_LEDGER_FILE_CONTEXT as $, acquireSingleRunLock as A, dominates as At, surfaceContentHash as B, JudgeParseError as Bt, detectScale as C, llmJudge as Ct, runCanaries as D, fileVerdictCache as Dt, pairHoldout as E, contentHash as Et, assertCodeSurfaceIdentity as F, BackendIntegrityError as Ft, DEFAULT_MUTATION_PRIMITIVES as G, campaignBreakdown as H, assertComponentSurface as I, assertRealAgentReceipts as It, planCampaignRun as J, buildReflectionPrompt as K, codeSurfaceIdentityMaterial as L, assertRealBackend as Lt, compareOptimizationMethods as M, paretoFrontierWithCrowding as Mt, costFromLedgerSummary as N, scalarScore as Nt, composeGate as O, inMemoryVerdictCache as Ot, optimizationTokenUsageFromSummary as P, recoverTruncatedJson as Pt, inMemoryCampaignStorage as Q, componentSurfaceIdentityMaterial as R, summarizeAgentReceiptIntegrity as Rt, defaultProductionGate as S, hashScenarios as St, heldoutSignificance as T, canonicalJson as Tt, campaignMeanComposite as U, surfaceHash as V, compareRankKeys as W, createRunCostLedger as X, runCampaign as Y, fsCampaignStorage as Z, buildEvidenceVector as _, redTeamReport as _t, emitLoopProvenance as a, assertCampaignDesign as at, powerPreflight as b, Dataset as bt, provenanceRecordPath as c, campaignSplitDigest as ct, runImprovementLoop as d, REFERENCE_EQUIVALENCE_INPUT_LIMITS as dt, SearchLedgerConflictError as et, runOptimization as f, REFERENCE_EQUIVALENCE_JUDGE_VERSION as ft, gepaOptimizationMethod as g, redTeamDataset as gt, labelTrustRank as h, DEFAULT_RED_TEAM_CORPUS as ht, canonicalDigest as i, tangleTracesRoot as it, assertOptimizationResult as j, paretoFrontier as jt, externalTextOptimizationMethod as k, crowdingDistance as kt, provenanceSpansPath as l, campaignSplitDigestFromIdentities as lt, isProposedCandidate as m, runReferenceEquivalenceJudge as mt, buildLoopProvenanceRecord as n, SearchLedgerIntegrityError as nt, loopProvenanceArgsFromResult as o, assertCampaignSplitIdentity as ot, runEval as p, createReferenceEquivalenceJudge as pt, parseReflectionResponse as q, campaignMeasurementDigest as r, resolveRunDir as rt, loopProvenanceSpans as s, campaignScenarioIdentity as st, skillOptOptimizationMethod as t, SearchLedgerError as tt, verifyLoopProvenanceRecord as u, openAutoPr as ut, paretoPolicy as v, scoreRedTeamOutput as vt, dimensionRegressions as w, cachedJudge as wt, heldOutGate as x, HoldoutLockedError as xt, paretoSignificanceGate as y, toolNamesForRun as yt, renderSurfaceDiff as z, summarizeBackendIntegrity as zt };
7836
+ export { SEARCH_LEDGER_FILE_CONTEXT as $, acquireSingleRunLock as A, dominates as At, surfaceContentHash as B, summarizeAgentReceiptIntegrity as Bt, detectScale as C, llmJudge as Ct, runCanaries as D, fileVerdictCache as Dt, pairHoldout as E, contentHash as Et, assertCodeSurfaceIdentity as F, minimumPairsForPairedDeltaTest as Ft, DEFAULT_MUTATION_PRIMITIVES as G, campaignBreakdown as H, JudgeParseError as Ht, assertComponentSurface as I, pairedDeltaTest as It, planCampaignRun as J, buildReflectionPrompt as K, codeSurfaceIdentityMaterial as L, BackendIntegrityError as Lt, compareOptimizationMethods as M, paretoFrontierWithCrowding as Mt, costFromLedgerSummary as N, scalarScore as Nt, composeGate as O, inMemoryVerdictCache as Ot, optimizationTokenUsageFromSummary as P, recoverTruncatedJson as Pt, inMemoryCampaignStorage as Q, componentSurfaceIdentityMaterial as R, assertRealAgentReceipts as Rt, defaultProductionGate as S, hashScenarios as St, heldoutSignificance as T, canonicalJson as Tt, campaignMeanComposite as U, surfaceHash as V, summarizeBackendIntegrity as Vt, compareRankKeys as W, createRunCostLedger as X, runCampaign as Y, fsCampaignStorage as Z, buildEvidenceVector as _, redTeamReport as _t, emitLoopProvenance as a, assertCampaignDesign as at, powerPreflight as b, Dataset as bt, provenanceRecordPath as c, campaignSplitDigest as ct, runImprovementLoop as d, REFERENCE_EQUIVALENCE_INPUT_LIMITS as dt, SearchLedgerConflictError as et, runOptimization as f, REFERENCE_EQUIVALENCE_JUDGE_VERSION as ft, gepaOptimizationMethod as g, redTeamDataset as gt, labelTrustRank as h, DEFAULT_RED_TEAM_CORPUS as ht, canonicalDigest as i, tangleTracesRoot as it, assertOptimizationResult as j, paretoFrontier as jt, externalTextOptimizationMethod as k, crowdingDistance as kt, provenanceSpansPath as l, campaignSplitDigestFromIdentities as lt, isProposedCandidate as m, runReferenceEquivalenceJudge as mt, buildLoopProvenanceRecord as n, SearchLedgerIntegrityError as nt, loopProvenanceArgsFromResult as o, assertCampaignSplitIdentity as ot, runEval as p, createReferenceEquivalenceJudge as pt, parseReflectionResponse as q, campaignMeasurementDigest as r, resolveRunDir as rt, loopProvenanceSpans as s, campaignScenarioIdentity as st, skillOptOptimizationMethod as t, SearchLedgerError as tt, verifyLoopProvenanceRecord as u, openAutoPr as ut, paretoPolicy as v, scoreRedTeamOutput as vt, dimensionRegressions as w, cachedJudge as wt, heldOutGate as x, HoldoutLockedError as xt, paretoSignificanceGate as y, toolNamesForRun as yt, renderSurfaceDiff as z, assertRealBackend as zt };
7721
7837
 
7722
- //# sourceMappingURL=skillopt-optimization-method-CF6a327Q.js.map
7838
+ //# sourceMappingURL=skillopt-optimization-method-BY6vKLJB.js.map