@tangle-network/agent-eval 0.135.0 → 0.135.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +24 -0
- package/dist/analyst/index.js +3 -3
- package/dist/{analyze-runs-DMo3Lb_y.d.ts → analyze-runs-Cda5Xkj1.d.ts} +3 -3
- package/dist/{analyze-runs-DMo3Lb_y.d.ts.map → analyze-runs-Cda5Xkj1.d.ts.map} +1 -1
- package/dist/{analyze-runs-qk8op0tN.js → analyze-runs-jjCmF8pU.js} +14 -7
- package/dist/analyze-runs-jjCmF8pU.js.map +1 -0
- package/dist/{baseline-BaPxoROc.js → baseline-BUeFcgrn.js} +2 -2
- package/dist/{baseline-BaPxoROc.js.map → baseline-BUeFcgrn.js.map} +1 -1
- package/dist/benchmarks/index.d.ts +1 -1
- package/dist/benchmarks/index.js +1 -1
- package/dist/{benchmarks-Dw1Wv_JQ.js → benchmarks-Mtu251Jz.js} +3 -3
- package/dist/{benchmarks-Dw1Wv_JQ.js.map → benchmarks-Mtu251Jz.js.map} +1 -1
- package/dist/builder-eval/index.js +1 -1
- package/dist/campaign/index.d.ts +2 -2
- package/dist/campaign/index.js +2 -2
- package/dist/{campaign-B1c1T0kv.js → campaign-RVIqtJh0.js} +7 -7
- package/dist/{campaign-B1c1T0kv.js.map → campaign-RVIqtJh0.js.map} +1 -1
- package/dist/cli.js +1 -1
- package/dist/{client-BIyh1RCr.d.ts → client-DcvgkaZi.d.ts} +13 -4
- package/dist/client-DcvgkaZi.d.ts.map +1 -0
- package/dist/contract/index.d.ts +3 -3
- package/dist/contract/index.d.ts.map +1 -1
- package/dist/contract/index.js +19 -7
- package/dist/contract/index.js.map +1 -1
- package/dist/{cost-ledger-ZAa_P4r0.js → cost-ledger-DHAjwNj7.js} +6 -2
- package/dist/{cost-ledger-ZAa_P4r0.js.map → cost-ledger-DHAjwNj7.js.map} +1 -1
- package/dist/{default-registry-CFUZyNeZ.js → default-registry-BAhV-lbE.js} +3 -3
- package/dist/{default-registry-CFUZyNeZ.js.map → default-registry-BAhV-lbE.js.map} +1 -1
- package/dist/{eval-campaign-CHFxPTVl.js → eval-campaign-Cc8WZJ6b.js} +3 -3
- package/dist/{eval-campaign-CHFxPTVl.js.map → eval-campaign-Cc8WZJ6b.js.map} +1 -1
- package/dist/fuzz.js +1 -1
- package/dist/hosted/index.d.ts +1 -1
- package/dist/{index-BoJNQR6n.d.ts → index-B4Fjfo5U.d.ts} +93 -15
- package/dist/index-B4Fjfo5U.d.ts.map +1 -0
- package/dist/{index-C21xKtxu.d.ts → index-CQsJcqch.d.ts} +3 -3
- package/dist/{index-C21xKtxu.d.ts.map → index-CQsJcqch.d.ts.map} +1 -1
- package/dist/{index-DSC51roc2.d.ts → index-DSC51roc.d.ts} +1 -1
- package/dist/index-DSC51roc.d.ts.map +1 -0
- package/dist/index.d.ts +41 -10
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +135 -57
- package/dist/index.js.map +1 -1
- package/dist/{llm-client-BNcP4v08.js → llm-client-DHx8pzyJ.js} +2 -2
- package/dist/{llm-client-BNcP4v08.js.map → llm-client-DHx8pzyJ.js.map} +1 -1
- package/dist/matrix/index.d.ts +1 -1
- package/dist/meta-eval/index.d.ts +1 -2
- package/dist/meta-eval/index.d.ts.map +1 -1
- package/dist/meta-eval/index.js +2 -2
- package/dist/multishot/index.d.ts +1 -1
- package/dist/openapi.json +1 -1
- package/dist/{paired-arms-CA_8pN01.js → paired-arms-BbFKrAU-.js} +2 -2
- package/dist/{paired-arms-CA_8pN01.js.map → paired-arms-BbFKrAU-.js.map} +1 -1
- package/dist/pipelines/index.js +2 -2
- package/dist/{release-report-BVZBmRZp.js → release-report-DooPguBc.js} +4 -3
- package/dist/{release-report-BVZBmRZp.js.map → release-report-DooPguBc.js.map} +1 -1
- package/dist/{release-report-CuULWKyk.d.ts → release-report-DpBxGGI1.d.ts} +2 -2
- package/dist/{release-report-CuULWKyk.d.ts.map → release-report-DpBxGGI1.d.ts.map} +1 -1
- package/dist/reporting.d.ts +3 -3
- package/dist/reporting.js +4 -4
- package/dist/{researcher-DVtruQ9U.d.ts → researcher-Doo95b50.d.ts} +2 -2
- package/dist/{researcher-DVtruQ9U.d.ts.map → researcher-Doo95b50.d.ts.map} +1 -1
- package/dist/{reward-hacking-DCdRK9TY.js → reward-hacking-a-kYs0-i.js} +2 -2
- package/dist/{reward-hacking-DCdRK9TY.js.map → reward-hacking-a-kYs0-i.js.map} +1 -1
- package/dist/rl.d.ts +45 -3
- package/dist/rl.d.ts.map +1 -1
- package/dist/rl.js +108 -22
- package/dist/rl.js.map +1 -1
- package/dist/{rubric-predictive-validity-D6Q6n9oq.js → rubric-predictive-validity-BJf-8ejY.js} +2 -2
- package/dist/{rubric-predictive-validity-D6Q6n9oq.js.map → rubric-predictive-validity-BJf-8ejY.js.map} +1 -1
- package/dist/{semantic-concept-judge-Ca--u10C.js → semantic-concept-judge-Btozx3Vc.js} +3 -3
- package/dist/{semantic-concept-judge-Ca--u10C.js.map → semantic-concept-judge-Btozx3Vc.js.map} +1 -1
- package/dist/{server-Dc_lsOYd.js → server-Bz3WQJs6.js} +3 -3
- package/dist/{server-Dc_lsOYd.js.map → server-Bz3WQJs6.js.map} +1 -1
- package/dist/{skillopt-optimization-method-Cl4XPkLC.js → skillopt-optimization-method-0UmPD6aP.js} +342 -56
- package/dist/skillopt-optimization-method-0UmPD6aP.js.map +1 -0
- package/dist/{skillopt-optimization-method-DJ3l4w8W.d.ts → skillopt-optimization-method-CwSYkv35.d.ts} +39 -9
- package/dist/skillopt-optimization-method-CwSYkv35.d.ts.map +1 -0
- package/dist/{statistics-D_4Snl-5.d.ts → statistics-CKOqre5S.d.ts} +329 -3
- package/dist/statistics-CKOqre5S.d.ts.map +1 -0
- package/dist/{statistics-RwRNu2__.js → statistics-CnGCLLqc.js} +315 -2
- package/dist/statistics-CnGCLLqc.js.map +1 -0
- package/dist/{summary-report-BxtossFi.js → summary-report-BEk8OFLs.js} +11 -6
- package/dist/summary-report-BEk8OFLs.js.map +1 -0
- package/dist/{summary-report-DGp0-_XO.d.ts → summary-report-CPMINBqs.d.ts} +182 -7
- package/dist/summary-report-CPMINBqs.d.ts.map +1 -0
- package/dist/wire/index.js +1 -1
- package/package.json +1 -1
- package/dist/analyze-runs-qk8op0tN.js.map +0 -1
- package/dist/client-BIyh1RCr.d.ts.map +0 -1
- package/dist/index-BoJNQR6n.d.ts.map +0 -1
- package/dist/index-DSC51roc2.d.ts.map +0 -1
- package/dist/judge-calibration-DFtEMlde.d.ts +0 -146
- package/dist/judge-calibration-DFtEMlde.d.ts.map +0 -1
- package/dist/skillopt-optimization-method-Cl4XPkLC.js.map +0 -1
- package/dist/skillopt-optimization-method-DJ3l4w8W.d.ts.map +0 -1
- package/dist/statistics-D_4Snl-5.d.ts.map +0 -1
- package/dist/statistics-RwRNu2__.js.map +0 -1
- package/dist/summary-report-BxtossFi.js.map +0 -1
- package/dist/summary-report-DGp0-_XO.d.ts.map +0 -1
package/dist/{skillopt-optimization-method-Cl4XPkLC.js → skillopt-optimization-method-0UmPD6aP.js}
RENAMED
|
@@ -1,11 +1,11 @@
|
|
|
1
1
|
import { i as JudgeError, s as ValidationError, t as AgentEvalError } from "./errors-8YnH8WlF.js";
|
|
2
|
-
import { c as costForTokenPricing, i as CostLedger, t as CostAccountingIncompleteError } from "./cost-ledger-
|
|
3
|
-
import { f as maximumChargeForLlmRequest, l as costReceiptFromLlm, m as stripFencedJson, u as costReceiptFromLlmError } from "./llm-client-
|
|
2
|
+
import { c as costForTokenPricing, i as CostLedger, t as CostAccountingIncompleteError } from "./cost-ledger-DHAjwNj7.js";
|
|
3
|
+
import { f as maximumChargeForLlmRequest, l as costReceiptFromLlm, m as stripFencedJson, u as costReceiptFromLlmError } from "./llm-client-DHx8pzyJ.js";
|
|
4
4
|
import { a as clamp01, o as combineAbortSignals, t as assertProposalFindings } from "./proposal-findings-DCawte-y.js";
|
|
5
5
|
import { n as mapConcurrent } from "./concurrency-MUjT7VjM.js";
|
|
6
|
-
import {
|
|
6
|
+
import { E as pairedBootstrap, H as weightedComposite, M as pairedRiskDifferenceScore, N as pairedSignTest, O as pairedDeltaTieFraction, T as pairedBinaryScale, d as confidenceInterval, j as pairedRiskDifferenceExact } from "./statistics-CnGCLLqc.js";
|
|
7
7
|
import { t as DEFAULT_REDACTION_RULES } from "./redact-7Aq1ukl-.js";
|
|
8
|
-
import { a as campaignCellExecutionEvidence, l as projectCampaignCellQuality, t as detectRewardHacking } from "./reward-hacking-
|
|
8
|
+
import { a as campaignCellExecutionEvidence, l as projectCampaignCellQuality, t as detectRewardHacking } from "./reward-hacking-a-kYs0-i.js";
|
|
9
9
|
import { a as appendLedgerLine, h as tryAcquireAtomicFileLock, m as probeAtomicFileLock, o as tryWithLedgerFileLock } from "./ledger-core-DAKFKRzi.js";
|
|
10
10
|
import { createRequire } from "node:module";
|
|
11
11
|
import { z } from "zod";
|
|
@@ -170,6 +170,32 @@ function minimumPairsForPairedDeltaTest(confidence = .95) {
|
|
|
170
170
|
* switches to a pre-registered one-sided exact sign test. The exact path is
|
|
171
171
|
* deliberately conservative: it requires both a point estimate above the
|
|
172
172
|
* threshold and enough consistently positive paired differences.
|
|
173
|
+
*
|
|
174
|
+
* ## A zero-width interval is never significant
|
|
175
|
+
*
|
|
176
|
+
* When every paired delta is identical the resample distribution is a point
|
|
177
|
+
* mass and the interval collapses: `[0, 0]` when all pairs tie, `[g, g]` on n
|
|
178
|
+
* identical deltas of g. Neither says the effect is certain — both say the
|
|
179
|
+
* sample carries no information about how far the estimate could be wrong, and
|
|
180
|
+
* `low > threshold` then answers on the point estimate alone. It fails in both
|
|
181
|
+
* directions: `[0, 0]` clears every NEGATIVE threshold, which is how a
|
|
182
|
+
* tie-dominated pass/fail comparison laundered a regression into a
|
|
183
|
+
* noninferiority pass, and `[g, g]` clears every threshold below g with no
|
|
184
|
+
* spread behind it. Under a bounded asymmetric null whose true mean paired
|
|
185
|
+
* delta is exactly 0 — 2 % of pairs dropping by 1.0, the rest gaining 0.0204 —
|
|
186
|
+
* every sample that misses the drop is exactly that shape, and deciding on
|
|
187
|
+
* `low > 0` promoted 65.65 % of samples at n = 20 against a nominal 5 %.
|
|
188
|
+
*
|
|
189
|
+
* So `indeterminate` is reported and `significant` is false whenever the
|
|
190
|
+
* interval has zero width, on BOTH paths: at small n the exact sign test is a
|
|
191
|
+
* test of the MEDIAN and a zero-spread sample is precisely where it stops
|
|
192
|
+
* saying anything about the mean the caller is thresholding.
|
|
193
|
+
*
|
|
194
|
+
* `threshold` may be negative — that is a noninferiority margin, and it is the
|
|
195
|
+
* regime the zero-width hole is worst in. For a two-point (pass/fail) outcome
|
|
196
|
+
* the percentile bootstrap is not a valid interval at a nonzero margin at all;
|
|
197
|
+
* use {@link decidePairedPromotion}, which routes those to Tango's score
|
|
198
|
+
* interval, rather than thresholding this function's bootstrap directly.
|
|
173
199
|
*/
|
|
174
200
|
function pairedDeltaTest(before, after, options = {}) {
|
|
175
201
|
const threshold = options.threshold ?? 0;
|
|
@@ -180,13 +206,15 @@ function pairedDeltaTest(before, after, options = {}) {
|
|
|
180
206
|
const minimumPairs = Math.max(requestedMinimum, exactMinimum);
|
|
181
207
|
const bootstrap = pairedBootstrap(before, after, options);
|
|
182
208
|
const sufficient = bootstrap.n >= minimumPairs;
|
|
209
|
+
const indeterminate = bootstrap.n > 0 && (!Number.isFinite(bootstrap.low) || !Number.isFinite(bootstrap.high) || bootstrap.low === bootstrap.high);
|
|
183
210
|
if (bootstrap.gateEligible) return {
|
|
184
211
|
bootstrap,
|
|
185
212
|
method: "bootstrap-ci",
|
|
186
213
|
pValue: null,
|
|
187
214
|
minimumPairs,
|
|
188
215
|
sufficient,
|
|
189
|
-
|
|
216
|
+
indeterminate,
|
|
217
|
+
significant: sufficient && !indeterminate && bootstrap.low > threshold
|
|
190
218
|
};
|
|
191
219
|
const exact = pairedSignTest(before.map((value, index) => after[index] - value - threshold), "greater");
|
|
192
220
|
const estimate = options.statistic === "mean" ? bootstrap.mean : bootstrap.median;
|
|
@@ -196,10 +224,174 @@ function pairedDeltaTest(before, after, options = {}) {
|
|
|
196
224
|
pValue: exact.pValue,
|
|
197
225
|
minimumPairs,
|
|
198
226
|
sufficient,
|
|
199
|
-
|
|
227
|
+
indeterminate,
|
|
228
|
+
significant: sufficient && !indeterminate && estimate > threshold && exact.pValue <= (1 - bootstrap.confidence) / 2
|
|
200
229
|
};
|
|
201
230
|
}
|
|
202
231
|
//#endregion
|
|
232
|
+
//#region src/paired-promotion-decision.ts
|
|
233
|
+
/**
|
|
234
|
+
* @module
|
|
235
|
+
* ONE rule for "does this paired interval clear a promotion threshold".
|
|
236
|
+
*
|
|
237
|
+
* The rule below was derived on `HeldOutGate` (#479) after the same estimator
|
|
238
|
+
* bug shipped twice. It then turned out that a SECOND gate — the composable
|
|
239
|
+
* `heldOutGate`, plus everything else routed through `heldoutSignificance` —
|
|
240
|
+
* still carried the original defect, because the rule had been written into one
|
|
241
|
+
* gate's method body rather than into a shared function. Two copies of a
|
|
242
|
+
* statistical rule is how a defect survives in one of them, so there is now
|
|
243
|
+
* exactly one copy and both gates call it.
|
|
244
|
+
*
|
|
245
|
+
* Three things the rule does that a bare `pairedBootstrap(...).low > threshold`
|
|
246
|
+
* does not:
|
|
247
|
+
*
|
|
248
|
+
* 1. **Two-point (pass/fail) outcomes decide on Tango's SCORE interval.** On a
|
|
249
|
+
* pass/fail eval the paired delta vector is dominated by ties, so the
|
|
250
|
+
* bootstrap of the mean is a resample of a lattice with three atoms and its
|
|
251
|
+
* percentile interval is not valid at a nonzero margin. The score interval
|
|
252
|
+
* (`pairedRiskDifferenceScore`) re-estimates the nuisance loss rate under
|
|
253
|
+
* each hypothesised margin instead of fixing it at the observed value, which
|
|
254
|
+
* is the only construction that stays a confidence interval as the margin
|
|
255
|
+
* moves off zero — the regime every noninferiority threshold lives in.
|
|
256
|
+
* Measured on the composable gate before this change, at a true risk
|
|
257
|
+
* difference sitting exactly on the production caller's -0.05 margin and a
|
|
258
|
+
* nominal 5 %: 14.60 % false promotion at n = 40 and 10.10 % at n = 76.
|
|
259
|
+
* 2. **McNemar's exact test holds a VETO at every non-negative threshold.**
|
|
260
|
+
* Redundant with the interval by construction and kept anyway, so that
|
|
261
|
+
* swapping the estimator for one without that duality cannot silently
|
|
262
|
+
* reintroduce "promotes what the exact test refuses". Witness: n = 6, b = 5,
|
|
263
|
+
* c = 0 — no exact argument reaches alpha = 0.05 with 5 discordant pairs
|
|
264
|
+
* (two-sided floor 2/2^5 = 0.0625), whatever an interval says. A NEGATIVE
|
|
265
|
+
* threshold is a noninferiority question, which McNemar's test of "no
|
|
266
|
+
* difference" is not the right test for, so the veto does not apply there.
|
|
267
|
+
* 3. **A ZERO-WIDTH interval is refused, wherever it sits.** At [0, 0] it
|
|
268
|
+
* cannot tell a gain from a regression and clears every negative threshold.
|
|
269
|
+
* Away from zero it fails the opposite way: n identical positive deltas give
|
|
270
|
+
* [g, g], which clears threshold 0 on no spread at all. Both are an absence
|
|
271
|
+
* of evidence. Measured on the composable gate before this change, under a
|
|
272
|
+
* bounded asymmetric null whose true mean paired delta is exactly 0: 88.50 %
|
|
273
|
+
* false promotion at n = 6 and 65.65 % at n = 20 against a nominal 5 %.
|
|
274
|
+
*
|
|
275
|
+
* Orthogonal to the small-sample switch inside {@link pairedDeltaTest}: that
|
|
276
|
+
* picks the TEST from the sample size (bootstrap CI at n >= 20, pre-registered
|
|
277
|
+
* exact sign test below it), this picks the ESTIMATOR from the outcome's shape.
|
|
278
|
+
* Both are needed — an exact sign test applied to a tie-pinned median is still
|
|
279
|
+
* blind, and a mean bootstrap CI at n = 6 is still not a valid test.
|
|
280
|
+
*/
|
|
281
|
+
/**
|
|
282
|
+
* Which estimator {@link decidePairedPromotion} would use on this data, and the
|
|
283
|
+
* shape facts behind it — for callers that must report the shape on a path
|
|
284
|
+
* where no interval is computed at all (an early rejection, or zero pairs).
|
|
285
|
+
* Cheap: no bootstrap, no interval.
|
|
286
|
+
*/
|
|
287
|
+
function pairedDecisionShape(before, after, statistic = "mean") {
|
|
288
|
+
const tieFraction = before.length === 0 ? null : pairedDeltaTieFraction(before, after);
|
|
289
|
+
if (statistic === "median") return {
|
|
290
|
+
statistic: "median_bootstrap",
|
|
291
|
+
binaryScale: null,
|
|
292
|
+
tieFraction
|
|
293
|
+
};
|
|
294
|
+
const binaryScale = pairedBinaryScale(before, after);
|
|
295
|
+
if (binaryScale !== null) return {
|
|
296
|
+
statistic: "paired_risk_difference",
|
|
297
|
+
binaryScale,
|
|
298
|
+
tieFraction
|
|
299
|
+
};
|
|
300
|
+
return {
|
|
301
|
+
statistic: "mean_bootstrap",
|
|
302
|
+
binaryScale: null,
|
|
303
|
+
tieFraction
|
|
304
|
+
};
|
|
305
|
+
}
|
|
306
|
+
/**
|
|
307
|
+
* Decide whether a paired candidate-minus-baseline delta clears a promotion
|
|
308
|
+
* threshold. `before` is the baseline arm, `after` the candidate arm, paired by
|
|
309
|
+
* position. Throws on unequal lengths.
|
|
310
|
+
*/
|
|
311
|
+
function decidePairedPromotion(before, after, options = {}) {
|
|
312
|
+
if (before.length !== after.length) throw new Error(`decidePairedPromotion: unequal sample sizes (${before.length} vs ${after.length})`);
|
|
313
|
+
const threshold = options.threshold ?? 0;
|
|
314
|
+
if (!Number.isFinite(threshold)) throw new Error(`decidePairedPromotion: threshold must be finite, got ${threshold}`);
|
|
315
|
+
const confidence = options.confidence ?? .95;
|
|
316
|
+
const exactMinimum = minimumPairsForPairedDeltaTest(confidence);
|
|
317
|
+
const requestedMinimum = options.minPairs ?? exactMinimum;
|
|
318
|
+
if (!Number.isInteger(requestedMinimum) || requestedMinimum < 1) throw new Error(`decidePairedPromotion: minPairs must be a positive integer, got ${requestedMinimum}`);
|
|
319
|
+
const minimumPairs = Math.max(requestedMinimum, exactMinimum);
|
|
320
|
+
const n = before.length;
|
|
321
|
+
const sufficient = n >= minimumPairs;
|
|
322
|
+
const { binaryScale, tieFraction } = pairedDecisionShape(before, after, options.statistic);
|
|
323
|
+
let core;
|
|
324
|
+
if (binaryScale !== null) {
|
|
325
|
+
const unitControl = before.map((v) => v / binaryScale);
|
|
326
|
+
const unitTreatment = after.map((v) => v / binaryScale);
|
|
327
|
+
const exact = pairedRiskDifferenceExact(unitControl, unitTreatment, confidence);
|
|
328
|
+
const score = pairedRiskDifferenceScore(unitControl, unitTreatment, confidence);
|
|
329
|
+
const low = score.lower * binaryScale;
|
|
330
|
+
core = {
|
|
331
|
+
statistic: "paired_risk_difference",
|
|
332
|
+
method: "score-interval",
|
|
333
|
+
delta: score.riskDifference * binaryScale,
|
|
334
|
+
low,
|
|
335
|
+
high: score.upper * binaryScale,
|
|
336
|
+
bootstrap: null,
|
|
337
|
+
mcnemar: {
|
|
338
|
+
b: exact.b,
|
|
339
|
+
c: exact.c,
|
|
340
|
+
nDiscordant: exact.nDiscordant,
|
|
341
|
+
pValue: exact.pValue
|
|
342
|
+
},
|
|
343
|
+
pValue: null,
|
|
344
|
+
clearsThreshold: low > threshold,
|
|
345
|
+
label: "success-rate",
|
|
346
|
+
methodDetail: ""
|
|
347
|
+
};
|
|
348
|
+
} else {
|
|
349
|
+
const bootstrapStatistic = options.statistic === "median" ? "median" : "mean";
|
|
350
|
+
const test = pairedDeltaTest(before, after, {
|
|
351
|
+
confidence,
|
|
352
|
+
resamples: options.resamples,
|
|
353
|
+
statistic: bootstrapStatistic,
|
|
354
|
+
seed: options.seed,
|
|
355
|
+
threshold,
|
|
356
|
+
minPairs: options.minPairs
|
|
357
|
+
});
|
|
358
|
+
const ci = test.bootstrap;
|
|
359
|
+
core = {
|
|
360
|
+
statistic: bootstrapStatistic === "mean" ? "mean_bootstrap" : "median_bootstrap",
|
|
361
|
+
method: test.method,
|
|
362
|
+
delta: bootstrapStatistic === "mean" ? ci.mean : ci.median,
|
|
363
|
+
low: ci.low,
|
|
364
|
+
high: ci.high,
|
|
365
|
+
bootstrap: ci,
|
|
366
|
+
mcnemar: null,
|
|
367
|
+
pValue: test.pValue,
|
|
368
|
+
clearsThreshold: test.significant,
|
|
369
|
+
label: bootstrapStatistic,
|
|
370
|
+
methodDetail: test.method === "exact-sign" ? ` Below ${test.minimumPairs} pairs the interval is descriptive only; the decision is the exact one-sided sign test, p=${fmt(test.pValue ?? 1)}.` : ""
|
|
371
|
+
};
|
|
372
|
+
}
|
|
373
|
+
const indeterminate = !Number.isFinite(core.low) || !Number.isFinite(core.high) || core.low === core.high;
|
|
374
|
+
const indeterminateCause = !indeterminate ? "" : tieFraction === 1 ? "every paired delta is an exact tie" : core.mcnemar !== null && core.mcnemar.nDiscordant === 0 ? "every pair is concordant (0 discordant pairs)" : `the ${core.label} CI collapsed to a point at ${fmt(core.low)}`;
|
|
375
|
+
const exactTestVetoes = core.mcnemar !== null && threshold >= 0 && !(core.mcnemar.pValue < 1 - confidence);
|
|
376
|
+
return {
|
|
377
|
+
n,
|
|
378
|
+
threshold,
|
|
379
|
+
confidence,
|
|
380
|
+
binaryScale,
|
|
381
|
+
tieFraction,
|
|
382
|
+
minimumPairs,
|
|
383
|
+
sufficient,
|
|
384
|
+
indeterminate,
|
|
385
|
+
indeterminateCause,
|
|
386
|
+
exactTestVetoes,
|
|
387
|
+
promote: sufficient && !indeterminate && core.clearsThreshold && !exactTestVetoes,
|
|
388
|
+
...core
|
|
389
|
+
};
|
|
390
|
+
}
|
|
391
|
+
function fmt(x) {
|
|
392
|
+
return x.toFixed(4);
|
|
393
|
+
}
|
|
394
|
+
//#endregion
|
|
203
395
|
//#region src/json-recovery.ts
|
|
204
396
|
/**
|
|
205
397
|
* Truncation-tolerant JSON recovery — shared by every parser that reads JSON
|
|
@@ -4178,7 +4370,7 @@ async function compareOptimizationMethods(opts) {
|
|
|
4178
4370
|
confidence: intervalConfidence,
|
|
4179
4371
|
statistic: "mean"
|
|
4180
4372
|
});
|
|
4181
|
-
const favored = boot.low > 0 ? best.name : boot.high < 0 ? other.name : "tie";
|
|
4373
|
+
const favored = !Number.isFinite(boot.low) || !Number.isFinite(boot.high) || boot.low === boot.high ? "tie" : boot.low > 0 ? best.name : boot.high < 0 ? other.name : "tie";
|
|
4182
4374
|
return {
|
|
4183
4375
|
a: best.name,
|
|
4184
4376
|
b: other.name,
|
|
@@ -5294,18 +5486,39 @@ function pairHoldout(candidate, baseline, scenarioIds, select) {
|
|
|
5294
5486
|
cellIds
|
|
5295
5487
|
};
|
|
5296
5488
|
}
|
|
5297
|
-
/**
|
|
5298
|
-
*
|
|
5299
|
-
*
|
|
5300
|
-
*
|
|
5301
|
-
*
|
|
5489
|
+
/**
|
|
5490
|
+
* Significance of the held-out composite lift: ship only when the lower bound
|
|
5491
|
+
* of the interval the outcome's shape ADMITS exceeds `deltaThreshold` (default
|
|
5492
|
+
* 0 ⇒ "confidently positive"). Interpret `deltaThreshold` in the judge's native
|
|
5493
|
+
* scale.
|
|
5494
|
+
*
|
|
5495
|
+
* The decision is delegated whole to {@link decidePairedPromotion}, the one
|
|
5496
|
+
* copy of the rule (`src/paired-promotion-decision.ts`), which `HeldOutGate`
|
|
5497
|
+
* also calls. That module's header carries the measurements; the short version
|
|
5498
|
+
* is three guards a bare `bootstrap.low > threshold` does not have:
|
|
5499
|
+
*
|
|
5500
|
+
* - a two-point (pass/fail) outcome decides on Tango's SCORE interval, the
|
|
5501
|
+
* only paired-binary construction that stays valid at a nonzero margin;
|
|
5502
|
+
* - McNemar's exact test VETOES at any non-negative threshold;
|
|
5503
|
+
* - a ZERO-WIDTH interval is refused rather than promoted, in either
|
|
5504
|
+
* direction — [0,0] clears every negative threshold and [g,g] clears every
|
|
5505
|
+
* threshold below g, and both are an absence of evidence, not a result.
|
|
5506
|
+
*
|
|
5507
|
+
* Measured on this function before those guards landed, at a nominal 5 %:
|
|
5508
|
+
* 14.60 % false promotion at n = 40 on a paired-binary noninferiority boundary,
|
|
5509
|
+
* and 88.50 % at n = 6 under a bounded asymmetric null whose true mean paired
|
|
5510
|
+
* delta is exactly 0.
|
|
5511
|
+
*
|
|
5512
|
+
* At small n, where the percentile bootstrap is descriptive only, a
|
|
5513
|
+
* pre-registered exact sign test still carries the bootstrap path.
|
|
5514
|
+
*/
|
|
5302
5515
|
function heldoutSignificance(paired, opts = {}) {
|
|
5303
5516
|
const deltaThreshold = opts.deltaThreshold ?? 0;
|
|
5304
5517
|
const confidence = opts.confidence ?? .95;
|
|
5305
5518
|
const resamples = opts.resamples ?? 2e3;
|
|
5306
5519
|
const seed = opts.seed ?? 1337;
|
|
5307
5520
|
const statistic = opts.statistic ?? "mean";
|
|
5308
|
-
const decision =
|
|
5521
|
+
const decision = decidePairedPromotion(paired.before, paired.after, {
|
|
5309
5522
|
confidence,
|
|
5310
5523
|
resamples,
|
|
5311
5524
|
statistic,
|
|
@@ -5313,7 +5526,12 @@ function heldoutSignificance(paired, opts = {}) {
|
|
|
5313
5526
|
threshold: deltaThreshold,
|
|
5314
5527
|
minPairs: opts.minProductiveRuns
|
|
5315
5528
|
});
|
|
5316
|
-
const bootstrap = decision.bootstrap
|
|
5529
|
+
const bootstrap = decision.bootstrap ?? pairedBootstrap(paired.before, paired.after, {
|
|
5530
|
+
confidence,
|
|
5531
|
+
resamples,
|
|
5532
|
+
statistic,
|
|
5533
|
+
seed
|
|
5534
|
+
});
|
|
5317
5535
|
const medianBootstrap = statistic === "median" ? bootstrap : pairedBootstrap(paired.before, paired.after, {
|
|
5318
5536
|
confidence,
|
|
5319
5537
|
resamples,
|
|
@@ -5328,19 +5546,20 @@ function heldoutSignificance(paired, opts = {}) {
|
|
|
5328
5546
|
if (Math.abs(after - before) < 1e-9) ties += 1;
|
|
5329
5547
|
}
|
|
5330
5548
|
const tieFraction = n === 0 ? 0 : ties / n;
|
|
5331
|
-
const fewRuns = !decision.sufficient;
|
|
5332
|
-
const significant = decision.significant;
|
|
5333
5549
|
return {
|
|
5334
5550
|
paired,
|
|
5335
5551
|
bootstrap,
|
|
5336
5552
|
medianBootstrap,
|
|
5553
|
+
decision,
|
|
5554
|
+
decisionStatistic: decision.statistic,
|
|
5555
|
+
mcnemar: decision.mcnemar,
|
|
5337
5556
|
tieFraction,
|
|
5338
5557
|
n,
|
|
5339
5558
|
minimumRequired: decision.minimumPairs,
|
|
5340
5559
|
decisionMethod: decision.method,
|
|
5341
5560
|
pValue: decision.pValue,
|
|
5342
|
-
significant,
|
|
5343
|
-
fewRuns
|
|
5561
|
+
significant: decision.promote,
|
|
5562
|
+
fewRuns: !decision.sufficient
|
|
5344
5563
|
};
|
|
5345
5564
|
}
|
|
5346
5565
|
/** Detect the native scale of a set of scores: 0-100 when any magnitude clears
|
|
@@ -5354,30 +5573,50 @@ function detectScale(values) {
|
|
|
5354
5573
|
* a dimension is "regressed" when the CI lower bound < −tolerance (conservative
|
|
5355
5574
|
* — blocks if the credible worst case exceeds tolerance, which is the right
|
|
5356
5575
|
* posture for safety dimensions like `hallucination_free`). When `tolerance`
|
|
5357
|
-
* is omitted it auto-scales: 0.05 on [0,1], 5 on 0-100.
|
|
5576
|
+
* is omitted it auto-scales: 0.05 on [0,1], 5 on 0-100.
|
|
5577
|
+
*
|
|
5578
|
+
* The interval comes from {@link decidePairedPromotion}, so a pass/fail
|
|
5579
|
+
* dimension is judged on Tango's score interval rather than a percentile
|
|
5580
|
+
* bootstrap of the mean — `tolerance` is a NONZERO margin, and the bootstrap
|
|
5581
|
+
* is not a valid interval at one. That matters most here because this guard
|
|
5582
|
+
* fails OPEN by construction: `tolerance` is positive, so an interval pinned at
|
|
5583
|
+
* [0,0] never satisfies `low < −tolerance` and a real regression on a safety
|
|
5584
|
+
* dimension would be reported as `regressed: false`. On the median it fails the
|
|
5585
|
+
* same way for the same reason — when most pairs tie, which is automatic for a
|
|
5586
|
+
* pass/fail dimension on {0,1} and on the 0-100 encoding `detectScale` exists
|
|
5587
|
+
* to support, the median CI collapses to [0,0]. Pass `statistic: 'median'` to
|
|
5588
|
+
* restore the pre-0.134 behaviour. */
|
|
5358
5589
|
function dimensionRegressions(candidate, baseline, scenarioIds, criticalDimensions, opts = {}) {
|
|
5359
5590
|
const out = [];
|
|
5360
5591
|
for (const dim of criticalDimensions) {
|
|
5361
5592
|
const paired = pairHoldout(candidate, baseline, scenarioIds, (s) => s.dimensions[dim]);
|
|
5362
5593
|
if (paired.before.length === 0) continue;
|
|
5363
5594
|
const tolerance = opts.tolerance ?? .05 * detectScale([...paired.before, ...paired.after]);
|
|
5364
|
-
const
|
|
5595
|
+
const bootstrapStatistic = opts.statistic ?? "mean";
|
|
5596
|
+
const shared = {
|
|
5365
5597
|
confidence: opts.confidence ?? .95,
|
|
5366
5598
|
resamples: opts.resamples ?? 2e3,
|
|
5367
|
-
statistic:
|
|
5599
|
+
statistic: bootstrapStatistic,
|
|
5368
5600
|
seed: opts.seed ?? 1337
|
|
5369
|
-
}
|
|
5370
|
-
const
|
|
5371
|
-
|
|
5372
|
-
|
|
5373
|
-
statistic: "median",
|
|
5374
|
-
seed: opts.seed ?? 1337,
|
|
5601
|
+
};
|
|
5602
|
+
const guard = decidePairedPromotion(paired.before, paired.after, shared);
|
|
5603
|
+
const regression = decidePairedPromotion(paired.after, paired.before, {
|
|
5604
|
+
...shared,
|
|
5375
5605
|
threshold: tolerance
|
|
5376
5606
|
});
|
|
5607
|
+
const bootstrap = guard.bootstrap ?? pairedBootstrap(paired.before, paired.after, shared);
|
|
5377
5608
|
out.push({
|
|
5378
5609
|
dimension: dim,
|
|
5379
5610
|
bootstrap,
|
|
5380
|
-
|
|
5611
|
+
bootstrapStatistic,
|
|
5612
|
+
ci: {
|
|
5613
|
+
low: guard.low,
|
|
5614
|
+
high: guard.high
|
|
5615
|
+
},
|
|
5616
|
+
decisionStatistic: guard.statistic,
|
|
5617
|
+
mcnemar: guard.mcnemar,
|
|
5618
|
+
indeterminate: guard.indeterminate,
|
|
5619
|
+
regressed: bootstrap.low < -tolerance || regression.promote,
|
|
5381
5620
|
tolerance,
|
|
5382
5621
|
n: paired.before.length
|
|
5383
5622
|
});
|
|
@@ -5430,7 +5669,8 @@ function defaultProductionGate(options) {
|
|
|
5430
5669
|
seed,
|
|
5431
5670
|
statistic: heldoutStatistic
|
|
5432
5671
|
});
|
|
5433
|
-
|
|
5672
|
+
const dec = sig.decision;
|
|
5673
|
+
delta = dec.delta;
|
|
5434
5674
|
const heldoutPass = sig.significant;
|
|
5435
5675
|
contributing.push({
|
|
5436
5676
|
name: "heldout-significance",
|
|
@@ -5438,13 +5678,20 @@ function defaultProductionGate(options) {
|
|
|
5438
5678
|
detail: {
|
|
5439
5679
|
n: sig.n,
|
|
5440
5680
|
delta,
|
|
5681
|
+
decisionStatistic: sig.decisionStatistic,
|
|
5682
|
+
decisionMethod: sig.decisionMethod,
|
|
5683
|
+
binaryScale: dec.binaryScale,
|
|
5684
|
+
mcnemar: dec.mcnemar,
|
|
5685
|
+
indeterminate: dec.indeterminate,
|
|
5441
5686
|
deltaMean: sig.bootstrap.mean,
|
|
5442
5687
|
deltaMedianDiagnostic: sig.medianBootstrap.median,
|
|
5443
5688
|
deltaMedian: sig.medianBootstrap.median,
|
|
5444
5689
|
tieFraction: sig.tieFraction,
|
|
5445
|
-
ciLow:
|
|
5446
|
-
ciHigh:
|
|
5447
|
-
|
|
5690
|
+
ciLow: dec.low,
|
|
5691
|
+
ciHigh: dec.high,
|
|
5692
|
+
bootstrapCiLow: sig.bootstrap.low,
|
|
5693
|
+
bootstrapCiHigh: sig.bootstrap.high,
|
|
5694
|
+
confidence: dec.confidence,
|
|
5448
5695
|
deltaThreshold,
|
|
5449
5696
|
fewRuns: sig.fewRuns
|
|
5450
5697
|
}
|
|
@@ -5452,7 +5699,8 @@ function defaultProductionGate(options) {
|
|
|
5452
5699
|
if (sig.fewRuns) requiredUnavailable.add("heldout-significance");
|
|
5453
5700
|
if (!heldoutPass) {
|
|
5454
5701
|
const tieNote = sig.tieFraction >= .4 ? `; ${(sig.tieFraction * 100).toFixed(0)}% tied scenarios` : "";
|
|
5455
|
-
|
|
5702
|
+
const ci = `${(dec.confidence * 100).toFixed(0)}% CI [${dec.low.toFixed(3)}, ${dec.high.toFixed(3)}]`;
|
|
5703
|
+
reasons.push(sig.fewRuns ? `held-out: only ${sig.n} paired runs (< ${sig.minimumRequired}) — too few to claim significance` : dec.indeterminate ? `held-out: ${dec.indeterminateCause}, so the paired CI is ${ci} and carries no direction — it cannot clear threshold ${deltaThreshold} on evidence${tieNote}` : dec.exactTestVetoes ? `held-out: McNemar exact p=${dec.mcnemar?.pValue.toExponential(2)} does not reject at α=${(1 - dec.confidence).toFixed(4)} (${dec.label} Δ ${delta.toFixed(3)}, ${ci}${tieNote})` : `held-out CI.low ${dec.low.toFixed(3)} ≤ threshold ${deltaThreshold} (${dec.label} Δ ${delta.toFixed(3)}, ${ci}${tieNote})`);
|
|
5456
5704
|
}
|
|
5457
5705
|
}
|
|
5458
5706
|
const dimensionsProvided = options.criticalDimensions !== void 0;
|
|
@@ -5480,7 +5728,11 @@ function defaultProductionGate(options) {
|
|
|
5480
5728
|
missingDimensions,
|
|
5481
5729
|
regressions: dimRegs.map((result) => ({
|
|
5482
5730
|
dimension: result.dimension,
|
|
5483
|
-
ciLow: result.
|
|
5731
|
+
ciLow: result.ci.low,
|
|
5732
|
+
ciHigh: result.ci.high,
|
|
5733
|
+
decisionStatistic: result.decisionStatistic,
|
|
5734
|
+
indeterminate: result.indeterminate,
|
|
5735
|
+
bootstrapCiLow: result.bootstrap.low,
|
|
5484
5736
|
median: result.bootstrap.median,
|
|
5485
5737
|
tolerance: result.tolerance,
|
|
5486
5738
|
n: result.n,
|
|
@@ -5492,7 +5744,7 @@ function defaultProductionGate(options) {
|
|
|
5492
5744
|
requiredUnavailable.add("dimension-regression");
|
|
5493
5745
|
reasons.push(`critical dimension(s) were not scored: ${missingDimensions.join(", ")}`);
|
|
5494
5746
|
}
|
|
5495
|
-
if (regressed.length > 0) reasons.push(`critical dimension(s) regressed: ${regressed.map((result) => `${result.dimension} CI.low ${result.
|
|
5747
|
+
if (regressed.length > 0) reasons.push(`critical dimension(s) regressed: ${regressed.map((result) => `${result.dimension} CI.low ${result.ci.low.toFixed(3)} < -${result.tolerance}`).join("; ")}`);
|
|
5496
5748
|
}
|
|
5497
5749
|
const budgetUsd = options.budgetUsd;
|
|
5498
5750
|
const budgetConfigured = budgetUsd !== void 0;
|
|
@@ -5646,8 +5898,10 @@ function extractText(artifact) {
|
|
|
5646
5898
|
//#endregion
|
|
5647
5899
|
//#region src/campaign/gates/heldout-gate.ts
|
|
5648
5900
|
/**
|
|
5649
|
-
* Composable held-out gate: ships only when the
|
|
5650
|
-
*
|
|
5901
|
+
* Composable held-out gate: ships only when the lower bound of the DECIDING
|
|
5902
|
+
* paired interval on the candidate-minus-baseline composite delta clears
|
|
5903
|
+
* `deltaThreshold` — Tango's score interval on a pass/fail holdout, the mean
|
|
5904
|
+
* bootstrap otherwise. See {@link decidePairedPromotion}.
|
|
5651
5905
|
*/
|
|
5652
5906
|
function heldOutGate(options) {
|
|
5653
5907
|
const deltaThreshold = options.deltaThreshold ?? .5;
|
|
@@ -5667,24 +5921,35 @@ function heldOutGate(options) {
|
|
|
5667
5921
|
resamples,
|
|
5668
5922
|
seed
|
|
5669
5923
|
});
|
|
5670
|
-
const
|
|
5924
|
+
const dec = sig.decision;
|
|
5925
|
+
const delta = dec.delta;
|
|
5671
5926
|
const passed = sig.significant;
|
|
5672
5927
|
const status = sig.fewRuns ? "not_evaluated" : passed ? "pass" : "fail";
|
|
5673
5928
|
const tieNote = sig.tieFraction >= .4 ? `, ${(sig.tieFraction * 100).toFixed(0)}% tied` : "";
|
|
5674
|
-
const ci = `${(
|
|
5929
|
+
const ci = `${(dec.confidence * 100).toFixed(0)}% CI [${dec.low.toFixed(3)}, ${dec.high.toFixed(3)}]`;
|
|
5930
|
+
const held = `held-out ${dec.label} Δ ${delta.toFixed(3)}`;
|
|
5931
|
+
const holdReason = sig.fewRuns ? `held-out: only ${sig.n} paired runs; ${sig.minimumRequired} required — too few to claim significance` : dec.indeterminate ? `held-out: ${dec.indeterminateCause}, so the paired CI is ${ci} and carries no direction — it cannot clear ${deltaThreshold} on evidence (n=${sig.n}${tieNote})` : dec.exactTestVetoes ? `${held}, McNemar exact p=${dec.mcnemar?.pValue.toExponential(2)} does not reject at α=${(1 - dec.confidence).toFixed(4)} (${ci}, n=${sig.n}${tieNote})` : `${held}, CI.low ${dec.low.toFixed(3)} ≤ ${deltaThreshold} (${ci}, n=${sig.n}${tieNote})`;
|
|
5675
5932
|
return {
|
|
5676
5933
|
decision: passed ? "ship" : "hold",
|
|
5677
|
-
reasons: passed ? [
|
|
5934
|
+
reasons: passed ? [`${held}, CI.low ${dec.low.toFixed(3)} > ${deltaThreshold} (${ci}, n=${sig.n}${tieNote})`] : [holdReason],
|
|
5678
5935
|
contributingGates: [{
|
|
5679
5936
|
name: "heldOutGate",
|
|
5680
5937
|
status,
|
|
5681
5938
|
detail: {
|
|
5682
|
-
deltaMean:
|
|
5939
|
+
deltaMean: sig.bootstrap.mean,
|
|
5940
|
+
decidingDelta: delta,
|
|
5941
|
+
decisionStatistic: sig.decisionStatistic,
|
|
5942
|
+
decisionMethod: sig.decisionMethod,
|
|
5943
|
+
binaryScale: dec.binaryScale,
|
|
5944
|
+
mcnemar: dec.mcnemar,
|
|
5945
|
+
indeterminate: dec.indeterminate,
|
|
5683
5946
|
deltaMedianDiagnostic: sig.medianBootstrap.median,
|
|
5684
5947
|
tieFraction: sig.tieFraction,
|
|
5685
|
-
ciLow:
|
|
5686
|
-
ciHigh:
|
|
5687
|
-
|
|
5948
|
+
ciLow: dec.low,
|
|
5949
|
+
ciHigh: dec.high,
|
|
5950
|
+
bootstrapCiLow: sig.bootstrap.low,
|
|
5951
|
+
bootstrapCiHigh: sig.bootstrap.high,
|
|
5952
|
+
confidence: dec.confidence,
|
|
5688
5953
|
n: sig.n,
|
|
5689
5954
|
deltaThreshold,
|
|
5690
5955
|
fewRuns: sig.fewRuns,
|
|
@@ -5793,29 +6058,44 @@ function buildEvidenceVector(ctx, objectives, opts = {}) {
|
|
|
5793
6058
|
const n = paired.before.length;
|
|
5794
6059
|
const floorTolerance = obj.floorTolerance ?? .05 * detectScale([...paired.before, ...paired.after]);
|
|
5795
6060
|
const gainThreshold = obj.gainThreshold ?? 0;
|
|
5796
|
-
const
|
|
6061
|
+
const bootstrapStatistic = opts.statistic ?? "mean";
|
|
6062
|
+
const improvement = decidePairedPromotion(before, after, {
|
|
5797
6063
|
confidence,
|
|
5798
6064
|
resamples,
|
|
5799
|
-
statistic:
|
|
6065
|
+
statistic: bootstrapStatistic,
|
|
5800
6066
|
seed,
|
|
5801
6067
|
threshold: gainThreshold,
|
|
5802
6068
|
minPairs: opts.minProductiveRuns
|
|
5803
6069
|
});
|
|
5804
|
-
const regression =
|
|
6070
|
+
const regression = decidePairedPromotion(after, before, {
|
|
5805
6071
|
confidence,
|
|
5806
6072
|
resamples,
|
|
5807
|
-
statistic:
|
|
6073
|
+
statistic: bootstrapStatistic,
|
|
5808
6074
|
seed,
|
|
5809
6075
|
threshold: floorTolerance,
|
|
5810
6076
|
minPairs: opts.minProductiveRuns
|
|
5811
6077
|
});
|
|
5812
|
-
const bootstrap = improvement.bootstrap
|
|
5813
|
-
|
|
6078
|
+
const bootstrap = improvement.bootstrap ?? pairedBootstrap(before, after, {
|
|
6079
|
+
confidence,
|
|
6080
|
+
resamples,
|
|
6081
|
+
statistic: bootstrapStatistic,
|
|
6082
|
+
seed
|
|
6083
|
+
});
|
|
6084
|
+
const floorBreached = bootstrap.low < -floorTolerance || regression.promote;
|
|
6085
|
+
const verdict = !improvement.sufficient ? "few_runs" : floorBreached ? "regressed" : improvement.promote ? "improved" : "flat";
|
|
5814
6086
|
axes.push({
|
|
5815
6087
|
name: obj.name,
|
|
5816
6088
|
source: obj.source,
|
|
5817
6089
|
direction: obj.direction,
|
|
5818
6090
|
bootstrap,
|
|
6091
|
+
bootstrapStatistic,
|
|
6092
|
+
ci: {
|
|
6093
|
+
low: improvement.low,
|
|
6094
|
+
high: improvement.high
|
|
6095
|
+
},
|
|
6096
|
+
decisionStatistic: improvement.statistic,
|
|
6097
|
+
mcnemar: improvement.mcnemar,
|
|
6098
|
+
indeterminate: improvement.indeterminate,
|
|
5819
6099
|
n,
|
|
5820
6100
|
minimumRequired: improvement.minimumPairs,
|
|
5821
6101
|
decisionMethod: improvement.method,
|
|
@@ -5851,8 +6131,14 @@ const paretoPolicy = (ev) => {
|
|
|
5851
6131
|
verdict: ax.verdict,
|
|
5852
6132
|
n: ax.n,
|
|
5853
6133
|
deltaMedian: ax.bootstrap.median,
|
|
5854
|
-
ciLow: ax.
|
|
5855
|
-
ciHigh: ax.
|
|
6134
|
+
ciLow: ax.ci.low,
|
|
6135
|
+
ciHigh: ax.ci.high,
|
|
6136
|
+
decisionStatistic: ax.decisionStatistic,
|
|
6137
|
+
decisionMethod: ax.decisionMethod,
|
|
6138
|
+
mcnemar: ax.mcnemar,
|
|
6139
|
+
indeterminate: ax.indeterminate,
|
|
6140
|
+
bootstrapCiLow: ax.bootstrap.low,
|
|
6141
|
+
bootstrapCiHigh: ax.bootstrap.high,
|
|
5856
6142
|
confidence: ax.bootstrap.confidence,
|
|
5857
6143
|
gainThreshold: ax.gainThreshold,
|
|
5858
6144
|
floorTolerance: ax.floorTolerance
|
|
@@ -5865,13 +6151,13 @@ const paretoPolicy = (ev) => {
|
|
|
5865
6151
|
const reasons = [];
|
|
5866
6152
|
if (regressed.length > 0) {
|
|
5867
6153
|
decision = "hold";
|
|
5868
|
-
for (const a of regressed) reasons.push(`objective '${a.name}' regressed: good-direction CI.low ${a.
|
|
6154
|
+
for (const a of regressed) reasons.push(`objective '${a.name}' regressed: good-direction CI.low ${a.ci.low.toFixed(3)} < -${a.floorTolerance} (n=${a.n})`);
|
|
5869
6155
|
} else if (fewRuns.length > 0) {
|
|
5870
6156
|
decision = "need_more_work";
|
|
5871
6157
|
for (const a of fewRuns) reasons.push(`objective '${a.name}' has only n=${a.n} paired runs — insufficient evidence to claim significance`);
|
|
5872
6158
|
} else if (improved.length > 0) {
|
|
5873
6159
|
decision = "ship";
|
|
5874
|
-
reasons.push(`Pareto improvement at the confidence level: ${improved.map((a) => `'${a.name}' +${a.bootstrap.
|
|
6160
|
+
reasons.push(`Pareto improvement at the confidence level: ${improved.map((a) => `'${a.name}' +${a.ci.low > 0 ? a.ci.low.toFixed(3) : a.bootstrap.mean.toFixed(3)} (CI.low ${a.ci.low.toFixed(3)})`).join(", ")}; no objective regressed`);
|
|
5875
6161
|
} else {
|
|
5876
6162
|
decision = "hold";
|
|
5877
6163
|
reasons.push("no Pareto improvement: candidate statistically equivalent to baseline on every objective");
|
|
@@ -7833,6 +8119,6 @@ function skillOptOptimizationMethod(config) {
|
|
|
7833
8119
|
};
|
|
7834
8120
|
}
|
|
7835
8121
|
//#endregion
|
|
7836
|
-
export { SEARCH_LEDGER_FILE_CONTEXT as $, acquireSingleRunLock as A, dominates as At, surfaceContentHash as B,
|
|
8122
|
+
export { SEARCH_LEDGER_FILE_CONTEXT as $, acquireSingleRunLock as A, dominates as At, surfaceContentHash as B, assertRealAgentReceipts as Bt, detectScale as C, llmJudge as Ct, runCanaries as D, fileVerdictCache as Dt, pairHoldout as E, contentHash as Et, assertCodeSurfaceIdentity as F, decidePairedPromotion as Ft, DEFAULT_MUTATION_PRIMITIVES as G, campaignBreakdown as H, summarizeAgentReceiptIntegrity as Ht, assertComponentSurface as I, pairedDecisionShape as It, planCampaignRun as J, buildReflectionPrompt as K, codeSurfaceIdentityMaterial as L, minimumPairsForPairedDeltaTest as Lt, compareOptimizationMethods as M, paretoFrontierWithCrowding as Mt, costFromLedgerSummary as N, scalarScore as Nt, composeGate as O, inMemoryVerdictCache as Ot, optimizationTokenUsageFromSummary as P, recoverTruncatedJson as Pt, inMemoryCampaignStorage as Q, componentSurfaceIdentityMaterial as R, pairedDeltaTest as Rt, defaultProductionGate as S, hashScenarios as St, heldoutSignificance as T, canonicalJson as Tt, campaignMeanComposite as U, summarizeBackendIntegrity as Ut, surfaceHash as V, assertRealBackend as Vt, compareRankKeys as W, JudgeParseError as Wt, createRunCostLedger as X, runCampaign as Y, fsCampaignStorage as Z, buildEvidenceVector as _, redTeamReport as _t, emitLoopProvenance as a, assertCampaignDesign as at, powerPreflight as b, Dataset as bt, provenanceRecordPath as c, campaignSplitDigest as ct, runImprovementLoop as d, REFERENCE_EQUIVALENCE_INPUT_LIMITS as dt, SearchLedgerConflictError as et, runOptimization as f, REFERENCE_EQUIVALENCE_JUDGE_VERSION as ft, gepaOptimizationMethod as g, redTeamDataset as gt, labelTrustRank as h, DEFAULT_RED_TEAM_CORPUS as ht, canonicalDigest as i, tangleTracesRoot as it, assertOptimizationResult as j, paretoFrontier as jt, externalTextOptimizationMethod as k, crowdingDistance as kt, provenanceSpansPath as l, campaignSplitDigestFromIdentities as lt, isProposedCandidate as m, runReferenceEquivalenceJudge as mt, buildLoopProvenanceRecord as n, SearchLedgerIntegrityError as nt, loopProvenanceArgsFromResult as o, assertCampaignSplitIdentity as ot, runEval as p, createReferenceEquivalenceJudge as pt, parseReflectionResponse as q, campaignMeasurementDigest as r, resolveRunDir as rt, loopProvenanceSpans as s, campaignScenarioIdentity as st, skillOptOptimizationMethod as t, SearchLedgerError as tt, verifyLoopProvenanceRecord as u, openAutoPr as ut, paretoPolicy as v, scoreRedTeamOutput as vt, dimensionRegressions as w, cachedJudge as wt, heldOutGate as x, HoldoutLockedError as xt, paretoSignificanceGate as y, toolNamesForRun as yt, renderSurfaceDiff as z, BackendIntegrityError as zt };
|
|
7837
8123
|
|
|
7838
|
-
//# sourceMappingURL=skillopt-optimization-method-
|
|
8124
|
+
//# sourceMappingURL=skillopt-optimization-method-0UmPD6aP.js.map
|