@tangle-network/agent-eval 0.133.2 → 0.133.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +156 -0
- package/dist/{analyze-runs-BmX-h_yn.d.ts → analyze-runs-BClW9OSe.d.ts} +3 -3
- package/dist/{analyze-runs-BmX-h_yn.d.ts.map → analyze-runs-BClW9OSe.d.ts.map} +1 -1
- package/dist/{analyze-runs-B-afTpCv.js → analyze-runs-qk8op0tN.js} +63 -42
- package/dist/analyze-runs-qk8op0tN.js.map +1 -0
- package/dist/baseline-BaPxoROc.js +149 -0
- package/dist/baseline-BaPxoROc.js.map +1 -0
- package/dist/{baseline-hG3K85h4.d.ts → baseline-D_fT6277.d.ts} +43 -11
- package/dist/baseline-D_fT6277.d.ts.map +1 -0
- package/dist/benchmarks/index.d.ts +1 -1
- package/dist/benchmarks/index.js +1 -1
- package/dist/{benchmarks-CJr1H1_a.js → benchmarks-BP9sgMia.js} +3 -3
- package/dist/{benchmarks-CJr1H1_a.js.map → benchmarks-BP9sgMia.js.map} +1 -1
- package/dist/builder-eval/index.js +1 -1
- package/dist/campaign/index.d.ts +2 -2
- package/dist/campaign/index.js +2 -2
- package/dist/{campaign-BJjn1rhw.js → campaign--V4ffEKR.js} +12 -6
- package/dist/{campaign-BJjn1rhw.js.map → campaign--V4ffEKR.js.map} +1 -1
- package/dist/{client-COvaLoQG.d.ts → client-Du7B81wW.d.ts} +28 -14
- package/dist/client-Du7B81wW.d.ts.map +1 -0
- package/dist/{client-CYzbdJOZ.js → client-LIuo-KPv.js} +19 -7
- package/dist/client-LIuo-KPv.js.map +1 -0
- package/dist/contract/index.d.ts +3 -3
- package/dist/contract/index.d.ts.map +1 -1
- package/dist/contract/index.js +9 -8
- package/dist/contract/index.js.map +1 -1
- package/dist/{eval-campaign-DXhpZghy.js → eval-campaign-CvPcvqXC.js} +2 -2
- package/dist/{eval-campaign-DXhpZghy.js.map → eval-campaign-CvPcvqXC.js.map} +1 -1
- package/dist/hosted/index.d.ts +1 -1
- package/dist/hosted/index.d.ts.map +1 -1
- package/dist/hosted/index.js +1 -1
- package/dist/{index-BREtv3ZZ.d.ts → index-B5MNN1f1.d.ts} +3 -3
- package/dist/{index-BREtv3ZZ.d.ts.map → index-B5MNN1f1.d.ts.map} +1 -1
- package/dist/{index-C7Wue8R6.d.ts → index-DOqvIJ8I.d.ts} +27 -10
- package/dist/index-DOqvIJ8I.d.ts.map +1 -0
- package/dist/{index-DSC51roc.d.ts → index-DSC51roc2.d.ts} +1 -1
- package/dist/index-DSC51roc2.d.ts.map +1 -0
- package/dist/{index-nhIYz9hn.d.ts → index-DuhJaaiH.d.ts} +68 -7
- package/dist/index-DuhJaaiH.d.ts.map +1 -0
- package/dist/index.d.ts +56 -10
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +29 -20
- package/dist/index.js.map +1 -1
- package/dist/ledger-core/index.d.ts +2 -2
- package/dist/ledger-core/index.js +2 -2
- package/dist/{ledger-core-CPZfcrC2.js → ledger-core-DAKFKRzi.js} +136 -18
- package/dist/ledger-core-DAKFKRzi.js.map +1 -0
- package/dist/matrix/index.d.ts +1 -1
- package/dist/meta-eval/index.d.ts +1 -1
- package/dist/meta-eval/index.js +2 -2
- package/dist/multishot/index.d.ts +1 -1
- package/dist/openapi.json +1 -1
- package/dist/{paired-arms-6XItKzd1.js → paired-arms-CA_8pN01.js} +2 -2
- package/dist/{paired-arms-6XItKzd1.js.map → paired-arms-CA_8pN01.js.map} +1 -1
- package/dist/pipelines/index.d.ts +1 -1
- package/dist/pipelines/index.js +3 -2
- package/dist/pipelines/index.js.map +1 -1
- package/dist/{release-report-wuilQkvK.js → release-report-BVZBmRZp.js} +2 -2
- package/dist/{release-report-wuilQkvK.js.map → release-report-BVZBmRZp.js.map} +1 -1
- package/dist/{release-report-CjHWa8Ia.d.ts → release-report-DKBtegGt.d.ts} +2 -2
- package/dist/{release-report-CjHWa8Ia.d.ts.map → release-report-DKBtegGt.d.ts.map} +1 -1
- package/dist/reporting.d.ts +3 -3
- package/dist/reporting.js +4 -4
- package/dist/{researcher-CbSKhK8z.d.ts → researcher-BtD5U1Up.d.ts} +2 -2
- package/dist/{researcher-CbSKhK8z.d.ts.map → researcher-BtD5U1Up.d.ts.map} +1 -1
- package/dist/{reward-hacking-Dl2UBzej.js → reward-hacking-DCdRK9TY.js} +2 -2
- package/dist/{reward-hacking-Dl2UBzej.js.map → reward-hacking-DCdRK9TY.js.map} +1 -1
- package/dist/rl.d.ts +1 -1
- package/dist/rl.js +4 -4
- package/dist/{rubric-predictive-validity-QG7ydk0s.js → rubric-predictive-validity-D6Q6n9oq.js} +2 -2
- package/dist/{rubric-predictive-validity-QG7ydk0s.js.map → rubric-predictive-validity-D6Q6n9oq.js.map} +1 -1
- package/dist/{skillopt-optimization-method-wHF5xsUv.d.ts → skillopt-optimization-method-Dxr8pdZd.d.ts} +12 -7
- package/dist/{skillopt-optimization-method-wHF5xsUv.d.ts.map → skillopt-optimization-method-Dxr8pdZd.d.ts.map} +1 -1
- package/dist/{skillopt-optimization-method-CF6a327Q.js → skillopt-optimization-method-vvJ4bMNI.js} +119 -24
- package/dist/skillopt-optimization-method-vvJ4bMNI.js.map +1 -0
- package/dist/{statistics-DbvkkDPa.d.ts → statistics-D_4Snl-5.d.ts} +158 -30
- package/dist/statistics-D_4Snl-5.d.ts.map +1 -0
- package/dist/{statistics-DWM_AyLe.js → statistics-RwRNu2__.js} +546 -98
- package/dist/statistics-RwRNu2__.js.map +1 -0
- package/dist/{summary-report-Ci17nIdU.js → summary-report-BxtossFi.js} +3 -3
- package/dist/{summary-report-Ci17nIdU.js.map → summary-report-BxtossFi.js.map} +1 -1
- package/dist/{summary-report-CFnQgNfg.d.ts → summary-report-DyOhItws.d.ts} +3 -2
- package/dist/summary-report-DyOhItws.d.ts.map +1 -0
- package/dist/{baseline-DcX5hQDv.js → tool-use-metrics-DEGMKycK.js} +2 -114
- package/dist/tool-use-metrics-DEGMKycK.js.map +1 -0
- package/docs/design/statistics-decisions.md +271 -0
- package/docs/design.md +1 -0
- package/docs/insight-report.md +1 -1
- package/docs/research-report-methodology.md +4 -1
- package/package.json +2 -1
- package/dist/analyze-runs-B-afTpCv.js.map +0 -1
- package/dist/baseline-DcX5hQDv.js.map +0 -1
- package/dist/baseline-hG3K85h4.d.ts.map +0 -1
- package/dist/client-COvaLoQG.d.ts.map +0 -1
- package/dist/client-CYzbdJOZ.js.map +0 -1
- package/dist/index-C7Wue8R6.d.ts.map +0 -1
- package/dist/index-DSC51roc.d.ts.map +0 -1
- package/dist/index-nhIYz9hn.d.ts.map +0 -1
- package/dist/ledger-core-CPZfcrC2.js.map +0 -1
- package/dist/skillopt-optimization-method-CF6a327Q.js.map +0 -1
- package/dist/statistics-DWM_AyLe.js.map +0 -1
- package/dist/statistics-DbvkkDPa.d.ts.map +0 -1
- package/dist/summary-report-CFnQgNfg.d.ts.map +0 -1
package/dist/{skillopt-optimization-method-CF6a327Q.js → skillopt-optimization-method-vvJ4bMNI.js}
RENAMED
|
@@ -3,10 +3,10 @@ import { c as costForTokenPricing, i as CostLedger, t as CostAccountingIncomplet
|
|
|
3
3
|
import { f as maximumChargeForLlmRequest, l as costReceiptFromLlm, m as stripFencedJson, u as costReceiptFromLlmError } from "./llm-client-ClPW-dWB.js";
|
|
4
4
|
import { i as combineAbortSignals, r as clamp01 } from "./run-score-iEEAWiBY.js";
|
|
5
5
|
import { n as mapConcurrent } from "./concurrency-MUjT7VjM.js";
|
|
6
|
-
import {
|
|
6
|
+
import { C as pairedBootstrap, D as pairedSignTest, I as weightedComposite, u as confidenceInterval } from "./statistics-RwRNu2__.js";
|
|
7
7
|
import { t as DEFAULT_REDACTION_RULES } from "./redact-7Aq1ukl-.js";
|
|
8
|
-
import { a as campaignCellExecutionEvidence, l as projectCampaignCellQuality, t as detectRewardHacking } from "./reward-hacking-
|
|
9
|
-
import { a as
|
|
8
|
+
import { a as campaignCellExecutionEvidence, l as projectCampaignCellQuality, t as detectRewardHacking } from "./reward-hacking-DCdRK9TY.js";
|
|
9
|
+
import { a as appendLedgerLine, h as tryAcquireAtomicFileLock, m as probeAtomicFileLock, o as tryWithLedgerFileLock } from "./ledger-core-DAKFKRzi.js";
|
|
10
10
|
import { createRequire } from "node:module";
|
|
11
11
|
import { z } from "zod";
|
|
12
12
|
import { appendFileSync, existsSync, readFileSync, writeFileSync } from "node:fs";
|
|
@@ -155,6 +155,51 @@ function assertBackendReport(report, opts) {
|
|
|
155
155
|
return report;
|
|
156
156
|
}
|
|
157
157
|
//#endregion
|
|
158
|
+
//#region src/paired-delta-test.ts
|
|
159
|
+
/** Smallest all-positive sample that can clear a one-sided exact sign test. */
|
|
160
|
+
function minimumPairsForPairedDeltaTest(confidence = .95) {
|
|
161
|
+
if (!Number.isFinite(confidence) || confidence <= 0 || confidence >= 1) throw new Error(`minimumPairsForPairedDeltaTest: confidence must be in (0,1), got ${confidence}`);
|
|
162
|
+
const oneSidedAlpha = (1 - confidence) / 2;
|
|
163
|
+
return Math.ceil(Math.log2(1 / oneSidedAlpha));
|
|
164
|
+
}
|
|
165
|
+
/**
|
|
166
|
+
* Tests whether a paired candidate-minus-baseline delta clears a threshold.
|
|
167
|
+
*
|
|
168
|
+
* At 20 or more pairs, the percentile bootstrap lower bound carries the
|
|
169
|
+
* decision. Below that point the interval is descriptive only, so the function
|
|
170
|
+
* switches to a pre-registered one-sided exact sign test. The exact path is
|
|
171
|
+
* deliberately conservative: it requires both a point estimate above the
|
|
172
|
+
* threshold and enough consistently positive paired differences.
|
|
173
|
+
*/
|
|
174
|
+
function pairedDeltaTest(before, after, options = {}) {
|
|
175
|
+
const threshold = options.threshold ?? 0;
|
|
176
|
+
if (!Number.isFinite(threshold)) throw new Error(`pairedDeltaTest: threshold must be finite, got ${threshold}`);
|
|
177
|
+
const exactMinimum = minimumPairsForPairedDeltaTest(options.confidence ?? .95);
|
|
178
|
+
const requestedMinimum = options.minPairs ?? exactMinimum;
|
|
179
|
+
if (!Number.isInteger(requestedMinimum) || requestedMinimum < 1) throw new Error(`pairedDeltaTest: minPairs must be a positive integer, got ${requestedMinimum}`);
|
|
180
|
+
const minimumPairs = Math.max(requestedMinimum, exactMinimum);
|
|
181
|
+
const bootstrap = pairedBootstrap(before, after, options);
|
|
182
|
+
const sufficient = bootstrap.n >= minimumPairs;
|
|
183
|
+
if (bootstrap.gateEligible) return {
|
|
184
|
+
bootstrap,
|
|
185
|
+
method: "bootstrap-ci",
|
|
186
|
+
pValue: null,
|
|
187
|
+
minimumPairs,
|
|
188
|
+
sufficient,
|
|
189
|
+
significant: sufficient && bootstrap.low > threshold
|
|
190
|
+
};
|
|
191
|
+
const exact = pairedSignTest(before.map((value, index) => after[index] - value - threshold), "greater");
|
|
192
|
+
const estimate = options.statistic === "mean" ? bootstrap.mean : bootstrap.median;
|
|
193
|
+
return {
|
|
194
|
+
bootstrap,
|
|
195
|
+
method: "exact-sign",
|
|
196
|
+
pValue: exact.pValue,
|
|
197
|
+
minimumPairs,
|
|
198
|
+
sufficient,
|
|
199
|
+
significant: sufficient && estimate > threshold && exact.pValue <= (1 - bootstrap.confidence) / 2
|
|
200
|
+
};
|
|
201
|
+
}
|
|
202
|
+
//#endregion
|
|
158
203
|
//#region src/json-recovery.ts
|
|
159
204
|
/**
|
|
160
205
|
* Truncation-tolerant JSON recovery — shared by every parser that reads JSON
|
|
@@ -5251,23 +5296,24 @@ function pairHoldout(candidate, baseline, scenarioIds, select) {
|
|
|
5251
5296
|
}
|
|
5252
5297
|
/** Significance of the held-out composite lift: ship only when the paired
|
|
5253
5298
|
* bootstrap CI lower bound on (candidate − baseline) exceeds `deltaThreshold`
|
|
5254
|
-
* (default 0 ⇒ "confidently positive").
|
|
5255
|
-
*
|
|
5256
|
-
*
|
|
5257
|
-
* composite scale. */
|
|
5299
|
+
* (default 0 ⇒ "confidently positive"). At small n, where the percentile
|
|
5300
|
+
* bootstrap is descriptive only, a pre-registered exact sign test carries
|
|
5301
|
+
* the decision. Interpret `deltaThreshold` in the judge's native scale. */
|
|
5258
5302
|
function heldoutSignificance(paired, opts = {}) {
|
|
5259
5303
|
const deltaThreshold = opts.deltaThreshold ?? 0;
|
|
5260
|
-
const minProductiveRuns = opts.minProductiveRuns ?? 3;
|
|
5261
5304
|
const confidence = opts.confidence ?? .95;
|
|
5262
5305
|
const resamples = opts.resamples ?? 2e3;
|
|
5263
5306
|
const seed = opts.seed ?? 1337;
|
|
5264
5307
|
const statistic = opts.statistic ?? "mean";
|
|
5265
|
-
const
|
|
5308
|
+
const decision = pairedDeltaTest(paired.before, paired.after, {
|
|
5266
5309
|
confidence,
|
|
5267
5310
|
resamples,
|
|
5268
5311
|
statistic,
|
|
5269
|
-
seed
|
|
5312
|
+
seed,
|
|
5313
|
+
threshold: deltaThreshold,
|
|
5314
|
+
minPairs: opts.minProductiveRuns
|
|
5270
5315
|
});
|
|
5316
|
+
const bootstrap = decision.bootstrap;
|
|
5271
5317
|
const medianBootstrap = statistic === "median" ? bootstrap : pairedBootstrap(paired.before, paired.after, {
|
|
5272
5318
|
confidence,
|
|
5273
5319
|
resamples,
|
|
@@ -5282,14 +5328,18 @@ function heldoutSignificance(paired, opts = {}) {
|
|
|
5282
5328
|
if (Math.abs(after - before) < 1e-9) ties += 1;
|
|
5283
5329
|
}
|
|
5284
5330
|
const tieFraction = n === 0 ? 0 : ties / n;
|
|
5285
|
-
const fewRuns =
|
|
5331
|
+
const fewRuns = !decision.sufficient;
|
|
5332
|
+
const significant = decision.significant;
|
|
5286
5333
|
return {
|
|
5287
5334
|
paired,
|
|
5288
5335
|
bootstrap,
|
|
5289
5336
|
medianBootstrap,
|
|
5290
5337
|
tieFraction,
|
|
5291
5338
|
n,
|
|
5292
|
-
|
|
5339
|
+
minimumRequired: decision.minimumPairs,
|
|
5340
|
+
decisionMethod: decision.method,
|
|
5341
|
+
pValue: decision.pValue,
|
|
5342
|
+
significant,
|
|
5293
5343
|
fewRuns
|
|
5294
5344
|
};
|
|
5295
5345
|
}
|
|
@@ -5317,10 +5367,17 @@ function dimensionRegressions(candidate, baseline, scenarioIds, criticalDimensio
|
|
|
5317
5367
|
statistic: "median",
|
|
5318
5368
|
seed: opts.seed ?? 1337
|
|
5319
5369
|
});
|
|
5370
|
+
const regression = pairedDeltaTest(paired.after, paired.before, {
|
|
5371
|
+
confidence: opts.confidence ?? .95,
|
|
5372
|
+
resamples: opts.resamples ?? 2e3,
|
|
5373
|
+
statistic: "median",
|
|
5374
|
+
seed: opts.seed ?? 1337,
|
|
5375
|
+
threshold: tolerance
|
|
5376
|
+
});
|
|
5320
5377
|
out.push({
|
|
5321
5378
|
dimension: dim,
|
|
5322
5379
|
bootstrap,
|
|
5323
|
-
regressed:
|
|
5380
|
+
regressed: regression.significant,
|
|
5324
5381
|
tolerance,
|
|
5325
5382
|
n: paired.before.length
|
|
5326
5383
|
});
|
|
@@ -5395,7 +5452,7 @@ function defaultProductionGate(options) {
|
|
|
5395
5452
|
if (sig.fewRuns) requiredUnavailable.add("heldout-significance");
|
|
5396
5453
|
if (!heldoutPass) {
|
|
5397
5454
|
const tieNote = sig.tieFraction >= .4 ? `; ${(sig.tieFraction * 100).toFixed(0)}% tied scenarios` : "";
|
|
5398
|
-
reasons.push(sig.fewRuns ? `held-out: only ${sig.n} paired runs (< ${
|
|
5455
|
+
reasons.push(sig.fewRuns ? `held-out: only ${sig.n} paired runs (< ${sig.minimumRequired}) — too few to claim significance` : `held-out CI.low ${sig.bootstrap.low.toFixed(3)} ≤ threshold ${deltaThreshold} (${heldoutStatistic} Δ ${delta.toFixed(3)}, ${(sig.bootstrap.confidence * 100).toFixed(0)}% CI [${sig.bootstrap.low.toFixed(3)}, ${sig.bootstrap.high.toFixed(3)}]${tieNote})`);
|
|
5399
5456
|
}
|
|
5400
5457
|
}
|
|
5401
5458
|
const dimensionsProvided = options.criticalDimensions !== void 0;
|
|
@@ -5617,7 +5674,7 @@ function heldOutGate(options) {
|
|
|
5617
5674
|
const ci = `${(sig.bootstrap.confidence * 100).toFixed(0)}% CI [${sig.bootstrap.low.toFixed(3)}, ${sig.bootstrap.high.toFixed(3)}]`;
|
|
5618
5675
|
return {
|
|
5619
5676
|
decision: passed ? "ship" : "hold",
|
|
5620
|
-
reasons: passed ? [`held-out mean Δ ${delta.toFixed(3)}, CI.low ${sig.bootstrap.low.toFixed(3)} > ${deltaThreshold} (${ci}, n=${sig.n}${tieNote})`] : [sig.fewRuns ? `held-out: only ${sig.n} paired runs — too few to claim significance` : `held-out mean Δ ${delta.toFixed(3)}, CI.low ${sig.bootstrap.low.toFixed(3)} ≤ ${deltaThreshold} (${ci}, n=${sig.n}${tieNote})`],
|
|
5677
|
+
reasons: passed ? [`held-out mean Δ ${delta.toFixed(3)}, CI.low ${sig.bootstrap.low.toFixed(3)} > ${deltaThreshold} (${ci}, n=${sig.n}${tieNote})`] : [sig.fewRuns ? `held-out: only ${sig.n} paired runs; ${sig.minimumRequired} required — too few to claim significance` : `held-out mean Δ ${delta.toFixed(3)}, CI.low ${sig.bootstrap.low.toFixed(3)} ≤ ${deltaThreshold} (${ci}, n=${sig.n}${tieNote})`],
|
|
5621
5678
|
contributingGates: [{
|
|
5622
5679
|
name: "heldOutGate",
|
|
5623
5680
|
status,
|
|
@@ -5686,6 +5743,30 @@ function powerPreflight(opts) {
|
|
|
5686
5743
|
//#endregion
|
|
5687
5744
|
//#region src/campaign/gates/promotion-policy.ts
|
|
5688
5745
|
/**
|
|
5746
|
+
* Promotion policy over the evidence VECTOR — the substrate's answer to "never
|
|
5747
|
+
* collapse the multi-objective promotion decision into one scalar." A
|
|
5748
|
+
* `defaultProductionGate` is one opinionated composition; this module factors
|
|
5749
|
+
* the decision into two reusable pieces so MANY policies can compete over the
|
|
5750
|
+
* SAME evidence (the quant-desk pattern: one evidence bus, plural strategies):
|
|
5751
|
+
*
|
|
5752
|
+
* buildEvidenceVector(ctx, objectives, opts) -> EvidenceVector // the bus
|
|
5753
|
+
* PromotionPolicy = (ev: EvidenceVector) => GateResult // a strategy
|
|
5754
|
+
* paretoPolicy(ev) // the default strategy
|
|
5755
|
+
* paretoSignificanceGate(options): Gate // bus + policy as a Gate
|
|
5756
|
+
*
|
|
5757
|
+
* The Pareto policy is SYMMETRIC multi-objective: every objective is BOTH a
|
|
5758
|
+
* potential gain source AND a safety floor (unlike `defaultProductionGate`,
|
|
5759
|
+
* where only `composite` can win and `criticalDimensions` are pure floors). A
|
|
5760
|
+
* candidate ships iff it weakly DOMINATES the baseline at the confidence level —
|
|
5761
|
+
* no objective credibly worse (CI floor breach) AND at least one objective
|
|
5762
|
+
* credibly better (CI gain). Insufficient evidence on ANY axis -> need_more_work
|
|
5763
|
+
* (NOT folded into hold: "gather more reps" and "reject" are different actions).
|
|
5764
|
+
*
|
|
5765
|
+
* Cost/latency are NOT CI axes here — `GateContext` carries only an aggregate
|
|
5766
|
+
* per-side cost, no per-cell observation vector to bootstrap. Treat them as hard
|
|
5767
|
+
* constraints (compose with a budget gate via `composeGate`), not faked CIs.
|
|
5768
|
+
*/
|
|
5769
|
+
/**
|
|
5689
5770
|
* The Evidence Bus. For each objective, pair candidate vs baseline by full
|
|
5690
5771
|
* cellId and bootstrap a CI on the good-direction paired delta. Reuses the
|
|
5691
5772
|
* exact `pairHoldout` + `pairedBootstrap` machinery the held-out gate uses, so
|
|
@@ -5693,7 +5774,6 @@ function powerPreflight(opts) {
|
|
|
5693
5774
|
*/
|
|
5694
5775
|
function buildEvidenceVector(ctx, objectives, opts = {}) {
|
|
5695
5776
|
if (objectives.length === 0) throw new Error("buildEvidenceVector: at least 1 objective required");
|
|
5696
|
-
const minProductiveRuns = opts.minProductiveRuns ?? 3;
|
|
5697
5777
|
const confidence = opts.confidence ?? .95;
|
|
5698
5778
|
const resamples = opts.resamples ?? 2e3;
|
|
5699
5779
|
const seed = opts.seed ?? 1337;
|
|
@@ -5708,22 +5788,37 @@ function buildEvidenceVector(ctx, objectives, opts = {}) {
|
|
|
5708
5788
|
select = (s) => s.dimensions[dim];
|
|
5709
5789
|
}
|
|
5710
5790
|
const paired = pairHoldout(ctx.judgeScores, baseline, scenarioIds, select);
|
|
5711
|
-
const
|
|
5791
|
+
const before = obj.direction === "maximize" ? paired.before : paired.after;
|
|
5792
|
+
const after = obj.direction === "maximize" ? paired.after : paired.before;
|
|
5793
|
+
const n = paired.before.length;
|
|
5794
|
+
const floorTolerance = obj.floorTolerance ?? .05 * detectScale([...paired.before, ...paired.after]);
|
|
5795
|
+
const gainThreshold = obj.gainThreshold ?? 0;
|
|
5796
|
+
const improvement = pairedDeltaTest(before, after, {
|
|
5712
5797
|
confidence,
|
|
5713
5798
|
resamples,
|
|
5714
5799
|
statistic: "median",
|
|
5715
|
-
seed
|
|
5800
|
+
seed,
|
|
5801
|
+
threshold: gainThreshold,
|
|
5802
|
+
minPairs: opts.minProductiveRuns
|
|
5716
5803
|
});
|
|
5717
|
-
const
|
|
5718
|
-
|
|
5719
|
-
|
|
5720
|
-
|
|
5804
|
+
const regression = pairedDeltaTest(after, before, {
|
|
5805
|
+
confidence,
|
|
5806
|
+
resamples,
|
|
5807
|
+
statistic: "median",
|
|
5808
|
+
seed,
|
|
5809
|
+
threshold: floorTolerance,
|
|
5810
|
+
minPairs: opts.minProductiveRuns
|
|
5811
|
+
});
|
|
5812
|
+
const bootstrap = improvement.bootstrap;
|
|
5813
|
+
const verdict = !improvement.sufficient ? "few_runs" : regression.significant ? "regressed" : improvement.significant ? "improved" : "flat";
|
|
5721
5814
|
axes.push({
|
|
5722
5815
|
name: obj.name,
|
|
5723
5816
|
source: obj.source,
|
|
5724
5817
|
direction: obj.direction,
|
|
5725
5818
|
bootstrap,
|
|
5726
5819
|
n,
|
|
5820
|
+
minimumRequired: improvement.minimumPairs,
|
|
5821
|
+
decisionMethod: improvement.method,
|
|
5727
5822
|
gainThreshold,
|
|
5728
5823
|
floorTolerance,
|
|
5729
5824
|
verdict
|
|
@@ -7717,6 +7812,6 @@ function skillOptOptimizationMethod(config) {
|
|
|
7717
7812
|
};
|
|
7718
7813
|
}
|
|
7719
7814
|
//#endregion
|
|
7720
|
-
export { SEARCH_LEDGER_FILE_CONTEXT as $, acquireSingleRunLock as A, dominates as At, surfaceContentHash as B,
|
|
7815
|
+
export { SEARCH_LEDGER_FILE_CONTEXT as $, acquireSingleRunLock as A, dominates as At, surfaceContentHash as B, summarizeAgentReceiptIntegrity as Bt, detectScale as C, llmJudge as Ct, runCanaries as D, fileVerdictCache as Dt, pairHoldout as E, contentHash as Et, assertCodeSurfaceIdentity as F, minimumPairsForPairedDeltaTest as Ft, DEFAULT_MUTATION_PRIMITIVES as G, campaignBreakdown as H, JudgeParseError as Ht, assertComponentSurface as I, pairedDeltaTest as It, planCampaignRun as J, buildReflectionPrompt as K, codeSurfaceIdentityMaterial as L, BackendIntegrityError as Lt, compareOptimizationMethods as M, paretoFrontierWithCrowding as Mt, costFromLedgerSummary as N, scalarScore as Nt, composeGate as O, inMemoryVerdictCache as Ot, optimizationTokenUsageFromSummary as P, recoverTruncatedJson as Pt, inMemoryCampaignStorage as Q, componentSurfaceIdentityMaterial as R, assertRealAgentReceipts as Rt, defaultProductionGate as S, hashScenarios as St, heldoutSignificance as T, canonicalJson as Tt, campaignMeanComposite as U, surfaceHash as V, summarizeBackendIntegrity as Vt, compareRankKeys as W, createRunCostLedger as X, runCampaign as Y, fsCampaignStorage as Z, buildEvidenceVector as _, redTeamReport as _t, emitLoopProvenance as a, assertCampaignDesign as at, powerPreflight as b, Dataset as bt, provenanceRecordPath as c, campaignSplitDigest as ct, runImprovementLoop as d, REFERENCE_EQUIVALENCE_INPUT_LIMITS as dt, SearchLedgerConflictError as et, runOptimization as f, REFERENCE_EQUIVALENCE_JUDGE_VERSION as ft, gepaOptimizationMethod as g, redTeamDataset as gt, labelTrustRank as h, DEFAULT_RED_TEAM_CORPUS as ht, canonicalDigest as i, tangleTracesRoot as it, assertOptimizationResult as j, paretoFrontier as jt, externalTextOptimizationMethod as k, crowdingDistance as kt, provenanceSpansPath as l, campaignSplitDigestFromIdentities as lt, isProposedCandidate as m, runReferenceEquivalenceJudge as mt, buildLoopProvenanceRecord as n, SearchLedgerIntegrityError as nt, loopProvenanceArgsFromResult as o, assertCampaignSplitIdentity as ot, runEval as p, createReferenceEquivalenceJudge as pt, parseReflectionResponse as q, campaignMeasurementDigest as r, resolveRunDir as rt, loopProvenanceSpans as s, campaignScenarioIdentity as st, skillOptOptimizationMethod as t, SearchLedgerError as tt, verifyLoopProvenanceRecord as u, openAutoPr as ut, paretoPolicy as v, scoreRedTeamOutput as vt, dimensionRegressions as w, cachedJudge as wt, heldOutGate as x, HoldoutLockedError as xt, paretoSignificanceGate as y, toolNamesForRun as yt, renderSurfaceDiff as z, assertRealBackend as zt };
|
|
7721
7816
|
|
|
7722
|
-
//# sourceMappingURL=skillopt-optimization-method-
|
|
7817
|
+
//# sourceMappingURL=skillopt-optimization-method-vvJ4bMNI.js.map
|