@tangle-network/agent-eval 0.133.2 → 0.134.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +218 -0
- package/dist/analyst/index.d.ts +11 -35
- package/dist/analyst/index.d.ts.map +1 -1
- package/dist/analyst/index.js +4 -53
- package/dist/analyst/index.js.map +1 -1
- package/dist/{analyze-runs-BmX-h_yn.d.ts → analyze-runs-DMo3Lb_y.d.ts} +4 -4
- package/dist/{analyze-runs-BmX-h_yn.d.ts.map → analyze-runs-DMo3Lb_y.d.ts.map} +1 -1
- package/dist/{analyze-runs-B-afTpCv.js → analyze-runs-qk8op0tN.js} +63 -42
- package/dist/analyze-runs-qk8op0tN.js.map +1 -0
- package/dist/baseline-BaPxoROc.js +149 -0
- package/dist/baseline-BaPxoROc.js.map +1 -0
- package/dist/{baseline-hG3K85h4.d.ts → baseline-D_fT6277.d.ts} +43 -11
- package/dist/baseline-D_fT6277.d.ts.map +1 -0
- package/dist/benchmarks/index.d.ts +1 -1
- package/dist/benchmarks/index.js +1 -1
- package/dist/{benchmarks-CJr1H1_a.js → benchmarks-v5piCeDl.js} +3 -3
- package/dist/{benchmarks-CJr1H1_a.js.map → benchmarks-v5piCeDl.js.map} +1 -1
- package/dist/builder-eval/index.js +1 -1
- package/dist/campaign/index.d.ts +5 -4
- package/dist/campaign/index.js +4 -3
- package/dist/{campaign-BJjn1rhw.js → campaign-DEC_7DLn.js} +12 -6
- package/dist/{campaign-BJjn1rhw.js.map → campaign-DEC_7DLn.js.map} +1 -1
- package/dist/{client-COvaLoQG.d.ts → client-BIyh1RCr.d.ts} +29 -15
- package/dist/client-BIyh1RCr.d.ts.map +1 -0
- package/dist/{client-CYzbdJOZ.js → client-LIuo-KPv.js} +19 -7
- package/dist/client-LIuo-KPv.js.map +1 -0
- package/dist/contract/index.d.ts +12 -11
- package/dist/contract/index.d.ts.map +1 -1
- package/dist/contract/index.js +12 -18
- package/dist/contract/index.js.map +1 -1
- package/dist/{default-registry-Cl3pHo4n.d.ts → default-registry-Brxr728w.d.ts} +4 -268
- package/dist/default-registry-Brxr728w.d.ts.map +1 -0
- package/dist/{default-registry-D3T9XbuY.js → default-registry-IjYs7T8l.js} +4 -61
- package/dist/default-registry-IjYs7T8l.js.map +1 -0
- package/dist/{eval-campaign-DXhpZghy.js → eval-campaign-CvPcvqXC.js} +2 -2
- package/dist/{eval-campaign-DXhpZghy.js.map → eval-campaign-CvPcvqXC.js.map} +1 -1
- package/dist/hosted/index.d.ts +2 -2
- package/dist/hosted/index.d.ts.map +1 -1
- package/dist/hosted/index.js +1 -1
- package/dist/{index-C7Wue8R6.d.ts → index-BoJNQR6n.d.ts} +29 -11
- package/dist/index-BoJNQR6n.d.ts.map +1 -0
- package/dist/{index-BREtv3ZZ.d.ts → index-C21xKtxu.d.ts} +4 -4
- package/dist/{index-BREtv3ZZ.d.ts.map → index-C21xKtxu.d.ts.map} +1 -1
- package/dist/{index-DSC51roc.d.ts → index-DSC51roc2.d.ts} +1 -1
- package/dist/index-DSC51roc2.d.ts.map +1 -0
- package/dist/{index-nhIYz9hn.d.ts → index-DuhJaaiH.d.ts} +68 -7
- package/dist/index-DuhJaaiH.d.ts.map +1 -0
- package/dist/index.d.ts +60 -13
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +134 -27
- package/dist/index.js.map +1 -1
- package/dist/ledger-core/index.d.ts +2 -2
- package/dist/ledger-core/index.js +2 -2
- package/dist/{ledger-core-CPZfcrC2.js → ledger-core-DAKFKRzi.js} +136 -18
- package/dist/ledger-core-DAKFKRzi.js.map +1 -0
- package/dist/matrix/index.d.ts +1 -1
- package/dist/meta-eval/index.d.ts +1 -1
- package/dist/meta-eval/index.js +2 -2
- package/dist/multishot/index.d.ts +2 -2
- package/dist/openapi.json +1 -1
- package/dist/{paired-arms-6XItKzd1.js → paired-arms-CA_8pN01.js} +2 -2
- package/dist/{paired-arms-6XItKzd1.js.map → paired-arms-CA_8pN01.js.map} +1 -1
- package/dist/pipelines/index.d.ts +1 -1
- package/dist/pipelines/index.js +3 -2
- package/dist/pipelines/index.js.map +1 -1
- package/dist/proposal-findings-DCawte-y.js +164 -0
- package/dist/proposal-findings-DCawte-y.js.map +1 -0
- package/dist/{release-report-wuilQkvK.js → release-report-BVZBmRZp.js} +2 -2
- package/dist/{release-report-wuilQkvK.js.map → release-report-BVZBmRZp.js.map} +1 -1
- package/dist/{release-report-CjHWa8Ia.d.ts → release-report-CuULWKyk.d.ts} +2 -2
- package/dist/{release-report-CjHWa8Ia.d.ts.map → release-report-CuULWKyk.d.ts.map} +1 -1
- package/dist/reporting.d.ts +3 -3
- package/dist/reporting.js +4 -4
- package/dist/{researcher-CbSKhK8z.d.ts → researcher-DVtruQ9U.d.ts} +2 -2
- package/dist/{researcher-CbSKhK8z.d.ts.map → researcher-DVtruQ9U.d.ts.map} +1 -1
- package/dist/{reward-hacking-Dl2UBzej.js → reward-hacking-DCdRK9TY.js} +2 -2
- package/dist/{reward-hacking-Dl2UBzej.js.map → reward-hacking-DCdRK9TY.js.map} +1 -1
- package/dist/rl.d.ts +2 -2
- package/dist/rl.d.ts.map +1 -1
- package/dist/rl.js +18 -5
- package/dist/rl.js.map +1 -1
- package/dist/{rubric-predictive-validity-QG7ydk0s.js → rubric-predictive-validity-D6Q6n9oq.js} +2 -2
- package/dist/{rubric-predictive-validity-QG7ydk0s.js.map → rubric-predictive-validity-D6Q6n9oq.js.map} +1 -1
- package/dist/{semantic-concept-judge-BypLt6Fw.js → semantic-concept-judge-C0P1VTXD.js} +2 -3
- package/dist/{semantic-concept-judge-BypLt6Fw.js.map → semantic-concept-judge-C0P1VTXD.js.map} +1 -1
- package/dist/{skill-usage-BaaxFSJR.d.ts → skill-usage-BDQVPIG1.d.ts} +3 -2
- package/dist/skill-usage-BDQVPIG1.d.ts.map +1 -0
- package/dist/{skillopt-optimization-method-CF6a327Q.js → skillopt-optimization-method-BY6vKLJB.js} +169 -53
- package/dist/skillopt-optimization-method-BY6vKLJB.js.map +1 -0
- package/dist/{skillopt-optimization-method-wHF5xsUv.d.ts → skillopt-optimization-method-DJ3l4w8W.d.ts} +20 -18
- package/dist/skillopt-optimization-method-DJ3l4w8W.d.ts.map +1 -0
- package/dist/{statistics-DbvkkDPa.d.ts → statistics-D_4Snl-5.d.ts} +158 -30
- package/dist/statistics-D_4Snl-5.d.ts.map +1 -0
- package/dist/{statistics-DWM_AyLe.js → statistics-RwRNu2__.js} +546 -98
- package/dist/statistics-RwRNu2__.js.map +1 -0
- package/dist/{summary-report-Ci17nIdU.js → summary-report-BxtossFi.js} +3 -3
- package/dist/{summary-report-Ci17nIdU.js.map → summary-report-BxtossFi.js.map} +1 -1
- package/dist/{summary-report-CFnQgNfg.d.ts → summary-report-DGp0-_XO.d.ts} +61 -4
- package/dist/summary-report-DGp0-_XO.d.ts.map +1 -0
- package/dist/{baseline-DcX5hQDv.js → tool-use-metrics-DEGMKycK.js} +2 -114
- package/dist/tool-use-metrics-DEGMKycK.js.map +1 -0
- package/dist/types-DVjczBM9.d.ts +276 -0
- package/dist/types-DVjczBM9.d.ts.map +1 -0
- package/dist/{types-BokuXvOG.d.ts → types-DiWLru6Z.d.ts} +20 -37
- package/dist/types-DiWLru6Z.d.ts.map +1 -0
- package/docs/campaign-proposers.md +5 -0
- package/docs/design/statistics-decisions.md +271 -0
- package/docs/design.md +1 -0
- package/docs/insight-report.md +1 -1
- package/docs/research-report-methodology.md +4 -1
- package/package.json +2 -1
- package/dist/analyze-runs-B-afTpCv.js.map +0 -1
- package/dist/baseline-DcX5hQDv.js.map +0 -1
- package/dist/baseline-hG3K85h4.d.ts.map +0 -1
- package/dist/client-COvaLoQG.d.ts.map +0 -1
- package/dist/client-CYzbdJOZ.js.map +0 -1
- package/dist/default-registry-Cl3pHo4n.d.ts.map +0 -1
- package/dist/default-registry-D3T9XbuY.js.map +0 -1
- package/dist/index-C7Wue8R6.d.ts.map +0 -1
- package/dist/index-DSC51roc.d.ts.map +0 -1
- package/dist/index-nhIYz9hn.d.ts.map +0 -1
- package/dist/ledger-core-CPZfcrC2.js.map +0 -1
- package/dist/run-score-iEEAWiBY.js +0 -41
- package/dist/run-score-iEEAWiBY.js.map +0 -1
- package/dist/skill-usage-BaaxFSJR.d.ts.map +0 -1
- package/dist/skillopt-optimization-method-CF6a327Q.js.map +0 -1
- package/dist/skillopt-optimization-method-wHF5xsUv.d.ts.map +0 -1
- package/dist/statistics-DWM_AyLe.js.map +0 -1
- package/dist/statistics-DbvkkDPa.d.ts.map +0 -1
- package/dist/summary-report-CFnQgNfg.d.ts.map +0 -1
- package/dist/types-BokuXvOG.d.ts.map +0 -1
package/dist/{skillopt-optimization-method-CF6a327Q.js → skillopt-optimization-method-BY6vKLJB.js}
RENAMED
|
@@ -1,12 +1,12 @@
|
|
|
1
1
|
import { i as JudgeError, s as ValidationError, t as AgentEvalError } from "./errors-8YnH8WlF.js";
|
|
2
2
|
import { c as costForTokenPricing, i as CostLedger, t as CostAccountingIncompleteError } from "./cost-ledger-BrJxbrMy.js";
|
|
3
3
|
import { f as maximumChargeForLlmRequest, l as costReceiptFromLlm, m as stripFencedJson, u as costReceiptFromLlmError } from "./llm-client-ClPW-dWB.js";
|
|
4
|
-
import {
|
|
4
|
+
import { a as clamp01, o as combineAbortSignals, t as assertProposalFindings } from "./proposal-findings-DCawte-y.js";
|
|
5
5
|
import { n as mapConcurrent } from "./concurrency-MUjT7VjM.js";
|
|
6
|
-
import {
|
|
6
|
+
import { C as pairedBootstrap, D as pairedSignTest, I as weightedComposite, u as confidenceInterval } from "./statistics-RwRNu2__.js";
|
|
7
7
|
import { t as DEFAULT_REDACTION_RULES } from "./redact-7Aq1ukl-.js";
|
|
8
|
-
import { a as campaignCellExecutionEvidence, l as projectCampaignCellQuality, t as detectRewardHacking } from "./reward-hacking-
|
|
9
|
-
import { a as
|
|
8
|
+
import { a as campaignCellExecutionEvidence, l as projectCampaignCellQuality, t as detectRewardHacking } from "./reward-hacking-DCdRK9TY.js";
|
|
9
|
+
import { a as appendLedgerLine, h as tryAcquireAtomicFileLock, m as probeAtomicFileLock, o as tryWithLedgerFileLock } from "./ledger-core-DAKFKRzi.js";
|
|
10
10
|
import { createRequire } from "node:module";
|
|
11
11
|
import { z } from "zod";
|
|
12
12
|
import { appendFileSync, existsSync, readFileSync, writeFileSync } from "node:fs";
|
|
@@ -155,6 +155,51 @@ function assertBackendReport(report, opts) {
|
|
|
155
155
|
return report;
|
|
156
156
|
}
|
|
157
157
|
//#endregion
|
|
158
|
+
//#region src/paired-delta-test.ts
|
|
159
|
+
/** Smallest all-positive sample that can clear a one-sided exact sign test. */
|
|
160
|
+
function minimumPairsForPairedDeltaTest(confidence = .95) {
|
|
161
|
+
if (!Number.isFinite(confidence) || confidence <= 0 || confidence >= 1) throw new Error(`minimumPairsForPairedDeltaTest: confidence must be in (0,1), got ${confidence}`);
|
|
162
|
+
const oneSidedAlpha = (1 - confidence) / 2;
|
|
163
|
+
return Math.ceil(Math.log2(1 / oneSidedAlpha));
|
|
164
|
+
}
|
|
165
|
+
/**
|
|
166
|
+
* Tests whether a paired candidate-minus-baseline delta clears a threshold.
|
|
167
|
+
*
|
|
168
|
+
* At 20 or more pairs, the percentile bootstrap lower bound carries the
|
|
169
|
+
* decision. Below that point the interval is descriptive only, so the function
|
|
170
|
+
* switches to a pre-registered one-sided exact sign test. The exact path is
|
|
171
|
+
* deliberately conservative: it requires both a point estimate above the
|
|
172
|
+
* threshold and enough consistently positive paired differences.
|
|
173
|
+
*/
|
|
174
|
+
function pairedDeltaTest(before, after, options = {}) {
|
|
175
|
+
const threshold = options.threshold ?? 0;
|
|
176
|
+
if (!Number.isFinite(threshold)) throw new Error(`pairedDeltaTest: threshold must be finite, got ${threshold}`);
|
|
177
|
+
const exactMinimum = minimumPairsForPairedDeltaTest(options.confidence ?? .95);
|
|
178
|
+
const requestedMinimum = options.minPairs ?? exactMinimum;
|
|
179
|
+
if (!Number.isInteger(requestedMinimum) || requestedMinimum < 1) throw new Error(`pairedDeltaTest: minPairs must be a positive integer, got ${requestedMinimum}`);
|
|
180
|
+
const minimumPairs = Math.max(requestedMinimum, exactMinimum);
|
|
181
|
+
const bootstrap = pairedBootstrap(before, after, options);
|
|
182
|
+
const sufficient = bootstrap.n >= minimumPairs;
|
|
183
|
+
if (bootstrap.gateEligible) return {
|
|
184
|
+
bootstrap,
|
|
185
|
+
method: "bootstrap-ci",
|
|
186
|
+
pValue: null,
|
|
187
|
+
minimumPairs,
|
|
188
|
+
sufficient,
|
|
189
|
+
significant: sufficient && bootstrap.low > threshold
|
|
190
|
+
};
|
|
191
|
+
const exact = pairedSignTest(before.map((value, index) => after[index] - value - threshold), "greater");
|
|
192
|
+
const estimate = options.statistic === "mean" ? bootstrap.mean : bootstrap.median;
|
|
193
|
+
return {
|
|
194
|
+
bootstrap,
|
|
195
|
+
method: "exact-sign",
|
|
196
|
+
pValue: exact.pValue,
|
|
197
|
+
minimumPairs,
|
|
198
|
+
sufficient,
|
|
199
|
+
significant: sufficient && estimate > threshold && exact.pValue <= (1 - bootstrap.confidence) / 2
|
|
200
|
+
};
|
|
201
|
+
}
|
|
202
|
+
//#endregion
|
|
158
203
|
//#region src/json-recovery.ts
|
|
159
204
|
/**
|
|
160
205
|
* Truncation-tolerant JSON recovery — shared by every parser that reads JSON
|
|
@@ -3835,7 +3880,7 @@ function describeExternalScenario(scenario, label, maxChars, describe) {
|
|
|
3835
3880
|
assertJsonValue(data, `${label} scenario '${scenario.id}'`);
|
|
3836
3881
|
const serializedChars = JSON.stringify(data).length;
|
|
3837
3882
|
if (serializedChars > maxChars) throw new Error(`${label} scenario '${scenario.id}' exceeds maxEvidenceChars (${serializedChars} > ${maxChars})`);
|
|
3838
|
-
return deepFreeze({
|
|
3883
|
+
return deepFreeze$1({
|
|
3839
3884
|
id: scenario.id,
|
|
3840
3885
|
data
|
|
3841
3886
|
});
|
|
@@ -3886,9 +3931,9 @@ async function scoreOneScenario(args) {
|
|
|
3886
3931
|
function cloneExternalTextCandidate$1(candidate) {
|
|
3887
3932
|
return typeof candidate === "string" ? candidate : { ...candidate };
|
|
3888
3933
|
}
|
|
3889
|
-
function deepFreeze(value) {
|
|
3934
|
+
function deepFreeze$1(value) {
|
|
3890
3935
|
if (value && typeof value === "object") {
|
|
3891
|
-
for (const child of Object.values(value)) deepFreeze(child);
|
|
3936
|
+
for (const child of Object.values(value)) deepFreeze$1(child);
|
|
3892
3937
|
Object.freeze(value);
|
|
3893
3938
|
}
|
|
3894
3939
|
return value;
|
|
@@ -5251,23 +5296,24 @@ function pairHoldout(candidate, baseline, scenarioIds, select) {
|
|
|
5251
5296
|
}
|
|
5252
5297
|
/** Significance of the held-out composite lift: ship only when the paired
|
|
5253
5298
|
* bootstrap CI lower bound on (candidate − baseline) exceeds `deltaThreshold`
|
|
5254
|
-
* (default 0 ⇒ "confidently positive").
|
|
5255
|
-
*
|
|
5256
|
-
*
|
|
5257
|
-
* composite scale. */
|
|
5299
|
+
* (default 0 ⇒ "confidently positive"). At small n, where the percentile
|
|
5300
|
+
* bootstrap is descriptive only, a pre-registered exact sign test carries
|
|
5301
|
+
* the decision. Interpret `deltaThreshold` in the judge's native scale. */
|
|
5258
5302
|
function heldoutSignificance(paired, opts = {}) {
|
|
5259
5303
|
const deltaThreshold = opts.deltaThreshold ?? 0;
|
|
5260
|
-
const minProductiveRuns = opts.minProductiveRuns ?? 3;
|
|
5261
5304
|
const confidence = opts.confidence ?? .95;
|
|
5262
5305
|
const resamples = opts.resamples ?? 2e3;
|
|
5263
5306
|
const seed = opts.seed ?? 1337;
|
|
5264
5307
|
const statistic = opts.statistic ?? "mean";
|
|
5265
|
-
const
|
|
5308
|
+
const decision = pairedDeltaTest(paired.before, paired.after, {
|
|
5266
5309
|
confidence,
|
|
5267
5310
|
resamples,
|
|
5268
5311
|
statistic,
|
|
5269
|
-
seed
|
|
5312
|
+
seed,
|
|
5313
|
+
threshold: deltaThreshold,
|
|
5314
|
+
minPairs: opts.minProductiveRuns
|
|
5270
5315
|
});
|
|
5316
|
+
const bootstrap = decision.bootstrap;
|
|
5271
5317
|
const medianBootstrap = statistic === "median" ? bootstrap : pairedBootstrap(paired.before, paired.after, {
|
|
5272
5318
|
confidence,
|
|
5273
5319
|
resamples,
|
|
@@ -5282,14 +5328,18 @@ function heldoutSignificance(paired, opts = {}) {
|
|
|
5282
5328
|
if (Math.abs(after - before) < 1e-9) ties += 1;
|
|
5283
5329
|
}
|
|
5284
5330
|
const tieFraction = n === 0 ? 0 : ties / n;
|
|
5285
|
-
const fewRuns =
|
|
5331
|
+
const fewRuns = !decision.sufficient;
|
|
5332
|
+
const significant = decision.significant;
|
|
5286
5333
|
return {
|
|
5287
5334
|
paired,
|
|
5288
5335
|
bootstrap,
|
|
5289
5336
|
medianBootstrap,
|
|
5290
5337
|
tieFraction,
|
|
5291
5338
|
n,
|
|
5292
|
-
|
|
5339
|
+
minimumRequired: decision.minimumPairs,
|
|
5340
|
+
decisionMethod: decision.method,
|
|
5341
|
+
pValue: decision.pValue,
|
|
5342
|
+
significant,
|
|
5293
5343
|
fewRuns
|
|
5294
5344
|
};
|
|
5295
5345
|
}
|
|
@@ -5317,10 +5367,17 @@ function dimensionRegressions(candidate, baseline, scenarioIds, criticalDimensio
|
|
|
5317
5367
|
statistic: "median",
|
|
5318
5368
|
seed: opts.seed ?? 1337
|
|
5319
5369
|
});
|
|
5370
|
+
const regression = pairedDeltaTest(paired.after, paired.before, {
|
|
5371
|
+
confidence: opts.confidence ?? .95,
|
|
5372
|
+
resamples: opts.resamples ?? 2e3,
|
|
5373
|
+
statistic: "median",
|
|
5374
|
+
seed: opts.seed ?? 1337,
|
|
5375
|
+
threshold: tolerance
|
|
5376
|
+
});
|
|
5320
5377
|
out.push({
|
|
5321
5378
|
dimension: dim,
|
|
5322
5379
|
bootstrap,
|
|
5323
|
-
regressed:
|
|
5380
|
+
regressed: regression.significant,
|
|
5324
5381
|
tolerance,
|
|
5325
5382
|
n: paired.before.length
|
|
5326
5383
|
});
|
|
@@ -5395,7 +5452,7 @@ function defaultProductionGate(options) {
|
|
|
5395
5452
|
if (sig.fewRuns) requiredUnavailable.add("heldout-significance");
|
|
5396
5453
|
if (!heldoutPass) {
|
|
5397
5454
|
const tieNote = sig.tieFraction >= .4 ? `; ${(sig.tieFraction * 100).toFixed(0)}% tied scenarios` : "";
|
|
5398
|
-
reasons.push(sig.fewRuns ? `held-out: only ${sig.n} paired runs (< ${
|
|
5455
|
+
reasons.push(sig.fewRuns ? `held-out: only ${sig.n} paired runs (< ${sig.minimumRequired}) — too few to claim significance` : `held-out CI.low ${sig.bootstrap.low.toFixed(3)} ≤ threshold ${deltaThreshold} (${heldoutStatistic} Δ ${delta.toFixed(3)}, ${(sig.bootstrap.confidence * 100).toFixed(0)}% CI [${sig.bootstrap.low.toFixed(3)}, ${sig.bootstrap.high.toFixed(3)}]${tieNote})`);
|
|
5399
5456
|
}
|
|
5400
5457
|
}
|
|
5401
5458
|
const dimensionsProvided = options.criticalDimensions !== void 0;
|
|
@@ -5617,7 +5674,7 @@ function heldOutGate(options) {
|
|
|
5617
5674
|
const ci = `${(sig.bootstrap.confidence * 100).toFixed(0)}% CI [${sig.bootstrap.low.toFixed(3)}, ${sig.bootstrap.high.toFixed(3)}]`;
|
|
5618
5675
|
return {
|
|
5619
5676
|
decision: passed ? "ship" : "hold",
|
|
5620
|
-
reasons: passed ? [`held-out mean Δ ${delta.toFixed(3)}, CI.low ${sig.bootstrap.low.toFixed(3)} > ${deltaThreshold} (${ci}, n=${sig.n}${tieNote})`] : [sig.fewRuns ? `held-out: only ${sig.n} paired runs — too few to claim significance` : `held-out mean Δ ${delta.toFixed(3)}, CI.low ${sig.bootstrap.low.toFixed(3)} ≤ ${deltaThreshold} (${ci}, n=${sig.n}${tieNote})`],
|
|
5677
|
+
reasons: passed ? [`held-out mean Δ ${delta.toFixed(3)}, CI.low ${sig.bootstrap.low.toFixed(3)} > ${deltaThreshold} (${ci}, n=${sig.n}${tieNote})`] : [sig.fewRuns ? `held-out: only ${sig.n} paired runs; ${sig.minimumRequired} required — too few to claim significance` : `held-out mean Δ ${delta.toFixed(3)}, CI.low ${sig.bootstrap.low.toFixed(3)} ≤ ${deltaThreshold} (${ci}, n=${sig.n}${tieNote})`],
|
|
5621
5678
|
contributingGates: [{
|
|
5622
5679
|
name: "heldOutGate",
|
|
5623
5680
|
status,
|
|
@@ -5686,6 +5743,30 @@ function powerPreflight(opts) {
|
|
|
5686
5743
|
//#endregion
|
|
5687
5744
|
//#region src/campaign/gates/promotion-policy.ts
|
|
5688
5745
|
/**
|
|
5746
|
+
* Promotion policy over the evidence VECTOR — the substrate's answer to "never
|
|
5747
|
+
* collapse the multi-objective promotion decision into one scalar." A
|
|
5748
|
+
* `defaultProductionGate` is one opinionated composition; this module factors
|
|
5749
|
+
* the decision into two reusable pieces so MANY policies can compete over the
|
|
5750
|
+
* SAME evidence (the quant-desk pattern: one evidence bus, plural strategies):
|
|
5751
|
+
*
|
|
5752
|
+
* buildEvidenceVector(ctx, objectives, opts) -> EvidenceVector // the bus
|
|
5753
|
+
* PromotionPolicy = (ev: EvidenceVector) => GateResult // a strategy
|
|
5754
|
+
* paretoPolicy(ev) // the default strategy
|
|
5755
|
+
* paretoSignificanceGate(options): Gate // bus + policy as a Gate
|
|
5756
|
+
*
|
|
5757
|
+
* The Pareto policy is SYMMETRIC multi-objective: every objective is BOTH a
|
|
5758
|
+
* potential gain source AND a safety floor (unlike `defaultProductionGate`,
|
|
5759
|
+
* where only `composite` can win and `criticalDimensions` are pure floors). A
|
|
5760
|
+
* candidate ships iff it weakly DOMINATES the baseline at the confidence level —
|
|
5761
|
+
* no objective credibly worse (CI floor breach) AND at least one objective
|
|
5762
|
+
* credibly better (CI gain). Insufficient evidence on ANY axis -> need_more_work
|
|
5763
|
+
* (NOT folded into hold: "gather more reps" and "reject" are different actions).
|
|
5764
|
+
*
|
|
5765
|
+
* Cost/latency are NOT CI axes here — `GateContext` carries only an aggregate
|
|
5766
|
+
* per-side cost, no per-cell observation vector to bootstrap. Treat them as hard
|
|
5767
|
+
* constraints (compose with a budget gate via `composeGate`), not faked CIs.
|
|
5768
|
+
*/
|
|
5769
|
+
/**
|
|
5689
5770
|
* The Evidence Bus. For each objective, pair candidate vs baseline by full
|
|
5690
5771
|
* cellId and bootstrap a CI on the good-direction paired delta. Reuses the
|
|
5691
5772
|
* exact `pairHoldout` + `pairedBootstrap` machinery the held-out gate uses, so
|
|
@@ -5693,7 +5774,6 @@ function powerPreflight(opts) {
|
|
|
5693
5774
|
*/
|
|
5694
5775
|
function buildEvidenceVector(ctx, objectives, opts = {}) {
|
|
5695
5776
|
if (objectives.length === 0) throw new Error("buildEvidenceVector: at least 1 objective required");
|
|
5696
|
-
const minProductiveRuns = opts.minProductiveRuns ?? 3;
|
|
5697
5777
|
const confidence = opts.confidence ?? .95;
|
|
5698
5778
|
const resamples = opts.resamples ?? 2e3;
|
|
5699
5779
|
const seed = opts.seed ?? 1337;
|
|
@@ -5708,22 +5788,37 @@ function buildEvidenceVector(ctx, objectives, opts = {}) {
|
|
|
5708
5788
|
select = (s) => s.dimensions[dim];
|
|
5709
5789
|
}
|
|
5710
5790
|
const paired = pairHoldout(ctx.judgeScores, baseline, scenarioIds, select);
|
|
5711
|
-
const
|
|
5791
|
+
const before = obj.direction === "maximize" ? paired.before : paired.after;
|
|
5792
|
+
const after = obj.direction === "maximize" ? paired.after : paired.before;
|
|
5793
|
+
const n = paired.before.length;
|
|
5794
|
+
const floorTolerance = obj.floorTolerance ?? .05 * detectScale([...paired.before, ...paired.after]);
|
|
5795
|
+
const gainThreshold = obj.gainThreshold ?? 0;
|
|
5796
|
+
const improvement = pairedDeltaTest(before, after, {
|
|
5712
5797
|
confidence,
|
|
5713
5798
|
resamples,
|
|
5714
5799
|
statistic: "median",
|
|
5715
|
-
seed
|
|
5800
|
+
seed,
|
|
5801
|
+
threshold: gainThreshold,
|
|
5802
|
+
minPairs: opts.minProductiveRuns
|
|
5716
5803
|
});
|
|
5717
|
-
const
|
|
5718
|
-
|
|
5719
|
-
|
|
5720
|
-
|
|
5804
|
+
const regression = pairedDeltaTest(after, before, {
|
|
5805
|
+
confidence,
|
|
5806
|
+
resamples,
|
|
5807
|
+
statistic: "median",
|
|
5808
|
+
seed,
|
|
5809
|
+
threshold: floorTolerance,
|
|
5810
|
+
minPairs: opts.minProductiveRuns
|
|
5811
|
+
});
|
|
5812
|
+
const bootstrap = improvement.bootstrap;
|
|
5813
|
+
const verdict = !improvement.sufficient ? "few_runs" : regression.significant ? "regressed" : improvement.significant ? "improved" : "flat";
|
|
5721
5814
|
axes.push({
|
|
5722
5815
|
name: obj.name,
|
|
5723
5816
|
source: obj.source,
|
|
5724
5817
|
direction: obj.direction,
|
|
5725
5818
|
bootstrap,
|
|
5726
5819
|
n,
|
|
5820
|
+
minimumRequired: improvement.minimumPairs,
|
|
5821
|
+
decisionMethod: improvement.method,
|
|
5727
5822
|
gainThreshold,
|
|
5728
5823
|
floorTolerance,
|
|
5729
5824
|
verdict
|
|
@@ -6413,6 +6508,8 @@ async function runOptimization(opts) {
|
|
|
6413
6508
|
const candidateConcurrency = opts.candidateConcurrency ?? 1;
|
|
6414
6509
|
if (typeof opts.runDir !== "string" || opts.runDir.trim().length === 0) throw new Error("runOptimization: runDir is required and must be a non-empty string");
|
|
6415
6510
|
if (!Number.isInteger(candidateConcurrency) || candidateConcurrency < 1) throw new Error("runOptimization: candidateConcurrency must be a positive integer");
|
|
6511
|
+
const initialFindings = immutableProposalSnapshot(assertProposalFindings(opts.findings ?? [], "runOptimization initial proposal findings"), "initial findings");
|
|
6512
|
+
const baselineSurface = immutableProposalSnapshot(opts.baselineSurface, "baseline surface");
|
|
6416
6513
|
opts.runDir = resolveRunDir(opts.runDir, opts.repo);
|
|
6417
6514
|
const storage = opts.storage ?? fsCampaignStorage();
|
|
6418
6515
|
const costLedger = opts.costLedger ?? createRunCostLedger({
|
|
@@ -6425,7 +6522,7 @@ async function runOptimization(opts) {
|
|
|
6425
6522
|
const premeasuredBaseline = opts.premeasuredBaseline;
|
|
6426
6523
|
const baselineCampaign = premeasuredBaseline ? validatedPremeasuredBaseline({
|
|
6427
6524
|
input: premeasuredBaseline,
|
|
6428
|
-
baselineSurface
|
|
6525
|
+
baselineSurface,
|
|
6429
6526
|
scenarios: opts.scenarios,
|
|
6430
6527
|
reps,
|
|
6431
6528
|
seed: opts.seed ?? 42
|
|
@@ -6433,7 +6530,7 @@ async function runOptimization(opts) {
|
|
|
6433
6530
|
...opts,
|
|
6434
6531
|
costLedger,
|
|
6435
6532
|
costPhase: "search.baseline",
|
|
6436
|
-
dispatch: (scenario, ctx) => opts.dispatchWithSurface(
|
|
6533
|
+
dispatch: (scenario, ctx) => opts.dispatchWithSurface(baselineSurface, scenario, ctx),
|
|
6437
6534
|
runDir: `${opts.runDir}/baseline`
|
|
6438
6535
|
});
|
|
6439
6536
|
const baselineCoverage = campaignCoverage(baselineCampaign.cells, opts.scenarios, reps, requireJudgeScore);
|
|
@@ -6443,10 +6540,10 @@ async function runOptimization(opts) {
|
|
|
6443
6540
|
}
|
|
6444
6541
|
const generations = [];
|
|
6445
6542
|
const history = [];
|
|
6446
|
-
let currentFindings =
|
|
6543
|
+
let currentFindings = initialFindings;
|
|
6447
6544
|
const selectionRankKey = opts.selectionRankKey ?? ((campaign) => [campaignMeanComposite(campaign)]);
|
|
6448
|
-
let winnerSurface =
|
|
6449
|
-
let winnerSurfaceHash = surfaceHash(
|
|
6545
|
+
let winnerSurface = baselineSurface;
|
|
6546
|
+
let winnerSurfaceHash = surfaceHash(baselineSurface);
|
|
6450
6547
|
let winnerComposite = campaignMeanComposite(baselineCampaign);
|
|
6451
6548
|
let winnerRankKey = selectionRankKey(baselineCampaign);
|
|
6452
6549
|
assertFiniteRankKey(winnerRankKey, "selectionRankKey for baseline");
|
|
@@ -6454,7 +6551,7 @@ async function runOptimization(opts) {
|
|
|
6454
6551
|
let winnerOutcome = baselineOutcome;
|
|
6455
6552
|
let winnerLabel;
|
|
6456
6553
|
let winnerRationale;
|
|
6457
|
-
const scored = [toParetoParent(
|
|
6554
|
+
const scored = [toParetoParent(baselineSurface, winnerSurfaceHash, baselineCampaign, -1)];
|
|
6458
6555
|
if (opts.analyzeGeneration && opts.maxGenerations > 0 && baselineCampaign.cells.length > 0) {
|
|
6459
6556
|
const fresh = await opts.analyzeGeneration({
|
|
6460
6557
|
generation: -1,
|
|
@@ -6468,31 +6565,34 @@ async function runOptimization(opts) {
|
|
|
6468
6565
|
costLedger,
|
|
6469
6566
|
costPhase: "analysis.baseline"
|
|
6470
6567
|
});
|
|
6471
|
-
if (Array.isArray(fresh))
|
|
6568
|
+
if (!Array.isArray(fresh)) throw new TypeError("runOptimization: analyzeGeneration must return an array");
|
|
6569
|
+
currentFindings = immutableProposalSnapshot(assertProposalFindings(fresh, "runOptimization baseline analysis findings"), "baseline analysis findings");
|
|
6472
6570
|
}
|
|
6473
6571
|
for (let gen = 0; gen < opts.maxGenerations; gen++) {
|
|
6474
|
-
|
|
6572
|
+
const proposalHistory = immutableProposalSnapshot(history, "history");
|
|
6573
|
+
if (proposer.decide?.({ history: proposalHistory }).stop) break;
|
|
6475
6574
|
const paretoParents = computeParetoFrontier(scored);
|
|
6476
6575
|
const parentSurfaceHash = winnerSurfaceHash;
|
|
6477
6576
|
const parentComposite = winnerComposite;
|
|
6478
|
-
const
|
|
6479
|
-
currentSurface: winnerSurface,
|
|
6480
|
-
history,
|
|
6481
|
-
findings: currentFindings,
|
|
6577
|
+
const proposalContext = Object.freeze({
|
|
6578
|
+
currentSurface: immutableProposalSnapshot(winnerSurface, "current surface"),
|
|
6579
|
+
history: proposalHistory,
|
|
6580
|
+
findings: immutableProposalSnapshot(assertProposalFindings(currentFindings, "runOptimization proposal findings"), "findings"),
|
|
6482
6581
|
populationSize: opts.populationSize,
|
|
6483
6582
|
generation: gen,
|
|
6484
|
-
signal: new AbortController().signal,
|
|
6485
|
-
baselineOutcome,
|
|
6486
|
-
incumbentOutcome: winnerOutcome,
|
|
6487
|
-
report: opts.report,
|
|
6488
|
-
dataset: opts.labeledStore && opts.labeledStore !== "off" ? opts.labeledStore : void 0,
|
|
6583
|
+
signal: opts.signal ?? new AbortController().signal,
|
|
6584
|
+
baselineOutcome: immutableProposalSnapshot(baselineOutcome, "baseline outcome"),
|
|
6585
|
+
incumbentOutcome: immutableProposalSnapshot(winnerOutcome, "incumbent outcome"),
|
|
6489
6586
|
maxImprovementShots: opts.maxImprovementShots,
|
|
6490
|
-
paretoParents,
|
|
6587
|
+
paretoParents: immutableProposalSnapshot(paretoParents, "Pareto parents"),
|
|
6491
6588
|
costLedger,
|
|
6492
6589
|
costPhase: "search.proposal"
|
|
6493
6590
|
});
|
|
6494
|
-
|
|
6495
|
-
|
|
6591
|
+
const proposed = await proposer.propose(proposalContext);
|
|
6592
|
+
if (!Array.isArray(proposed)) throw new TypeError("runOptimization: proposer must return an array");
|
|
6593
|
+
const proposalSnapshot = immutableProposalSnapshot(proposed, "candidate outputs");
|
|
6594
|
+
if (proposalSnapshot.length === 0) break;
|
|
6595
|
+
const surfaceResults = await mapConcurrent(proposalSnapshot.map((p) => isProposedCandidate(p) ? p : {
|
|
6496
6596
|
surface: p,
|
|
6497
6597
|
label: "",
|
|
6498
6598
|
rationale: ""
|
|
@@ -6589,11 +6689,13 @@ async function runOptimization(opts) {
|
|
|
6589
6689
|
costLedger,
|
|
6590
6690
|
costPhase: "analysis.generation"
|
|
6591
6691
|
});
|
|
6592
|
-
if (Array.isArray(fresh))
|
|
6692
|
+
if (!Array.isArray(fresh)) throw new TypeError("runOptimization: analyzeGeneration must return an array");
|
|
6693
|
+
currentFindings = immutableProposalSnapshot(assertProposalFindings(fresh, "runOptimization generation analysis findings"), "generation analysis findings");
|
|
6593
6694
|
}
|
|
6594
6695
|
}
|
|
6595
6696
|
return {
|
|
6596
6697
|
generations,
|
|
6698
|
+
baselineSurface,
|
|
6597
6699
|
winnerSurface,
|
|
6598
6700
|
winnerSurfaceHash,
|
|
6599
6701
|
winnerLabel,
|
|
@@ -6603,6 +6705,19 @@ async function runOptimization(opts) {
|
|
|
6603
6705
|
cost: costLedger.summary()
|
|
6604
6706
|
};
|
|
6605
6707
|
}
|
|
6708
|
+
function immutableProposalSnapshot(value, label) {
|
|
6709
|
+
try {
|
|
6710
|
+
return deepFreeze(structuredClone(value));
|
|
6711
|
+
} catch (cause) {
|
|
6712
|
+
throw new TypeError(`runOptimization: proposal ${label} must contain snapshot-safe data`, { cause });
|
|
6713
|
+
}
|
|
6714
|
+
}
|
|
6715
|
+
function deepFreeze(value, seen = /* @__PURE__ */ new WeakSet()) {
|
|
6716
|
+
if (typeof value !== "object" || value === null || seen.has(value)) return value;
|
|
6717
|
+
seen.add(value);
|
|
6718
|
+
for (const descriptor of Object.values(Object.getOwnPropertyDescriptors(value))) if ("value" in descriptor) deepFreeze(descriptor.value, seen);
|
|
6719
|
+
return Object.freeze(value);
|
|
6720
|
+
}
|
|
6606
6721
|
function validatedPremeasuredBaseline(args) {
|
|
6607
6722
|
const { input } = args;
|
|
6608
6723
|
if (input.surfaceHash !== surfaceHash(args.baselineSurface)) throw new Error("runOptimization: premeasured baseline surface hash does not match baselineSurface");
|
|
@@ -6712,7 +6827,8 @@ async function runImprovementLoop(opts) {
|
|
|
6712
6827
|
dispatchTimeoutMs,
|
|
6713
6828
|
costLedger
|
|
6714
6829
|
});
|
|
6715
|
-
const
|
|
6830
|
+
const baselineSurface = optimization.baselineSurface;
|
|
6831
|
+
const winnerIsBaseline = optimization.winnerSurfaceHash === surfaceHash(baselineSurface);
|
|
6716
6832
|
const holdoutDeferred = (opts.holdout ?? "measured") === "deferred";
|
|
6717
6833
|
const baselineOnHoldout = holdoutDeferred ? await runCampaign({
|
|
6718
6834
|
...opts,
|
|
@@ -6732,7 +6848,7 @@ async function runImprovementLoop(opts) {
|
|
|
6732
6848
|
costPhase: "holdout.baseline",
|
|
6733
6849
|
dispatchTimeoutMs,
|
|
6734
6850
|
scenarios: opts.holdoutScenarios,
|
|
6735
|
-
dispatch: (scenario, ctx) => opts.dispatchWithSurface(
|
|
6851
|
+
dispatch: (scenario, ctx) => opts.dispatchWithSurface(baselineSurface, scenario, ctx),
|
|
6736
6852
|
runDir: `${opts.runDir}/holdout-baseline`
|
|
6737
6853
|
});
|
|
6738
6854
|
const winnerOnHoldout = winnerIsBaseline || holdoutDeferred ? baselineOnHoldout : await runCampaign({
|
|
@@ -6772,7 +6888,7 @@ async function runImprovementLoop(opts) {
|
|
|
6772
6888
|
let neutralizedOnHoldout;
|
|
6773
6889
|
let neutralizedSurface;
|
|
6774
6890
|
if (opts.neutralize && !winnerIsBaseline && !holdoutDeferred) {
|
|
6775
|
-
const surface = opts.neutralize(optimization.winnerSurface,
|
|
6891
|
+
const surface = opts.neutralize(optimization.winnerSurface, baselineSurface);
|
|
6776
6892
|
neutralizedSurface = surface;
|
|
6777
6893
|
neutralizedOnHoldout = await runCampaign({
|
|
6778
6894
|
...opts,
|
|
@@ -6825,7 +6941,7 @@ async function runImprovementLoop(opts) {
|
|
|
6825
6941
|
costPhase: "promotion.gate",
|
|
6826
6942
|
signal: new AbortController().signal
|
|
6827
6943
|
});
|
|
6828
|
-
const promotedDiff = optimization.winnerSurfaceHash === surfaceHash(
|
|
6944
|
+
const promotedDiff = optimization.winnerSurfaceHash === surfaceHash(baselineSurface) ? "" : renderSurfaceDiff(optimization.winnerSurface, baselineSurface);
|
|
6829
6945
|
let prResult;
|
|
6830
6946
|
if (opts.autoOnPromote === "pr" && gateResult.decision === "ship") prResult = openAutoPr({
|
|
6831
6947
|
result: winnerOnHoldout,
|
|
@@ -7717,6 +7833,6 @@ function skillOptOptimizationMethod(config) {
|
|
|
7717
7833
|
};
|
|
7718
7834
|
}
|
|
7719
7835
|
//#endregion
|
|
7720
|
-
export { SEARCH_LEDGER_FILE_CONTEXT as $, acquireSingleRunLock as A, dominates as At, surfaceContentHash as B,
|
|
7836
|
+
export { SEARCH_LEDGER_FILE_CONTEXT as $, acquireSingleRunLock as A, dominates as At, surfaceContentHash as B, summarizeAgentReceiptIntegrity as Bt, detectScale as C, llmJudge as Ct, runCanaries as D, fileVerdictCache as Dt, pairHoldout as E, contentHash as Et, assertCodeSurfaceIdentity as F, minimumPairsForPairedDeltaTest as Ft, DEFAULT_MUTATION_PRIMITIVES as G, campaignBreakdown as H, JudgeParseError as Ht, assertComponentSurface as I, pairedDeltaTest as It, planCampaignRun as J, buildReflectionPrompt as K, codeSurfaceIdentityMaterial as L, BackendIntegrityError as Lt, compareOptimizationMethods as M, paretoFrontierWithCrowding as Mt, costFromLedgerSummary as N, scalarScore as Nt, composeGate as O, inMemoryVerdictCache as Ot, optimizationTokenUsageFromSummary as P, recoverTruncatedJson as Pt, inMemoryCampaignStorage as Q, componentSurfaceIdentityMaterial as R, assertRealAgentReceipts as Rt, defaultProductionGate as S, hashScenarios as St, heldoutSignificance as T, canonicalJson as Tt, campaignMeanComposite as U, surfaceHash as V, summarizeBackendIntegrity as Vt, compareRankKeys as W, createRunCostLedger as X, runCampaign as Y, fsCampaignStorage as Z, buildEvidenceVector as _, redTeamReport as _t, emitLoopProvenance as a, assertCampaignDesign as at, powerPreflight as b, Dataset as bt, provenanceRecordPath as c, campaignSplitDigest as ct, runImprovementLoop as d, REFERENCE_EQUIVALENCE_INPUT_LIMITS as dt, SearchLedgerConflictError as et, runOptimization as f, REFERENCE_EQUIVALENCE_JUDGE_VERSION as ft, gepaOptimizationMethod as g, redTeamDataset as gt, labelTrustRank as h, DEFAULT_RED_TEAM_CORPUS as ht, canonicalDigest as i, tangleTracesRoot as it, assertOptimizationResult as j, paretoFrontier as jt, externalTextOptimizationMethod as k, crowdingDistance as kt, provenanceSpansPath as l, campaignSplitDigestFromIdentities as lt, isProposedCandidate as m, runReferenceEquivalenceJudge as mt, buildLoopProvenanceRecord as n, SearchLedgerIntegrityError as nt, loopProvenanceArgsFromResult as o, assertCampaignSplitIdentity as ot, runEval as p, createReferenceEquivalenceJudge as pt, parseReflectionResponse as q, campaignMeasurementDigest as r, resolveRunDir as rt, loopProvenanceSpans as s, campaignScenarioIdentity as st, skillOptOptimizationMethod as t, SearchLedgerError as tt, verifyLoopProvenanceRecord as u, openAutoPr as ut, paretoPolicy as v, scoreRedTeamOutput as vt, dimensionRegressions as w, cachedJudge as wt, heldOutGate as x, HoldoutLockedError as xt, paretoSignificanceGate as y, toolNamesForRun as yt, renderSurfaceDiff as z, assertRealBackend as zt };
|
|
7721
7837
|
|
|
7722
|
-
//# sourceMappingURL=skillopt-optimization-method-
|
|
7838
|
+
//# sourceMappingURL=skillopt-optimization-method-BY6vKLJB.js.map
|