@tangle-network/agent-eval 0.135.0 → 0.135.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +24 -0
- package/dist/analyst/index.js +3 -3
- package/dist/{analyze-runs-DMo3Lb_y.d.ts → analyze-runs-Cda5Xkj1.d.ts} +3 -3
- package/dist/{analyze-runs-DMo3Lb_y.d.ts.map → analyze-runs-Cda5Xkj1.d.ts.map} +1 -1
- package/dist/{analyze-runs-qk8op0tN.js → analyze-runs-jjCmF8pU.js} +14 -7
- package/dist/analyze-runs-jjCmF8pU.js.map +1 -0
- package/dist/{baseline-BaPxoROc.js → baseline-BUeFcgrn.js} +2 -2
- package/dist/{baseline-BaPxoROc.js.map → baseline-BUeFcgrn.js.map} +1 -1
- package/dist/benchmarks/index.d.ts +1 -1
- package/dist/benchmarks/index.js +1 -1
- package/dist/{benchmarks-Dw1Wv_JQ.js → benchmarks-Mtu251Jz.js} +3 -3
- package/dist/{benchmarks-Dw1Wv_JQ.js.map → benchmarks-Mtu251Jz.js.map} +1 -1
- package/dist/builder-eval/index.js +1 -1
- package/dist/campaign/index.d.ts +2 -2
- package/dist/campaign/index.js +2 -2
- package/dist/{campaign-B1c1T0kv.js → campaign-RVIqtJh0.js} +7 -7
- package/dist/{campaign-B1c1T0kv.js.map → campaign-RVIqtJh0.js.map} +1 -1
- package/dist/cli.js +1 -1
- package/dist/{client-BIyh1RCr.d.ts → client-DcvgkaZi.d.ts} +13 -4
- package/dist/client-DcvgkaZi.d.ts.map +1 -0
- package/dist/contract/index.d.ts +3 -3
- package/dist/contract/index.d.ts.map +1 -1
- package/dist/contract/index.js +19 -7
- package/dist/contract/index.js.map +1 -1
- package/dist/{cost-ledger-ZAa_P4r0.js → cost-ledger-DHAjwNj7.js} +6 -2
- package/dist/{cost-ledger-ZAa_P4r0.js.map → cost-ledger-DHAjwNj7.js.map} +1 -1
- package/dist/{default-registry-CFUZyNeZ.js → default-registry-BAhV-lbE.js} +3 -3
- package/dist/{default-registry-CFUZyNeZ.js.map → default-registry-BAhV-lbE.js.map} +1 -1
- package/dist/{eval-campaign-CHFxPTVl.js → eval-campaign-Cc8WZJ6b.js} +3 -3
- package/dist/{eval-campaign-CHFxPTVl.js.map → eval-campaign-Cc8WZJ6b.js.map} +1 -1
- package/dist/fuzz.js +1 -1
- package/dist/hosted/index.d.ts +1 -1
- package/dist/{index-BoJNQR6n.d.ts → index-B4Fjfo5U.d.ts} +93 -15
- package/dist/index-B4Fjfo5U.d.ts.map +1 -0
- package/dist/{index-C21xKtxu.d.ts → index-CQsJcqch.d.ts} +3 -3
- package/dist/{index-C21xKtxu.d.ts.map → index-CQsJcqch.d.ts.map} +1 -1
- package/dist/{index-DSC51roc2.d.ts → index-DSC51roc.d.ts} +1 -1
- package/dist/index-DSC51roc.d.ts.map +1 -0
- package/dist/index.d.ts +41 -10
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +135 -57
- package/dist/index.js.map +1 -1
- package/dist/{llm-client-BNcP4v08.js → llm-client-DHx8pzyJ.js} +2 -2
- package/dist/{llm-client-BNcP4v08.js.map → llm-client-DHx8pzyJ.js.map} +1 -1
- package/dist/matrix/index.d.ts +1 -1
- package/dist/meta-eval/index.d.ts +1 -2
- package/dist/meta-eval/index.d.ts.map +1 -1
- package/dist/meta-eval/index.js +2 -2
- package/dist/multishot/index.d.ts +1 -1
- package/dist/openapi.json +1 -1
- package/dist/{paired-arms-CA_8pN01.js → paired-arms-BbFKrAU-.js} +2 -2
- package/dist/{paired-arms-CA_8pN01.js.map → paired-arms-BbFKrAU-.js.map} +1 -1
- package/dist/pipelines/index.js +2 -2
- package/dist/{release-report-BVZBmRZp.js → release-report-DooPguBc.js} +4 -3
- package/dist/{release-report-BVZBmRZp.js.map → release-report-DooPguBc.js.map} +1 -1
- package/dist/{release-report-CuULWKyk.d.ts → release-report-DpBxGGI1.d.ts} +2 -2
- package/dist/{release-report-CuULWKyk.d.ts.map → release-report-DpBxGGI1.d.ts.map} +1 -1
- package/dist/reporting.d.ts +3 -3
- package/dist/reporting.js +4 -4
- package/dist/{researcher-DVtruQ9U.d.ts → researcher-Doo95b50.d.ts} +2 -2
- package/dist/{researcher-DVtruQ9U.d.ts.map → researcher-Doo95b50.d.ts.map} +1 -1
- package/dist/{reward-hacking-DCdRK9TY.js → reward-hacking-a-kYs0-i.js} +2 -2
- package/dist/{reward-hacking-DCdRK9TY.js.map → reward-hacking-a-kYs0-i.js.map} +1 -1
- package/dist/rl.d.ts +45 -3
- package/dist/rl.d.ts.map +1 -1
- package/dist/rl.js +108 -22
- package/dist/rl.js.map +1 -1
- package/dist/{rubric-predictive-validity-D6Q6n9oq.js → rubric-predictive-validity-BJf-8ejY.js} +2 -2
- package/dist/{rubric-predictive-validity-D6Q6n9oq.js.map → rubric-predictive-validity-BJf-8ejY.js.map} +1 -1
- package/dist/{semantic-concept-judge-Ca--u10C.js → semantic-concept-judge-Btozx3Vc.js} +3 -3
- package/dist/{semantic-concept-judge-Ca--u10C.js.map → semantic-concept-judge-Btozx3Vc.js.map} +1 -1
- package/dist/{server-Dc_lsOYd.js → server-Bz3WQJs6.js} +3 -3
- package/dist/{server-Dc_lsOYd.js.map → server-Bz3WQJs6.js.map} +1 -1
- package/dist/{skillopt-optimization-method-Cl4XPkLC.js → skillopt-optimization-method-0UmPD6aP.js} +342 -56
- package/dist/skillopt-optimization-method-0UmPD6aP.js.map +1 -0
- package/dist/{skillopt-optimization-method-DJ3l4w8W.d.ts → skillopt-optimization-method-CwSYkv35.d.ts} +39 -9
- package/dist/skillopt-optimization-method-CwSYkv35.d.ts.map +1 -0
- package/dist/{statistics-D_4Snl-5.d.ts → statistics-CKOqre5S.d.ts} +329 -3
- package/dist/statistics-CKOqre5S.d.ts.map +1 -0
- package/dist/{statistics-RwRNu2__.js → statistics-CnGCLLqc.js} +315 -2
- package/dist/statistics-CnGCLLqc.js.map +1 -0
- package/dist/{summary-report-BxtossFi.js → summary-report-BEk8OFLs.js} +11 -6
- package/dist/summary-report-BEk8OFLs.js.map +1 -0
- package/dist/{summary-report-DGp0-_XO.d.ts → summary-report-CPMINBqs.d.ts} +182 -7
- package/dist/summary-report-CPMINBqs.d.ts.map +1 -0
- package/dist/wire/index.js +1 -1
- package/package.json +1 -1
- package/dist/analyze-runs-qk8op0tN.js.map +0 -1
- package/dist/client-BIyh1RCr.d.ts.map +0 -1
- package/dist/index-BoJNQR6n.d.ts.map +0 -1
- package/dist/index-DSC51roc2.d.ts.map +0 -1
- package/dist/judge-calibration-DFtEMlde.d.ts +0 -146
- package/dist/judge-calibration-DFtEMlde.d.ts.map +0 -1
- package/dist/skillopt-optimization-method-Cl4XPkLC.js.map +0 -1
- package/dist/skillopt-optimization-method-DJ3l4w8W.d.ts.map +0 -1
- package/dist/statistics-D_4Snl-5.d.ts.map +0 -1
- package/dist/statistics-RwRNu2__.js.map +0 -1
- package/dist/summary-report-BxtossFi.js.map +0 -1
- package/dist/summary-report-DGp0-_XO.d.ts.map +0 -1
|
@@ -1379,6 +1379,39 @@ function wilson(successes, n, confidence = .95) {
|
|
|
1379
1379
|
};
|
|
1380
1380
|
}
|
|
1381
1381
|
/**
|
|
1382
|
+
* Are these per-item outcomes binary (every value exactly 0 or 1)?
|
|
1383
|
+
*
|
|
1384
|
+
* The discriminator a promotion gate needs before choosing a paired statistic.
|
|
1385
|
+
* On binary outcomes the paired delta vector lives in {-1, 0, +1} and is
|
|
1386
|
+
* normally dominated by zeros (both arms solve, or both arms miss, most items),
|
|
1387
|
+
* so its MEDIAN is pinned at exactly 0 no matter how large the real shift in
|
|
1388
|
+
* success rate is — and a bootstrap CI on that median collapses to [0, 0].
|
|
1389
|
+
* A gate keying on `ci.low > threshold` is then structurally unable to see
|
|
1390
|
+
* either a gain or a regression. Detect this shape and switch to the
|
|
1391
|
+
* paired-binary estimators ({@link mcnemar}, {@link pairedRiskDifference})
|
|
1392
|
+
* instead of silently answering "no" forever.
|
|
1393
|
+
*
|
|
1394
|
+
* Empty input is NOT binary: there is no evidence of the outcome's shape, and
|
|
1395
|
+
* defaulting an empty vector into the binary branch would pick a statistic on
|
|
1396
|
+
* no data at all.
|
|
1397
|
+
*
|
|
1398
|
+
* NOT the right discriminator for a gate. It recognises the literal {0, 1}
|
|
1399
|
+
* encoding and nothing else, so a pass/fail dimension emitted on 0-100 — which
|
|
1400
|
+
* judges in this codebase do routinely — reads as non-binary, and a single
|
|
1401
|
+
* partial-credit score in an otherwise pass/fail vector flips it to false while
|
|
1402
|
+
* leaving the median just as blind. Gates want {@link pairedBinaryScale} (any
|
|
1403
|
+
* two-point encoding). This predicate remains for callers that specifically
|
|
1404
|
+
* mean "literally 0/1".
|
|
1405
|
+
*/
|
|
1406
|
+
function isBinaryOutcomeVector(values) {
|
|
1407
|
+
if (values.length === 0) return false;
|
|
1408
|
+
for (let i = 0; i < values.length; i++) {
|
|
1409
|
+
const v = values[i];
|
|
1410
|
+
if (v !== 0 && v !== 1) return false;
|
|
1411
|
+
}
|
|
1412
|
+
return true;
|
|
1413
|
+
}
|
|
1414
|
+
/**
|
|
1382
1415
|
* McNemar's test for paired binary outcomes — the correct significance test for
|
|
1383
1416
|
* "does treatment change the success rate vs control on the SAME items". Only
|
|
1384
1417
|
* discordant pairs (one arm right, the other wrong) carry information; concordant
|
|
@@ -1419,6 +1452,13 @@ function mcnemar(control, treatment) {
|
|
|
1419
1452
|
* the discordant counts, not the independent-samples formula (which overstates
|
|
1420
1453
|
* the interval by ignoring the pairing). Inputs are paired 0/1 (or boolean)
|
|
1421
1454
|
* arrays, control first. Throws on unequal lengths.
|
|
1455
|
+
*
|
|
1456
|
+
* REPORTING ONLY — do NOT decide a promotion on this interval. The CI is a Wald
|
|
1457
|
+
* normal approximation, which badly UNDERCOVERS when only a handful of pairs are
|
|
1458
|
+
* discordant: at n = 3 with b = 2, c = 0 it returns [0.133, 1.000], excluding 0,
|
|
1459
|
+
* while McNemar's exact test on the same data gives p = 0.50. A gate keying on
|
|
1460
|
+
* `lower > 0` would promote noise. Use {@link pairedRiskDifferenceExact}, whose
|
|
1461
|
+
* interval is dual to the exact test by construction, for any decision.
|
|
1422
1462
|
*/
|
|
1423
1463
|
function pairedRiskDifference(control, treatment, confidence = .95) {
|
|
1424
1464
|
if (control.length !== treatment.length) throw new Error(`pairedRiskDifference: unequal sample sizes (${control.length} vs ${treatment.length})`);
|
|
@@ -1454,6 +1494,279 @@ function pairedRiskDifference(control, treatment, confidence = .95) {
|
|
|
1454
1494
|
};
|
|
1455
1495
|
}
|
|
1456
1496
|
/**
|
|
1497
|
+
* Paired risk difference with the EXACT CONDITIONAL interval — the estimator a
|
|
1498
|
+
* promotion gate may decide on.
|
|
1499
|
+
*
|
|
1500
|
+
* Conditional on the number of discordant pairs m = b + c, the treatment-win
|
|
1501
|
+
* count b is Binomial(m, π) with π = P(treatment wins | discordant), and the
|
|
1502
|
+
* risk difference is an exact reparameterisation: RD = (2π − 1)·m/n. So a
|
|
1503
|
+
* Clopper-Pearson exact interval for π maps straight onto RD. This buys the
|
|
1504
|
+
* property the Wald interval in {@link pairedRiskDifference} does not have:
|
|
1505
|
+
*
|
|
1506
|
+
* **`lower > 0` ⟺ McNemar's exact test rejects at α = 1 − confidence.**
|
|
1507
|
+
*
|
|
1508
|
+
* Clopper-Pearson excludes π = 0.5 exactly when the two-sided exact binomial
|
|
1509
|
+
* test of π = 0.5 rejects, and that test IS {@link mcnemar}'s p-value — so the
|
|
1510
|
+
* interval and the test can never disagree, and a gate keyed on `lower` cannot
|
|
1511
|
+
* promote what the exact test refuses. The exact p is returned in the same
|
|
1512
|
+
* object so the two are impossible to compute apart.
|
|
1513
|
+
*
|
|
1514
|
+
* The interval is conservative (exact intervals over-cover; conditioning on m
|
|
1515
|
+
* discards the concordant pairs' information about m itself). That is the
|
|
1516
|
+
* correct direction for a promotion gate: it refuses more often, never less.
|
|
1517
|
+
*
|
|
1518
|
+
* With m = 0 there are no discordant pairs and π is not identified: the result
|
|
1519
|
+
* is the degenerate [0, 0] with p = 1. That is NOT evidence of equivalence —
|
|
1520
|
+
* callers must treat a zero-width interval as "cannot decide", not as "no
|
|
1521
|
+
* difference". Inputs are paired 0/1 (or boolean) arrays, control first.
|
|
1522
|
+
* Throws on unequal lengths.
|
|
1523
|
+
*/
|
|
1524
|
+
function pairedRiskDifferenceExact(control, treatment, confidence = .95) {
|
|
1525
|
+
if (control.length !== treatment.length) throw new Error(`pairedRiskDifferenceExact: unequal sample sizes (${control.length} vs ${treatment.length})`);
|
|
1526
|
+
if (confidence <= 0 || confidence >= 1) throw new Error(`pairedRiskDifferenceExact: confidence must be in (0,1), got ${confidence}`);
|
|
1527
|
+
const n = control.length;
|
|
1528
|
+
if (n === 0) return {
|
|
1529
|
+
n: 0,
|
|
1530
|
+
b: 0,
|
|
1531
|
+
c: 0,
|
|
1532
|
+
nDiscordant: 0,
|
|
1533
|
+
riskDifference: 0,
|
|
1534
|
+
lower: 0,
|
|
1535
|
+
upper: 0,
|
|
1536
|
+
confidence,
|
|
1537
|
+
pValue: 1
|
|
1538
|
+
};
|
|
1539
|
+
let b = 0;
|
|
1540
|
+
let c = 0;
|
|
1541
|
+
for (let i = 0; i < n; i++) {
|
|
1542
|
+
const ctrl = control[i] ? 1 : 0;
|
|
1543
|
+
const treat = treatment[i] ? 1 : 0;
|
|
1544
|
+
if (treat === 1 && ctrl === 0) b++;
|
|
1545
|
+
else if (treat === 0 && ctrl === 1) c++;
|
|
1546
|
+
}
|
|
1547
|
+
const m = b + c;
|
|
1548
|
+
const riskDifference = (b - c) / n;
|
|
1549
|
+
const pValue = binomialSignTwoSided(b, c);
|
|
1550
|
+
if (m === 0) return {
|
|
1551
|
+
n,
|
|
1552
|
+
b,
|
|
1553
|
+
c,
|
|
1554
|
+
nDiscordant: 0,
|
|
1555
|
+
riskDifference: 0,
|
|
1556
|
+
lower: 0,
|
|
1557
|
+
upper: 0,
|
|
1558
|
+
confidence,
|
|
1559
|
+
pValue
|
|
1560
|
+
};
|
|
1561
|
+
const alpha = 1 - confidence;
|
|
1562
|
+
const piLow = b === 0 ? 0 : betaQuantile(alpha / 2, b, m - b + 1);
|
|
1563
|
+
const piHigh = b === m ? 1 : betaQuantile(1 - alpha / 2, b + 1, m - b);
|
|
1564
|
+
const scale = m / n;
|
|
1565
|
+
return {
|
|
1566
|
+
n,
|
|
1567
|
+
b,
|
|
1568
|
+
c,
|
|
1569
|
+
nDiscordant: m,
|
|
1570
|
+
riskDifference,
|
|
1571
|
+
lower: Math.max(-1, (2 * piLow - 1) * scale),
|
|
1572
|
+
upper: Math.min(1, (2 * piHigh - 1) * scale),
|
|
1573
|
+
confidence,
|
|
1574
|
+
pValue
|
|
1575
|
+
};
|
|
1576
|
+
}
|
|
1577
|
+
/** Inverse regularized incomplete beta by bisection on
|
|
1578
|
+
* {@link regularizedIncompleteBeta}, which is monotone increasing in x. 80
|
|
1579
|
+
* halvings of [0,1] resolve to ~8e-25, far past the continued fraction's own
|
|
1580
|
+
* 3e-15 tolerance, so the quantile is as exact as the CDF it inverts. */
|
|
1581
|
+
function betaQuantile(p, a, b) {
|
|
1582
|
+
if (p <= 0) return 0;
|
|
1583
|
+
if (p >= 1) return 1;
|
|
1584
|
+
let lo = 0;
|
|
1585
|
+
let hi = 1;
|
|
1586
|
+
for (let i = 0; i < 80; i++) {
|
|
1587
|
+
const mid = (lo + hi) / 2;
|
|
1588
|
+
if (regularizedIncompleteBeta(mid, a, b) < p) lo = mid;
|
|
1589
|
+
else hi = mid;
|
|
1590
|
+
}
|
|
1591
|
+
return (lo + hi) / 2;
|
|
1592
|
+
}
|
|
1593
|
+
/**
|
|
1594
|
+
* Constrained MLE of q = P(treatment loses) under the hypothesis RD = `delta`.
|
|
1595
|
+
*
|
|
1596
|
+
* Profiling the two concordant cells out of the multinomial leaves
|
|
1597
|
+
* `L(q) = b·log(q+delta) + c·log(q) + e·log(1 − 2q − delta)` with `e = n − b − c`,
|
|
1598
|
+
* whose stationary point is the positive root of
|
|
1599
|
+
* `2n·q² − [(b + c) − delta·(b + 3c + 2e)]·q − c·delta·(1 − delta) = 0`.
|
|
1600
|
+
* At `delta = 0` this returns `(b + c) / 2n`, the familiar null.
|
|
1601
|
+
*/
|
|
1602
|
+
function constrainedLossRate(b, c, n, delta) {
|
|
1603
|
+
const e = n - b - c;
|
|
1604
|
+
const quadratic = 2 * n;
|
|
1605
|
+
const linear = -(b + c - delta * (b + 3 * c + 2 * e));
|
|
1606
|
+
const constant = -c * delta * (1 - delta);
|
|
1607
|
+
const discriminant = linear * linear - 4 * quadratic * constant;
|
|
1608
|
+
const root = discriminant > 0 ? Math.sqrt(discriminant) : 0;
|
|
1609
|
+
const q = (-linear + root) / (2 * quadratic);
|
|
1610
|
+
return Math.min(Math.max(q, Math.max(0, -delta)), Math.max(0, (1 - delta) / 2));
|
|
1611
|
+
}
|
|
1612
|
+
/** Tango's score statistic for H0: RD = `delta`. `Var(b − c) = n·(2q + delta −
|
|
1613
|
+
* delta²)` under that hypothesis, evaluated at the constrained MLE of q. */
|
|
1614
|
+
function tangoScore(b, c, n, delta) {
|
|
1615
|
+
const numerator = b - c - n * delta;
|
|
1616
|
+
const variance = n * (2 * constrainedLossRate(b, c, n, delta) + delta * (1 - delta));
|
|
1617
|
+
if (!(variance > 0)) {
|
|
1618
|
+
if (numerator === 0) return 0;
|
|
1619
|
+
return numerator > 0 ? Number.POSITIVE_INFINITY : Number.NEGATIVE_INFINITY;
|
|
1620
|
+
}
|
|
1621
|
+
return numerator / Math.sqrt(variance);
|
|
1622
|
+
}
|
|
1623
|
+
/**
|
|
1624
|
+
* Paired risk difference with TANGO'S (1998) SCORE INTERVAL — the estimator a
|
|
1625
|
+
* promotion gate may decide on **at a nonzero margin**.
|
|
1626
|
+
*
|
|
1627
|
+
* {@link pairedRiskDifferenceExact} conditions on the observed discordant count
|
|
1628
|
+
* `m = b + c`, builds a Clopper-Pearson interval for the win share among those
|
|
1629
|
+
* `m` pairs, and multiplies by the observed `m/n`. That is exact for testing
|
|
1630
|
+
* RD = 0 — it is dual to McNemar — but it is NOT a confidence interval for the
|
|
1631
|
+
* population risk difference at a nonzero margin, because the sampling
|
|
1632
|
+
* variability of `m/n` itself is discarded. The gap is not academic: with the
|
|
1633
|
+
* production caller's `pairedDeltaThreshold: -0.05`, a process whose true risk
|
|
1634
|
+
* difference sits exactly on that margin clears a nominal-95 % `lower > margin`
|
|
1635
|
+
* check 24.75 % of the time at n = 40 and 43.95 % at n = 76 (2000 replicates
|
|
1636
|
+
* each) when the conditional interval decides.
|
|
1637
|
+
*
|
|
1638
|
+
* Tango's interval inverts the score test of RD = delta, which estimates the
|
|
1639
|
+
* nuisance loss rate under each hypothesised delta instead of fixing it at the
|
|
1640
|
+
* observed value, so `m` contributes its own uncertainty. It is the method
|
|
1641
|
+
* `ratesci::scorepairci` uses for paired risk-difference noninferiority, and it
|
|
1642
|
+
* is not conditional, so it stays valid as the margin moves away from zero.
|
|
1643
|
+
*
|
|
1644
|
+
* The bounds are found by bisecting `tangoScore(delta) = ±z` — the score is
|
|
1645
|
+
* monotone decreasing in delta, so each crossing is unique. Inputs are paired
|
|
1646
|
+
* 0/1 (or boolean) arrays, control first. Throws on unequal lengths.
|
|
1647
|
+
*/
|
|
1648
|
+
function pairedRiskDifferenceScore(control, treatment, confidence = .95) {
|
|
1649
|
+
if (control.length !== treatment.length) throw new Error(`pairedRiskDifferenceScore: unequal sample sizes (${control.length} vs ${treatment.length})`);
|
|
1650
|
+
if (confidence <= 0 || confidence >= 1) throw new Error(`pairedRiskDifferenceScore: confidence must be in (0,1), got ${confidence}`);
|
|
1651
|
+
const n = control.length;
|
|
1652
|
+
if (n === 0) return {
|
|
1653
|
+
n: 0,
|
|
1654
|
+
b: 0,
|
|
1655
|
+
c: 0,
|
|
1656
|
+
nDiscordant: 0,
|
|
1657
|
+
riskDifference: 0,
|
|
1658
|
+
lower: -1,
|
|
1659
|
+
upper: 1,
|
|
1660
|
+
confidence
|
|
1661
|
+
};
|
|
1662
|
+
let b = 0;
|
|
1663
|
+
let c = 0;
|
|
1664
|
+
for (let i = 0; i < n; i++) {
|
|
1665
|
+
const ctrl = control[i] ? 1 : 0;
|
|
1666
|
+
const treat = treatment[i] ? 1 : 0;
|
|
1667
|
+
if (treat === 1 && ctrl === 0) b++;
|
|
1668
|
+
else if (treat === 0 && ctrl === 1) c++;
|
|
1669
|
+
}
|
|
1670
|
+
const riskDifference = (b - c) / n;
|
|
1671
|
+
const z = zQuantile(1 - (1 - confidence) / 2);
|
|
1672
|
+
let lo = -1;
|
|
1673
|
+
let hi = riskDifference;
|
|
1674
|
+
for (let i = 0; i < 200; i++) {
|
|
1675
|
+
const mid = (lo + hi) / 2;
|
|
1676
|
+
if (tangoScore(b, c, n, mid) > z) lo = mid;
|
|
1677
|
+
else hi = mid;
|
|
1678
|
+
}
|
|
1679
|
+
const lower = (lo + hi) / 2;
|
|
1680
|
+
let ulo = riskDifference;
|
|
1681
|
+
let uhi = 1;
|
|
1682
|
+
for (let i = 0; i < 200; i++) {
|
|
1683
|
+
const mid = (ulo + uhi) / 2;
|
|
1684
|
+
if (tangoScore(b, c, n, mid) > -z) ulo = mid;
|
|
1685
|
+
else uhi = mid;
|
|
1686
|
+
}
|
|
1687
|
+
const upper = (ulo + uhi) / 2;
|
|
1688
|
+
return {
|
|
1689
|
+
n,
|
|
1690
|
+
b,
|
|
1691
|
+
c,
|
|
1692
|
+
nDiscordant: b + c,
|
|
1693
|
+
riskDifference,
|
|
1694
|
+
lower: Math.max(-1, lower),
|
|
1695
|
+
upper: Math.min(1, upper),
|
|
1696
|
+
confidence
|
|
1697
|
+
};
|
|
1698
|
+
}
|
|
1699
|
+
/**
|
|
1700
|
+
* The common positive level `s` such that EVERY value across both paired arms is
|
|
1701
|
+
* exactly 0 or `s` — i.e. the outcome is pass/fail, whatever encoding it arrived
|
|
1702
|
+
* in. Returns null when the outcomes are not two-point, when the two arms use
|
|
1703
|
+
* different levels, or when no positive value was observed at all (all-zero
|
|
1704
|
+
* arms: the level is not identified, and there is nothing to decide anyway).
|
|
1705
|
+
*
|
|
1706
|
+
* This is the scale-aware successor to {@link isBinaryOutcomeVector}, which only
|
|
1707
|
+
* recognises literal {0, 1}. Judges in this codebase emit dimensions on 0-100 as
|
|
1708
|
+
* well as [0,1] (see `detectScale` in `campaign/gates/statistical-heldout.ts`),
|
|
1709
|
+
* so a pass/fail dimension routinely arrives as {0, 100} and a {0,1}-only test
|
|
1710
|
+
* silently sends it down the median path that cannot see it. Any positive level
|
|
1711
|
+
* is accepted, not just 1 and 100: for a two-point {0, s} outcome the mean paired
|
|
1712
|
+
* delta is exactly s·(b − c)/n, so the binary estimators apply after dividing by
|
|
1713
|
+
* s and rescaling the result back into the caller's native units.
|
|
1714
|
+
*
|
|
1715
|
+
* Non-finite values ⇒ null: an unusable outcome must not be classified as a
|
|
1716
|
+
* clean pass/fail shape.
|
|
1717
|
+
*/
|
|
1718
|
+
function pairedBinaryScale(before, after) {
|
|
1719
|
+
let level = null;
|
|
1720
|
+
for (const arm of [before, after]) for (let i = 0; i < arm.length; i++) {
|
|
1721
|
+
const v = arm[i];
|
|
1722
|
+
if (!Number.isFinite(v)) return null;
|
|
1723
|
+
if (v === 0) continue;
|
|
1724
|
+
if (v < 0) return null;
|
|
1725
|
+
if (level === null) level = v;
|
|
1726
|
+
else if (v !== level) return null;
|
|
1727
|
+
}
|
|
1728
|
+
return level;
|
|
1729
|
+
}
|
|
1730
|
+
/** Fraction of paired observations whose delta is an exact tie (|after − before|
|
|
1731
|
+
* < 1e-9). Throws on unequal sample sizes; 0 pairs ⇒ 0. */
|
|
1732
|
+
function pairedDeltaTieFraction(before, after) {
|
|
1733
|
+
if (before.length !== after.length) throw new Error(`pairedDeltaTieFraction: unequal sample sizes (${before.length} vs ${after.length})`);
|
|
1734
|
+
const n = before.length;
|
|
1735
|
+
if (n === 0) return 0;
|
|
1736
|
+
let ties = 0;
|
|
1737
|
+
for (let i = 0; i < n; i++) if (Math.abs(after[i] - before[i]) < 1e-9) ties++;
|
|
1738
|
+
return ties / n;
|
|
1739
|
+
}
|
|
1740
|
+
/**
|
|
1741
|
+
* The paired-delta statistic a DECISION is computed on, package-wide.
|
|
1742
|
+
*
|
|
1743
|
+
* The mean paired delta is the estimator that answers the question a promotion
|
|
1744
|
+
* gate asks — "by how much did the candidate move the score" — in the caller's
|
|
1745
|
+
* own units, and it equals the aggregate lift everyone quotes. The MEDIAN
|
|
1746
|
+
* answers a different question and loses the answer to this one in every regime
|
|
1747
|
+
* eval data actually lands in:
|
|
1748
|
+
* - TWO-POINT (pass/fail) outcomes on any encoding: the delta vector lives in
|
|
1749
|
+
* {−s, 0, +s} dominated by zeros, so the median and its whole bootstrap CI
|
|
1750
|
+
* are pinned at exactly 0 however large the shift. (Decide these on
|
|
1751
|
+
* {@link pairedRiskDifferenceExact} instead — same estimand, exact interval.)
|
|
1752
|
+
* - TIE-DOMINATED outcomes: at half the pairs tied the sample median is 0 by
|
|
1753
|
+
* construction, and `ci.low > threshold` then answers "no" forever at a
|
|
1754
|
+
* non-negative threshold and "yes" forever at a negative one.
|
|
1755
|
+
* - LOW-CARDINALITY outcomes, even well below half ties: judge dimensions on
|
|
1756
|
+
* integer 0-100, and block scores like {⅔, 1} from averaging pass/fail
|
|
1757
|
+
* leaves, put the median on a coarse lattice whose bootstrap percentiles
|
|
1758
|
+
* land on atoms. Measured: 26 blocks of 3 pass/fail leaves carrying a real
|
|
1759
|
+
* +12.8pp lift, only 23% of pairs tied, gives a median CI of [0, 0.333] —
|
|
1760
|
+
* lower bound exactly 0, so a gate at threshold 0 refuses a real lift.
|
|
1761
|
+
* That last case is why there is no tie-fraction threshold here: any cutoff on
|
|
1762
|
+
* ties leaves the lattice case open on the other side of it.
|
|
1763
|
+
*
|
|
1764
|
+
* `heldoutSignificance` has defaulted to the mean since #316 for the same
|
|
1765
|
+
* reason. The median remains available per call site for callers who
|
|
1766
|
+
* specifically want outlier robustness and accept the blindness.
|
|
1767
|
+
*/
|
|
1768
|
+
const DECISION_PAIRED_DELTA_STATISTIC = "mean";
|
|
1769
|
+
/**
|
|
1457
1770
|
* Unbiased pass@k for code generation (Chen et al. 2021, "Evaluating Large
|
|
1458
1771
|
* Language Models Trained on Code"). Given `n` independent samples for one
|
|
1459
1772
|
* problem of which `c` pass, the probability that at least one of a random k of
|
|
@@ -1894,6 +2207,6 @@ function mulberry32(seed) {
|
|
|
1894
2207
|
};
|
|
1895
2208
|
}
|
|
1896
2209
|
//#endregion
|
|
1897
|
-
export {
|
|
2210
|
+
export { selfPreference as $, pairedRiskDifference as A, requiredSampleSize as B, mulberry32 as C, pairedCohensDz as D, pairedBootstrap as E, partialCredit as F, wilson as G, weightedComposite as H, passAtK as I, normalCdf as J, studentTCdf as K, pearsonR as L, pairedRiskDifferenceScore as M, pairedSignTest as N, pairedDeltaTieFraction as O, pairedTTest as P, positionalBias as Q, ranks as R, mcnemarRequiredN as S, pairedBinaryScale as T, weightedMean as U, spearmanR as V, wilcoxonSignedRank as W, calibrateJudgeContinuous as X, calibrateJudge as Y, continuousAgreement as Z, interpretCliffs as _, MANN_WHITNEY_EXACT_MAX_WORK as a, mcnemar as b, bonferroni as c, confidenceInterval as d, verbosityBias as et, corpusInterRaterAgreement as f, interRaterReliability as g, holm as h, MANN_WHITNEY_EXACT_MAX_STATES as i, pairedRiskDifferenceExact as j, pairedMde as k, cliffsDelta as l, eProcess as m, DECISION_PAIRED_DELTA_STATISTIC as n, WILCOXON_EXACT_MAX_N as o, corpusInterRaterAgreementFromJudgeScores as p, studentTQuantile as q, DEFAULT_PERMUTATIONS as r, benjaminiHochberg as s, BOOTSTRAP_GATE_MIN_N as t, cohensD as u, isBinaryOutcomeVector as v, normalizeScores as w, mcnemarPower as x, mannWhitneyU as y, requiredPairedSampleSize as z };
|
|
1898
2211
|
|
|
1899
|
-
//# sourceMappingURL=statistics-
|
|
2212
|
+
//# sourceMappingURL=statistics-CnGCLLqc.js.map
|