abtestresult-mcp 1.0.8 → 1.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +12 -3
- package/dist/index.js +446 -75
- package/package.json +1 -1
package/README.md
CHANGED
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
|
|
3
3
|
Statistical tools for A/B testing, available as a [Model Context Protocol](https://modelcontextprotocol.io) (MCP) server. Powered by [abtestresult.com](https://abtestresult.com).
|
|
4
4
|
|
|
5
|
-
Give any AI assistant (Claude, Cursor, VS Code, etc.) the ability to analyze A/B tests, calculate sample sizes, run Bayesian analysis, and more. All calculations run locally on your machine — no API keys, your test data never leaves your device.
|
|
5
|
+
Give any AI assistant (Claude, Codex, Cursor, VS Code, etc.) the ability to analyze A/B tests, calculate sample sizes, run Bayesian analysis, and more. All calculations run locally on your machine — no API keys, your test data never leaves your device.
|
|
6
6
|
|
|
7
7
|
This package ships readable source. It is source-available for local personal use only and is not licensed for commercial use.
|
|
8
8
|
|
|
@@ -23,7 +23,8 @@ Add to your MCP configuration:
|
|
|
23
23
|
|
|
24
24
|
**Where to add this:**
|
|
25
25
|
|
|
26
|
-
- **Claude Desktop:**
|
|
26
|
+
- **Claude Desktop:** `~/Library/Application Support/Claude/claude_desktop_config.json`
|
|
27
|
+
- **Codex:** `~/.codex/config.toml`
|
|
27
28
|
- **Claude Code:** `.mcp.json` in your project root
|
|
28
29
|
- **Cursor:** `.cursor/mcp.json`
|
|
29
30
|
- **VS Code:** `.vscode/settings.json` (use `mcp.servers` instead of `mcpServers`)
|
|
@@ -35,7 +36,7 @@ That's it. No API keys, no authentication. The server starts automatically when
|
|
|
35
36
|
| Tool | Description |
|
|
36
37
|
|------|-------------|
|
|
37
38
|
| `analyze_ab_test` | Z-test for rate metrics (conversion rate, CTR, etc.) |
|
|
38
|
-
| `analyze_ab_test_average` | T-test for continuous metrics (revenue, time on page, etc.) |
|
|
39
|
+
| `analyze_ab_test_average` | Welch's T-test for continuous metrics (revenue, time on page, etc.); no equal-variance assumption |
|
|
39
40
|
| `calculate_sample_size` | Required sample size for rate or average metrics |
|
|
40
41
|
| `calculate_mde` | Minimum detectable effect given a fixed sample size |
|
|
41
42
|
| `bayesian_ab_test` | Bayesian analysis with probability to be best & expected loss |
|
|
@@ -43,6 +44,14 @@ That's it. No API keys, no authentication. The server starts automatically when
|
|
|
43
44
|
| `paired_test` | Paired T-test or Wilcoxon signed-rank for before/after data |
|
|
44
45
|
| `survey_sample_size` | Cochran's formula with finite population correction |
|
|
45
46
|
|
|
47
|
+
## Shareable result links
|
|
48
|
+
|
|
49
|
+
Every analysis tool (except `paired_test`) returns a `view_url` in its response — a
|
|
50
|
+
deeplink to [abtestresult.com](https://abtestresult.com) that pre-fills the matching
|
|
51
|
+
calculator with your inputs and auto-runs it. Open it to inspect the result visually
|
|
52
|
+
(charts, confidence intervals, recommendations) or share it with your team. No data is
|
|
53
|
+
stored server-side: the inputs are encoded directly into the URL.
|
|
54
|
+
|
|
46
55
|
## Usage Examples
|
|
47
56
|
|
|
48
57
|
Once connected, just ask your AI assistant in natural language:
|
package/dist/index.js
CHANGED
|
@@ -33997,10 +33997,24 @@ function normalInterval(confidence, mean, std) {
|
|
|
33997
33997
|
const z2 = normalPpf(1 - alpha / 2);
|
|
33998
33998
|
return [mean - z2 * std, mean + z2 * std];
|
|
33999
33999
|
}
|
|
34000
|
+
var LARGE_DF = 1e4;
|
|
34000
34001
|
function tCdf(x, df) {
|
|
34002
|
+
if (df > LARGE_DF) {
|
|
34003
|
+
const z2 = x * (1 - 1 / (4 * df)) / Math.sqrt(1 + x * x / (2 * df));
|
|
34004
|
+
return normalCdf(z2);
|
|
34005
|
+
}
|
|
34001
34006
|
return import_jstat.default.studentt.cdf(x, df);
|
|
34002
34007
|
}
|
|
34003
34008
|
function tPpf(p, df) {
|
|
34009
|
+
if (df > LARGE_DF) {
|
|
34010
|
+
const z2 = normalPpf(p);
|
|
34011
|
+
const z22 = z2 * z2;
|
|
34012
|
+
const g1 = (z22 + 1) * z2 / 4;
|
|
34013
|
+
const g2 = ((5 * z22 + 16) * z22 + 3) * z2 / 96;
|
|
34014
|
+
const g3 = (((3 * z22 + 19) * z22 + 17) * z22 - 15) * z2 / 384;
|
|
34015
|
+
const g4 = ((((79 * z22 + 776) * z22 + 1482) * z22 - 1920) * z22 - 945) * z2 / 92160;
|
|
34016
|
+
return z2 + g1 / df + g2 / df ** 2 + g3 / df ** 3 + g4 / df ** 4;
|
|
34017
|
+
}
|
|
34004
34018
|
return import_jstat.default.studentt.inv(p, df);
|
|
34005
34019
|
}
|
|
34006
34020
|
function chiSquareCdf(x, df) {
|
|
@@ -34064,12 +34078,23 @@ function zTestForProportions(controlUsers, controlConversions, variantUsers, var
|
|
|
34064
34078
|
};
|
|
34065
34079
|
}
|
|
34066
34080
|
function tTestForAverages(controlUsers, controlMean, controlStdDev, variantUsers, variantMean, variantStdDev, confidenceLevelAdj, sided, niMarginPercent = 0) {
|
|
34081
|
+
if (!(controlUsers >= 2) || !(variantUsers >= 2)) {
|
|
34082
|
+
throw new RangeError("Welch T-test needs at least 2 users in each group");
|
|
34083
|
+
}
|
|
34084
|
+
if (!(controlStdDev >= 0) || !(variantStdDev >= 0) || !Number.isFinite(controlStdDev) || !Number.isFinite(variantStdDev)) {
|
|
34085
|
+
throw new RangeError("Standard deviations must be finite and non-negative");
|
|
34086
|
+
}
|
|
34087
|
+
if (controlStdDev === 0 && variantStdDev === 0) {
|
|
34088
|
+
throw new RangeError("At least one group must have a standard deviation above 0");
|
|
34089
|
+
}
|
|
34067
34090
|
const niMarginDecimal = niMarginPercent / 100;
|
|
34091
|
+
const controlVarMean = controlStdDev ** 2 / controlUsers;
|
|
34092
|
+
const variantVarMean = variantStdDev ** 2 / variantUsers;
|
|
34093
|
+
const sem = Math.sqrt(controlVarMean + variantVarMean);
|
|
34094
|
+
const df = (controlVarMean + variantVarMean) ** 2 / (controlVarMean ** 2 / (controlUsers - 1) + variantVarMean ** 2 / (variantUsers - 1));
|
|
34068
34095
|
const pooledStd = Math.sqrt(
|
|
34069
34096
|
((controlUsers - 1) * controlStdDev ** 2 + (variantUsers - 1) * variantStdDev ** 2) / (controlUsers + variantUsers - 2)
|
|
34070
34097
|
);
|
|
34071
|
-
const sem = pooledStd * Math.sqrt(1 / controlUsers + 1 / variantUsers);
|
|
34072
|
-
const df = controlUsers + variantUsers - 2;
|
|
34073
34098
|
let tScore = (variantMean - controlMean + niMarginDecimal * controlMean) / sem;
|
|
34074
34099
|
if (sided === 2) {
|
|
34075
34100
|
tScore = Math.abs(tScore);
|
|
@@ -34081,7 +34106,8 @@ function tTestForAverages(controlUsers, controlMean, controlStdDev, variantUsers
|
|
|
34081
34106
|
pValue = 1 - tCdf(tScore, df);
|
|
34082
34107
|
}
|
|
34083
34108
|
const alphaT = 1 - confidenceLevelAdj;
|
|
34084
|
-
const
|
|
34109
|
+
const criticalT = (dof) => sided === 1 ? tPpf(1 - alphaT, dof) : tPpf(1 - alphaT / 2, dof);
|
|
34110
|
+
const tCritical = criticalT(df);
|
|
34085
34111
|
const isSignificant = tScore > tCritical;
|
|
34086
34112
|
const delta = variantMean - controlMean;
|
|
34087
34113
|
const deltaRelative = controlMean !== 0 ? delta / controlMean : 0;
|
|
@@ -34089,18 +34115,21 @@ function tTestForAverages(controlUsers, controlMean, controlStdDev, variantUsers
|
|
|
34089
34115
|
const meanCriticalLess = controlMean - tCritical * sem;
|
|
34090
34116
|
const deltaCritical = tCritical * sem;
|
|
34091
34117
|
const deltaCriticalLess = -tCritical * sem;
|
|
34118
|
+
const controlSe = controlStdDev / Math.sqrt(controlUsers);
|
|
34119
|
+
const variantSe = variantStdDev / Math.sqrt(variantUsers);
|
|
34120
|
+
const controlT = criticalT(controlUsers - 1);
|
|
34121
|
+
const variantT = criticalT(variantUsers - 1);
|
|
34092
34122
|
const controlCi = [
|
|
34093
|
-
controlMean -
|
|
34094
|
-
controlMean +
|
|
34123
|
+
controlMean - controlT * controlSe,
|
|
34124
|
+
controlMean + controlT * controlSe
|
|
34095
34125
|
];
|
|
34096
34126
|
const variationCi = [
|
|
34097
|
-
variantMean -
|
|
34098
|
-
variantMean +
|
|
34127
|
+
variantMean - variantT * variantSe,
|
|
34128
|
+
variantMean + variantT * variantSe
|
|
34099
34129
|
];
|
|
34100
|
-
const diffSe = Math.sqrt(controlStdDev ** 2 / controlUsers + variantStdDev ** 2 / variantUsers);
|
|
34101
34130
|
const diffCi = [
|
|
34102
|
-
|
|
34103
|
-
|
|
34131
|
+
delta - tCritical * sem,
|
|
34132
|
+
delta + tCritical * sem
|
|
34104
34133
|
];
|
|
34105
34134
|
return {
|
|
34106
34135
|
tScore,
|
|
@@ -34112,6 +34141,8 @@ function tTestForAverages(controlUsers, controlMean, controlStdDev, variantUsers
|
|
|
34112
34141
|
pooledStd,
|
|
34113
34142
|
sem,
|
|
34114
34143
|
df,
|
|
34144
|
+
controlSe,
|
|
34145
|
+
variantSe,
|
|
34115
34146
|
controlCi,
|
|
34116
34147
|
variationCi,
|
|
34117
34148
|
diffCi,
|
|
@@ -34123,16 +34154,18 @@ function tTestForAverages(controlUsers, controlMean, controlStdDev, variantUsers
|
|
|
34123
34154
|
variantMean
|
|
34124
34155
|
};
|
|
34125
34156
|
}
|
|
34126
|
-
function checkSRM(controlUsers, variantUsers) {
|
|
34157
|
+
function checkSRM(controlUsers, variantUsers, expectedControlShare = 0.5) {
|
|
34127
34158
|
const total = controlUsers + variantUsers;
|
|
34128
|
-
const
|
|
34129
|
-
const
|
|
34159
|
+
const expectedControl = total * expectedControlShare;
|
|
34160
|
+
const expectedVariant = total - expectedControl;
|
|
34161
|
+
const chiSquare = Math.pow(controlUsers - expectedControl, 2) / expectedControl + Math.pow(variantUsers - expectedVariant, 2) / expectedVariant;
|
|
34130
34162
|
const pValue = 1 - chiSquareCdf(chiSquare, 1);
|
|
34131
34163
|
const hasMismatch = pValue < 0.01;
|
|
34132
34164
|
return {
|
|
34133
34165
|
chiSquare,
|
|
34134
34166
|
pValue,
|
|
34135
|
-
hasMismatch
|
|
34167
|
+
hasMismatch,
|
|
34168
|
+
expectedControlShare
|
|
34136
34169
|
};
|
|
34137
34170
|
}
|
|
34138
34171
|
function calculateSampleSizeRate(baseline, mde, confidenceLevel, power, sided, numVariants = 1, successMetrics = 1, guardrailMetrics = 0, niMarginPercent = 0, dailyTraffic = null) {
|
|
@@ -34201,15 +34234,30 @@ function calculateSampleSizeAverage(baseline, mde, stdDev, confidenceLevel, powe
|
|
|
34201
34234
|
mdeAbsolute: p2 - baseline
|
|
34202
34235
|
};
|
|
34203
34236
|
}
|
|
34237
|
+
function allocateSampleSize(samplesEachBalanced, numVariants, controlShare) {
|
|
34238
|
+
const variantShare = (1 - controlShare) / numVariants;
|
|
34239
|
+
const total = samplesEachBalanced / 2 * (1 / controlShare + 1 / variantShare);
|
|
34240
|
+
const control = Math.ceil(total * controlShare - 1e-9);
|
|
34241
|
+
const perVariant = Math.ceil(total * variantShare - 1e-9);
|
|
34242
|
+
const allocatedTotal = control + perVariant * numVariants;
|
|
34243
|
+
return {
|
|
34244
|
+
control,
|
|
34245
|
+
perVariant,
|
|
34246
|
+
total: allocatedTotal,
|
|
34247
|
+
multiplier: allocatedTotal / (samplesEachBalanced * (numVariants + 1))
|
|
34248
|
+
};
|
|
34249
|
+
}
|
|
34204
34250
|
function calculateSurveySampleSize(population, marginOfError, confidenceLevel) {
|
|
34205
34251
|
const estimatedProportion = 0.5;
|
|
34206
34252
|
const zScore = Math.abs(normalPpf((1 - confidenceLevel) / 2));
|
|
34207
34253
|
const sampleSize = population * Math.pow(zScore, 2) * estimatedProportion * (1 - estimatedProportion) / (Math.pow(marginOfError, 2) * (population - 1) + Math.pow(zScore, 2) * estimatedProportion * (1 - estimatedProportion));
|
|
34208
34254
|
return Math.ceil(sampleSize);
|
|
34209
34255
|
}
|
|
34210
|
-
function estimateMDE(traffic, stdDev, power, confidenceLevel, numVariants = 1, successMetrics = 1, guardrailMetrics = 0) {
|
|
34256
|
+
function estimateMDE(traffic, stdDev, power, confidenceLevel, numVariants = 1, successMetrics = 1, guardrailMetrics = 0, controlShare) {
|
|
34211
34257
|
const nGroups = numVariants + 1;
|
|
34212
|
-
const
|
|
34258
|
+
const cShare = controlShare ?? 1 / nGroups;
|
|
34259
|
+
const nControl = traffic * cShare;
|
|
34260
|
+
const nVariant = traffic * (1 - cShare) / numVariants;
|
|
34213
34261
|
let alpha = 1 - confidenceLevel;
|
|
34214
34262
|
const comparisons = numVariants + (successMetrics - 1);
|
|
34215
34263
|
if (comparisons > 0) {
|
|
@@ -34222,9 +34270,68 @@ function estimateMDE(traffic, stdDev, power, confidenceLevel, numVariants = 1, s
|
|
|
34222
34270
|
}
|
|
34223
34271
|
const zAlpha = normalPpf(1 - alpha / 2);
|
|
34224
34272
|
const zPower = normalPpf(adjustedPower);
|
|
34225
|
-
const mde = (zPower + zAlpha) * Math.sqrt(
|
|
34273
|
+
const mde = (zPower + zAlpha) * stdDev * Math.sqrt(1 / nControl + 1 / nVariant);
|
|
34226
34274
|
return mde;
|
|
34227
34275
|
}
|
|
34276
|
+
function twoProportionPower(nControl, nVariant, baselineRate, lift, alpha, sided = 2) {
|
|
34277
|
+
const p1 = baselineRate;
|
|
34278
|
+
const p2 = baselineRate + lift;
|
|
34279
|
+
const pooled = (nControl * p1 + nVariant * p2) / (nControl + nVariant);
|
|
34280
|
+
const se0 = Math.sqrt(pooled * (1 - pooled) * (1 / nControl + 1 / nVariant));
|
|
34281
|
+
const se1 = Math.sqrt(p1 * (1 - p1) / nControl + p2 * (1 - p2) / nVariant);
|
|
34282
|
+
if (sided === 1) {
|
|
34283
|
+
return normalCdf((lift - normalPpf(1 - alpha) * se0) / se1);
|
|
34284
|
+
}
|
|
34285
|
+
const z2 = normalPpf(1 - alpha / 2);
|
|
34286
|
+
return normalCdf((lift - z2 * se0) / se1) + normalCdf((-lift - z2 * se0) / se1);
|
|
34287
|
+
}
|
|
34288
|
+
function estimateMDERate({
|
|
34289
|
+
traffic,
|
|
34290
|
+
baselineRate,
|
|
34291
|
+
power,
|
|
34292
|
+
confidenceLevel,
|
|
34293
|
+
numVariants = 1,
|
|
34294
|
+
successMetrics = 1,
|
|
34295
|
+
guardrailMetrics = 0,
|
|
34296
|
+
controlShare,
|
|
34297
|
+
sided = 2
|
|
34298
|
+
}) {
|
|
34299
|
+
const p = baselineRate;
|
|
34300
|
+
if (!(traffic > 0) || !(p > 0 && p < 1)) return NaN;
|
|
34301
|
+
const cShare = controlShare ?? 1 / (numVariants + 1);
|
|
34302
|
+
const nControl = traffic * cShare;
|
|
34303
|
+
const nVariant = traffic * (1 - cShare) / numVariants;
|
|
34304
|
+
let alpha = 1 - confidenceLevel;
|
|
34305
|
+
const comparisons = numVariants + (successMetrics - 1);
|
|
34306
|
+
if (comparisons > 0) {
|
|
34307
|
+
alpha = sidakCorrection(alpha, comparisons);
|
|
34308
|
+
}
|
|
34309
|
+
let adjustedPower = power;
|
|
34310
|
+
if (guardrailMetrics > 0) {
|
|
34311
|
+
adjustedPower = 1 - (1 - power) / (guardrailMetrics + 1);
|
|
34312
|
+
}
|
|
34313
|
+
const powerAt = (d) => twoProportionPower(nControl, nVariant, p, d, alpha, sided);
|
|
34314
|
+
const maxLift = 1 - p - 1e-12;
|
|
34315
|
+
const steps = 200;
|
|
34316
|
+
const minLift = maxLift * 1e-6;
|
|
34317
|
+
let lo = 0;
|
|
34318
|
+
let hi = NaN;
|
|
34319
|
+
for (let i = 0; i <= steps; i++) {
|
|
34320
|
+
const d = minLift * Math.pow(maxLift / minLift, i / steps);
|
|
34321
|
+
if (powerAt(d) >= adjustedPower) {
|
|
34322
|
+
hi = d;
|
|
34323
|
+
break;
|
|
34324
|
+
}
|
|
34325
|
+
lo = d;
|
|
34326
|
+
}
|
|
34327
|
+
if (Number.isNaN(hi)) return Infinity;
|
|
34328
|
+
for (let i = 0; i < 100; i++) {
|
|
34329
|
+
const mid = (lo + hi) / 2;
|
|
34330
|
+
if (powerAt(mid) >= adjustedPower) hi = mid;
|
|
34331
|
+
else lo = mid;
|
|
34332
|
+
}
|
|
34333
|
+
return hi;
|
|
34334
|
+
}
|
|
34228
34335
|
function pairedTTest(sample1, sample2, confidenceLevel) {
|
|
34229
34336
|
if (sample1.length !== sample2.length) {
|
|
34230
34337
|
throw new Error("Samples must have the same length");
|
|
@@ -34420,6 +34527,31 @@ function betaPosterior(priorAlpha, priorBeta, successes, trials) {
|
|
|
34420
34527
|
beta: priorBeta + (trials - successes)
|
|
34421
34528
|
};
|
|
34422
34529
|
}
|
|
34530
|
+
function computeVariantLifts(samples, tail) {
|
|
34531
|
+
const EPSILON = 1e-10;
|
|
34532
|
+
const control = samples[0];
|
|
34533
|
+
const percentileCi = (sorted) => sorted.length > 0 ? [sorted[Math.floor(tail * sorted.length)], sorted[Math.floor((1 - tail) * sorted.length)]] : [0, 0];
|
|
34534
|
+
const mean = (arr) => arr.length > 0 ? arr.reduce((s, v) => s + v, 0) / arr.length : 0;
|
|
34535
|
+
const lifts = [];
|
|
34536
|
+
for (let v = 1; v < samples.length; v++) {
|
|
34537
|
+
const abs = [];
|
|
34538
|
+
const rel = [];
|
|
34539
|
+
for (let i = 0; i < control.length; i++) {
|
|
34540
|
+
abs.push(samples[v][i] - control[i]);
|
|
34541
|
+
if (Math.abs(control[i]) > EPSILON) rel.push((samples[v][i] - control[i]) / control[i]);
|
|
34542
|
+
}
|
|
34543
|
+
abs.sort((a, b) => a - b);
|
|
34544
|
+
rel.sort((a, b) => a - b);
|
|
34545
|
+
lifts.push({
|
|
34546
|
+
variantIndex: v,
|
|
34547
|
+
mean: mean(abs),
|
|
34548
|
+
ci: percentileCi(abs),
|
|
34549
|
+
relativeMean: mean(rel),
|
|
34550
|
+
relativeCi: percentileCi(rel)
|
|
34551
|
+
});
|
|
34552
|
+
}
|
|
34553
|
+
return lifts;
|
|
34554
|
+
}
|
|
34423
34555
|
function bayesianMonteCarlo(variants, simulations = 1e5, credibility = 0.95) {
|
|
34424
34556
|
const n = variants.length;
|
|
34425
34557
|
const samples = [];
|
|
@@ -34531,7 +34663,8 @@ function bayesianMonteCarlo(variants, simulations = 1e5, credibility = 0.95) {
|
|
|
34531
34663
|
credibleIntervals,
|
|
34532
34664
|
posteriorMeans,
|
|
34533
34665
|
liftDistribution: { mean: liftMean, ci: liftCi, histogram },
|
|
34534
|
-
relativeLiftDistribution: { mean: relLiftMean, ci: relLiftCi, histogram: relHistogram }
|
|
34666
|
+
relativeLiftDistribution: { mean: relLiftMean, ci: relLiftCi, histogram: relHistogram },
|
|
34667
|
+
variantLifts: computeVariantLifts(samples, tail)
|
|
34535
34668
|
};
|
|
34536
34669
|
}
|
|
34537
34670
|
function normalPosterior(sampleMean, sampleStdDev, sampleSize) {
|
|
@@ -34651,7 +34784,8 @@ function bayesianMonteCarloNormal(variants, simulations = 1e5, credibility = 0.9
|
|
|
34651
34784
|
credibleIntervals,
|
|
34652
34785
|
posteriorMeans,
|
|
34653
34786
|
liftDistribution: { mean: liftMean, ci: liftCi, histogram },
|
|
34654
|
-
relativeLiftDistribution: { mean: relLiftMean, ci: relLiftCi, histogram: relHistogram }
|
|
34787
|
+
relativeLiftDistribution: { mean: relLiftMean, ci: relLiftCi, histogram: relHistogram },
|
|
34788
|
+
variantLifts: computeVariantLifts(samples, tail)
|
|
34655
34789
|
};
|
|
34656
34790
|
}
|
|
34657
34791
|
|
|
@@ -34719,9 +34853,29 @@ function trackToolUsage(toolName) {
|
|
|
34719
34853
|
}
|
|
34720
34854
|
var server = new McpServer({
|
|
34721
34855
|
name: "ABTestResult",
|
|
34722
|
-
version: "1.
|
|
34856
|
+
version: "1.3.0",
|
|
34723
34857
|
description: "Statistical tools for A/B testing \u2014 significance testing, sample size calculation, Bayesian analysis, and more. Powered by abtestresult.com"
|
|
34724
34858
|
});
|
|
34859
|
+
var SITE_URL = (process.env.ABTESTRESULT_BASE_URL || "https://abtestresult.com").replace(/\/+$/, "");
|
|
34860
|
+
function toBase64Url(str) {
|
|
34861
|
+
return Buffer.from(str, "utf-8").toString("base64url");
|
|
34862
|
+
}
|
|
34863
|
+
function buildDeeplink(path, compact) {
|
|
34864
|
+
const clean = {};
|
|
34865
|
+
for (const [k, v] of Object.entries(compact)) {
|
|
34866
|
+
if (v !== void 0 && v !== null && v !== "") clean[k] = v;
|
|
34867
|
+
}
|
|
34868
|
+
return `${SITE_URL}${path}?d=${toBase64Url(JSON.stringify(clean))}`;
|
|
34869
|
+
}
|
|
34870
|
+
function pipeJoin(parts) {
|
|
34871
|
+
const arr = parts.map((p) => p === void 0 || p === null ? "" : String(p));
|
|
34872
|
+
while (arr.length > 0 && arr[arr.length - 1] === "") arr.pop();
|
|
34873
|
+
return arr.join("|");
|
|
34874
|
+
}
|
|
34875
|
+
function tailCode(testType, niMargin) {
|
|
34876
|
+
if (niMargin > 0) return "n";
|
|
34877
|
+
return testType === "one-sided" ? "1" : "2";
|
|
34878
|
+
}
|
|
34725
34879
|
server.tool(
|
|
34726
34880
|
"analyze_ab_test",
|
|
34727
34881
|
`Analyze an A/B test with rate/proportion metrics (e.g. conversion rate, click-through rate).
|
|
@@ -34735,13 +34889,15 @@ Example: "Control had 5000 visitors with 250 conversions, variant had 5000 visit
|
|
|
34735
34889
|
variant_conversions: external_exports3.number().int().min(0).describe("Number of conversions in the variant group"),
|
|
34736
34890
|
confidence_level: external_exports3.number().min(0.5).max(0.999).default(0.95).describe("Confidence level (e.g. 0.95 for 95%). Default: 0.95"),
|
|
34737
34891
|
test_type: external_exports3.enum(["two-sided", "one-sided"]).default("two-sided").describe("Two-sided tests for any difference, one-sided for improvement only. Default: two-sided"),
|
|
34738
|
-
non_inferiority_margin: external_exports3.number().min(0).default(0).describe("Non-inferiority margin as percentage (e.g.
|
|
34892
|
+
non_inferiority_margin: external_exports3.number().min(0).default(0).describe("Non-inferiority margin as a RELATIVE percentage of the control rate (e.g. 2 means the variant may be up to 2% of the control rate worse: at an 80% control rate that is 1.6 percentage points, not 2 points). When > 0, a one-sided non-inferiority test is run and test_type is ignored. Default: 0 (superiority test)"),
|
|
34893
|
+
expected_control_share: external_exports3.number().gt(0).lt(1).optional().describe("Optional planned share of users in control, as a decimal, for intentionally unequal splits (e.g. 0.05 for a 5% holdout). When set, the response includes a sample ratio mismatch check against this planned split. Analyse the real user counts; never rescale a group to equal size.")
|
|
34739
34894
|
},
|
|
34740
34895
|
async (params) => {
|
|
34741
34896
|
const rl = checkRateLimit();
|
|
34742
34897
|
if (rl) return { content: rl, isError: true };
|
|
34743
34898
|
trackToolUsage("analyze_ab_test");
|
|
34744
|
-
const
|
|
34899
|
+
const isNonInferiority = params.non_inferiority_margin > 0;
|
|
34900
|
+
const sided = isNonInferiority || params.test_type === "one-sided" ? 1 : 2;
|
|
34745
34901
|
const result = zTestForProportions(
|
|
34746
34902
|
params.control_users,
|
|
34747
34903
|
params.control_conversions,
|
|
@@ -34751,14 +34907,29 @@ Example: "Control had 5000 visitors with 250 conversions, variant had 5000 visit
|
|
|
34751
34907
|
sided,
|
|
34752
34908
|
params.non_inferiority_margin
|
|
34753
34909
|
);
|
|
34754
|
-
const
|
|
34910
|
+
const marginAbsolute = params.non_inferiority_margin / 100 * result.controlRate;
|
|
34911
|
+
const marginText = `${params.non_inferiority_margin}% of the control rate (${round(marginAbsolute * 100, 4)} percentage points; variant must stay above ${toPercent(result.controlRate - marginAbsolute)})`;
|
|
34912
|
+
const summary = isNonInferiority ? nonInferioritySummary(result.isSignificant, marginText) : superioritySummary(result.isSignificant, result.delta);
|
|
34913
|
+
const ctrlRate = (params.control_conversions / params.control_users * 100).toFixed(2);
|
|
34914
|
+
const varRate = (params.variant_conversions / params.variant_users * 100).toFixed(2);
|
|
34915
|
+
const viewUrl = buildDeeplink("/statistical-significance-calculator", {
|
|
34916
|
+
c: pipeJoin([params.control_users, params.control_conversions, ctrlRate]),
|
|
34917
|
+
v: pipeJoin([params.variant_users, params.variant_conversions, varRate]),
|
|
34918
|
+
m: "r",
|
|
34919
|
+
t: tailCode(params.test_type, params.non_inferiority_margin),
|
|
34920
|
+
l: String(Math.round(params.confidence_level * 100)),
|
|
34921
|
+
n: params.non_inferiority_margin > 0 ? String(params.non_inferiority_margin) : "",
|
|
34922
|
+
r: params.expected_control_share !== void 0 ? String(round(params.expected_control_share * 100, 4)) : ""
|
|
34923
|
+
});
|
|
34755
34924
|
return {
|
|
34756
34925
|
content: [{
|
|
34757
34926
|
type: "text",
|
|
34758
34927
|
text: JSON.stringify({
|
|
34759
34928
|
summary,
|
|
34929
|
+
view_url: viewUrl,
|
|
34760
34930
|
significant: result.isSignificant,
|
|
34761
|
-
p_value:
|
|
34931
|
+
p_value: roundPValue(result.pValue),
|
|
34932
|
+
p_value_display: formatPValue(result.pValue),
|
|
34762
34933
|
z_score: round(result.zScore, 4),
|
|
34763
34934
|
control_rate: toPercent(result.controlRate),
|
|
34764
34935
|
variant_rate: toPercent(result.variantRate),
|
|
@@ -34777,7 +34948,15 @@ Example: "Control had 5000 visitors with 250 conversions, variant had 5000 visit
|
|
|
34777
34948
|
upper: toPercent(result.ciVariation[1])
|
|
34778
34949
|
},
|
|
34779
34950
|
confidence_level: toPercent(params.confidence_level),
|
|
34780
|
-
test_type: params.test_type
|
|
34951
|
+
test_type: isNonInferiority ? "non-inferiority (one-sided)" : params.test_type,
|
|
34952
|
+
...isNonInferiority && {
|
|
34953
|
+
non_inferiority_margin: {
|
|
34954
|
+
relative_to_control: `${params.non_inferiority_margin}%`,
|
|
34955
|
+
absolute_percentage_points: round(marginAbsolute * 100, 4),
|
|
34956
|
+
threshold_rate: toPercent(result.controlRate - marginAbsolute)
|
|
34957
|
+
}
|
|
34958
|
+
},
|
|
34959
|
+
...plannedSplitSRM(params.control_users, params.variant_users, params.expected_control_share)
|
|
34781
34960
|
}, null, 2)
|
|
34782
34961
|
}]
|
|
34783
34962
|
};
|
|
@@ -34786,25 +34965,27 @@ Example: "Control had 5000 visitors with 250 conversions, variant had 5000 visit
|
|
|
34786
34965
|
server.tool(
|
|
34787
34966
|
"analyze_ab_test_average",
|
|
34788
34967
|
`Analyze an A/B test with continuous/average metrics (e.g. revenue per user, time on page, order value).
|
|
34789
|
-
Uses
|
|
34968
|
+
Uses Welch's T-test (unequal variances), which stays accurate when groups differ in size or spread (e.g. 95/5 holdouts). Returns significance, p-value, confidence intervals, and lift.
|
|
34790
34969
|
|
|
34791
34970
|
Example: "Control had 1000 users with mean $45.20 (std $12.50), variant had 1000 users with mean $48.80 (std $13.10)"`,
|
|
34792
34971
|
{
|
|
34793
|
-
control_users: external_exports3.number().int().
|
|
34972
|
+
control_users: external_exports3.number().int().min(2).describe("Number of users in the control group (at least 2)"),
|
|
34794
34973
|
control_mean: external_exports3.number().describe("Mean value for the control group"),
|
|
34795
34974
|
control_std_dev: external_exports3.number().positive().describe("Standard deviation for the control group"),
|
|
34796
|
-
variant_users: external_exports3.number().int().
|
|
34975
|
+
variant_users: external_exports3.number().int().min(2).describe("Number of users in the variant group (at least 2)"),
|
|
34797
34976
|
variant_mean: external_exports3.number().describe("Mean value for the variant group"),
|
|
34798
34977
|
variant_std_dev: external_exports3.number().positive().describe("Standard deviation for the variant group"),
|
|
34799
34978
|
confidence_level: external_exports3.number().min(0.5).max(0.999).default(0.95).describe("Confidence level (e.g. 0.95 for 95%). Default: 0.95"),
|
|
34800
34979
|
test_type: external_exports3.enum(["two-sided", "one-sided"]).default("two-sided").describe("Two-sided or one-sided test. Default: two-sided"),
|
|
34801
|
-
non_inferiority_margin: external_exports3.number().min(0).default(0).describe("Non-inferiority margin as percentage. Default: 0")
|
|
34980
|
+
non_inferiority_margin: external_exports3.number().min(0).default(0).describe("Non-inferiority margin as a RELATIVE percentage of the control mean (e.g. 2 means the variant mean may be up to 2% of the control mean lower). When > 0, a one-sided non-inferiority test is run and test_type is ignored. Default: 0 (superiority test)"),
|
|
34981
|
+
expected_control_share: external_exports3.number().gt(0).lt(1).optional().describe("Optional planned share of users in control, as a decimal, for intentionally unequal splits (e.g. 0.05 for a 5% holdout). When set, the response includes a sample ratio mismatch check against this planned split. Analyse the real user counts; never rescale a group to equal size.")
|
|
34802
34982
|
},
|
|
34803
34983
|
async (params) => {
|
|
34804
34984
|
const rl = checkRateLimit();
|
|
34805
34985
|
if (rl) return { content: rl, isError: true };
|
|
34806
34986
|
trackToolUsage("analyze_ab_test_average");
|
|
34807
|
-
const
|
|
34987
|
+
const isNonInferiority = params.non_inferiority_margin > 0;
|
|
34988
|
+
const sided = isNonInferiority || params.test_type === "one-sided" ? 1 : 2;
|
|
34808
34989
|
const result = tTestForAverages(
|
|
34809
34990
|
params.control_users,
|
|
34810
34991
|
params.control_mean,
|
|
@@ -34816,16 +34997,29 @@ Example: "Control had 1000 users with mean $45.20 (std $12.50), variant had 1000
|
|
|
34816
34997
|
sided,
|
|
34817
34998
|
params.non_inferiority_margin
|
|
34818
34999
|
);
|
|
34819
|
-
const
|
|
35000
|
+
const marginAbsolute = params.non_inferiority_margin / 100 * result.controlMean;
|
|
35001
|
+
const marginText = `${params.non_inferiority_margin}% of the control mean (${round(marginAbsolute, 4)} in metric units; variant mean must stay above ${round(result.controlMean - marginAbsolute, 4)})`;
|
|
35002
|
+
const summary = isNonInferiority ? nonInferioritySummary(result.isSignificant, marginText) : superioritySummary(result.isSignificant, result.delta);
|
|
35003
|
+
const viewUrl = buildDeeplink("/statistical-significance-calculator", {
|
|
35004
|
+
c: pipeJoin([params.control_users, params.control_mean, "", params.control_std_dev]),
|
|
35005
|
+
v: pipeJoin([params.variant_users, params.variant_mean, "", params.variant_std_dev]),
|
|
35006
|
+
m: "a",
|
|
35007
|
+
t: tailCode(params.test_type, params.non_inferiority_margin),
|
|
35008
|
+
l: String(Math.round(params.confidence_level * 100)),
|
|
35009
|
+
n: params.non_inferiority_margin > 0 ? String(params.non_inferiority_margin) : "",
|
|
35010
|
+
r: params.expected_control_share !== void 0 ? String(round(params.expected_control_share * 100, 4)) : ""
|
|
35011
|
+
});
|
|
34820
35012
|
return {
|
|
34821
35013
|
content: [{
|
|
34822
35014
|
type: "text",
|
|
34823
35015
|
text: JSON.stringify({
|
|
34824
35016
|
summary,
|
|
35017
|
+
view_url: viewUrl,
|
|
34825
35018
|
significant: result.isSignificant,
|
|
34826
|
-
p_value:
|
|
35019
|
+
p_value: roundPValue(result.pValue),
|
|
35020
|
+
p_value_display: formatPValue(result.pValue),
|
|
34827
35021
|
t_score: round(result.tScore, 4),
|
|
34828
|
-
degrees_of_freedom: result.df,
|
|
35022
|
+
degrees_of_freedom: round(result.df, 2),
|
|
34829
35023
|
control_mean: round(result.controlMean, 4),
|
|
34830
35024
|
variant_mean: round(result.variantMean, 4),
|
|
34831
35025
|
absolute_lift: round(result.delta, 4),
|
|
@@ -34843,7 +35037,15 @@ Example: "Control had 1000 users with mean $45.20 (std $12.50), variant had 1000
|
|
|
34843
35037
|
upper: round(result.variationCi[1], 4)
|
|
34844
35038
|
},
|
|
34845
35039
|
confidence_level: toPercent(params.confidence_level),
|
|
34846
|
-
test_type: params.test_type
|
|
35040
|
+
test_type: isNonInferiority ? "non-inferiority (one-sided)" : params.test_type,
|
|
35041
|
+
...isNonInferiority && {
|
|
35042
|
+
non_inferiority_margin: {
|
|
35043
|
+
relative_to_control: `${params.non_inferiority_margin}%`,
|
|
35044
|
+
absolute: round(marginAbsolute, 4),
|
|
35045
|
+
threshold_mean: round(result.controlMean - marginAbsolute, 4)
|
|
35046
|
+
}
|
|
35047
|
+
},
|
|
35048
|
+
...plannedSplitSRM(params.control_users, params.variant_users, params.expected_control_share)
|
|
34847
35049
|
}, null, 2)
|
|
34848
35050
|
}]
|
|
34849
35051
|
};
|
|
@@ -34865,7 +35067,8 @@ Example: "How many users do I need to detect a 10% relative lift on a 5% baselin
|
|
|
34865
35067
|
power: external_exports3.number().min(0.5).max(0.999).default(0.8).describe("Statistical power. Default: 0.80"),
|
|
34866
35068
|
test_type: external_exports3.enum(["two-sided", "one-sided"]).default("two-sided").describe("Test sidedness. Default: two-sided"),
|
|
34867
35069
|
num_variants: external_exports3.number().int().min(1).default(1).describe("Number of variant groups (not counting control). Default: 1"),
|
|
34868
|
-
daily_traffic: external_exports3.number().int().positive().optional().describe("Daily traffic to estimate test duration in days")
|
|
35070
|
+
daily_traffic: external_exports3.number().int().positive().optional().describe("Daily traffic to estimate test duration in days"),
|
|
35071
|
+
control_share: external_exports3.number().gt(0).lt(1).optional().describe("Planned share of traffic in control, as a decimal, for unequal splits (e.g. 0.05 for a 5% holdout, 0.95 for a 5% canary variant). Omit for an equal split.")
|
|
34869
35072
|
},
|
|
34870
35073
|
async (params) => {
|
|
34871
35074
|
const rl = checkRateLimit();
|
|
@@ -34910,14 +35113,35 @@ Example: "How many users do I need to detect a 10% relative lift on a 5% baselin
|
|
|
34910
35113
|
params.daily_traffic ?? null
|
|
34911
35114
|
);
|
|
34912
35115
|
}
|
|
35116
|
+
const viewUrl = buildDeeplink("/sample-size-calculator", {
|
|
35117
|
+
m: params.metric_type === "rate" ? "r" : "a",
|
|
35118
|
+
t: params.test_type === "one-sided" ? "1" : "2",
|
|
35119
|
+
b: String(params.baseline),
|
|
35120
|
+
d: String(params.mde),
|
|
35121
|
+
s: params.metric_type === "average" && params.std_dev ? String(params.std_dev) : "",
|
|
35122
|
+
c: String(Math.round(params.confidence_level * 100)),
|
|
35123
|
+
p: String(Math.round(params.power * 100)),
|
|
35124
|
+
v: params.num_variants !== 1 ? String(params.num_variants) : "",
|
|
35125
|
+
r: params.control_share !== void 0 ? String(round(params.control_share * 100, 4)) : ""
|
|
35126
|
+
});
|
|
35127
|
+
const equalShare = 1 / (params.num_variants + 1);
|
|
35128
|
+
const allocation = params.control_share !== void 0 && Math.abs(params.control_share - equalShare) > 1e-9 ? allocateSampleSize(result.samplesEach, params.num_variants, params.control_share) : null;
|
|
35129
|
+
const totalSamples = allocation ? allocation.total : result.samplesTotal;
|
|
35130
|
+
const estimatedDays = params.daily_traffic ? Math.ceil(totalSamples / params.daily_traffic) : null;
|
|
34913
35131
|
return {
|
|
34914
35132
|
content: [{
|
|
34915
35133
|
type: "text",
|
|
34916
35134
|
text: JSON.stringify({
|
|
34917
|
-
|
|
34918
|
-
|
|
35135
|
+
view_url: viewUrl,
|
|
35136
|
+
...allocation ? {
|
|
35137
|
+
control_samples: allocation.control,
|
|
35138
|
+
samples_per_variant: allocation.perVariant,
|
|
35139
|
+
equal_split_samples_per_group: result.samplesEach,
|
|
35140
|
+
traffic_multiplier_vs_equal_split: round(allocation.multiplier, 3)
|
|
35141
|
+
} : { samples_per_group: result.samplesEach },
|
|
35142
|
+
total_samples: totalSamples,
|
|
34919
35143
|
number_of_groups: params.num_variants + 1,
|
|
34920
|
-
estimated_days:
|
|
35144
|
+
estimated_days: estimatedDays,
|
|
34921
35145
|
effect_size: round(result.effectSize, 4),
|
|
34922
35146
|
absolute_mde: round(result.mdeAbsolute, 6),
|
|
34923
35147
|
settings: {
|
|
@@ -34946,7 +35170,8 @@ Example: "I have 50,000 total users and a 5% conversion rate. What's my MDE?"`,
|
|
|
34946
35170
|
std_dev: external_exports3.number().positive().optional().describe("Standard deviation. Required for average metrics. For rate metrics, auto-calculated from baseline if omitted."),
|
|
34947
35171
|
confidence_level: external_exports3.number().min(0.5).max(0.999).default(0.95).describe("Confidence level. Default: 0.95"),
|
|
34948
35172
|
power: external_exports3.number().min(0.5).max(0.999).default(0.8).describe("Statistical power. Default: 0.80"),
|
|
34949
|
-
num_variants: external_exports3.number().int().min(1).default(1).describe("Number of variant groups. Default: 1")
|
|
35173
|
+
num_variants: external_exports3.number().int().min(1).default(1).describe("Number of variant groups. Default: 1"),
|
|
35174
|
+
control_share: external_exports3.number().gt(0).lt(1).optional().describe("Share of total traffic in control, as a decimal, for unequal splits (e.g. 0.05 for a 5% holdout). Omit for an equal split.")
|
|
34950
35175
|
},
|
|
34951
35176
|
async (params) => {
|
|
34952
35177
|
const rl = checkRateLimit();
|
|
@@ -34968,25 +35193,59 @@ Example: "I have 50,000 total users and a 5% conversion rate. What's my MDE?"`,
|
|
|
34968
35193
|
}
|
|
34969
35194
|
stdDev = params.std_dev;
|
|
34970
35195
|
}
|
|
34971
|
-
const mdeAbsolute =
|
|
35196
|
+
const mdeAbsolute = params.metric_type === "rate" ? estimateMDERate({
|
|
35197
|
+
traffic: params.total_traffic,
|
|
35198
|
+
baselineRate: params.baseline / 100,
|
|
35199
|
+
power: params.power,
|
|
35200
|
+
confidenceLevel: params.confidence_level,
|
|
35201
|
+
numVariants: params.num_variants,
|
|
35202
|
+
controlShare: params.control_share
|
|
35203
|
+
}) : estimateMDE(
|
|
34972
35204
|
params.total_traffic,
|
|
34973
35205
|
stdDev,
|
|
34974
35206
|
params.power,
|
|
34975
35207
|
params.confidence_level,
|
|
34976
|
-
params.num_variants
|
|
35208
|
+
params.num_variants,
|
|
35209
|
+
1,
|
|
35210
|
+
0,
|
|
35211
|
+
params.control_share
|
|
34977
35212
|
);
|
|
35213
|
+
if (!Number.isFinite(mdeAbsolute)) {
|
|
35214
|
+
return {
|
|
35215
|
+
content: [{
|
|
35216
|
+
type: "text",
|
|
35217
|
+
text: params.metric_type === "rate" && !(params.baseline > 0 && params.baseline < 100) ? "Error: baseline must be between 0 and 100 (percent) for rate metrics." : "No detectable effect: with this traffic, even a lift to a 100% conversion rate would not reach the target power. Increase traffic or lower the power or confidence level."
|
|
35218
|
+
}],
|
|
35219
|
+
isError: true
|
|
35220
|
+
};
|
|
35221
|
+
}
|
|
35222
|
+
const controlShare = params.control_share ?? 1 / (params.num_variants + 1);
|
|
34978
35223
|
const baselineValue = params.metric_type === "rate" ? params.baseline / 100 : params.baseline;
|
|
34979
35224
|
const mdeRelative = baselineValue !== 0 ? mdeAbsolute / baselineValue : 0;
|
|
35225
|
+
const viewUrl = buildDeeplink("/mde-calculator", {
|
|
35226
|
+
t: String(params.total_traffic),
|
|
35227
|
+
s: params.std_dev ? String(params.std_dev) : "",
|
|
35228
|
+
r: String(params.baseline),
|
|
35229
|
+
p: String(params.power),
|
|
35230
|
+
c: String(params.confidence_level),
|
|
35231
|
+
m: params.metric_type === "average" ? "c" : "r",
|
|
35232
|
+
k: params.control_share !== void 0 ? String(round(params.control_share * 100, 4)) : "",
|
|
35233
|
+
v: params.num_variants !== 1 ? String(params.num_variants) : ""
|
|
35234
|
+
});
|
|
34980
35235
|
return {
|
|
34981
35236
|
content: [{
|
|
34982
35237
|
type: "text",
|
|
34983
35238
|
text: JSON.stringify({
|
|
35239
|
+
view_url: viewUrl,
|
|
34984
35240
|
mde_absolute: round(mdeAbsolute, 6),
|
|
34985
35241
|
mde_relative_percent: toPercent(mdeRelative),
|
|
34986
35242
|
interpretation: params.metric_type === "rate" ? `You can detect an absolute change of ${toPercent(mdeAbsolute)} (${toPercent(mdeRelative)} relative lift) from the ${params.baseline}% baseline.` : `You can detect an absolute change of ${round(mdeAbsolute, 4)} (${toPercent(mdeRelative)} relative lift) from the ${params.baseline} baseline.`,
|
|
34987
35243
|
settings: {
|
|
34988
35244
|
total_traffic: params.total_traffic,
|
|
34989
|
-
traffic_per_group: Math.floor(params.total_traffic / (params.num_variants + 1))
|
|
35245
|
+
...params.control_share === void 0 ? { traffic_per_group: Math.floor(params.total_traffic / (params.num_variants + 1)) } : {
|
|
35246
|
+
control_traffic: Math.round(params.total_traffic * controlShare),
|
|
35247
|
+
traffic_per_variant: Math.round(params.total_traffic * (1 - controlShare) / params.num_variants)
|
|
35248
|
+
},
|
|
34990
35249
|
baseline: params.baseline,
|
|
34991
35250
|
confidence_level: toPercent(params.confidence_level),
|
|
34992
35251
|
power: toPercent(params.power)
|
|
@@ -34998,10 +35257,17 @@ Example: "I have 50,000 total users and a 5% conversion rate. What's my MDE?"`,
|
|
|
34998
35257
|
);
|
|
34999
35258
|
server.tool(
|
|
35000
35259
|
"bayesian_ab_test",
|
|
35001
|
-
`Run Bayesian A/B test analysis. Returns probability to be best, expected loss,
|
|
35260
|
+
`Run Bayesian A/B test analysis. Returns probability to be best, expected loss, credible intervals,
|
|
35261
|
+
and the lift of every variant vs the control (first variant).
|
|
35002
35262
|
Supports both rate metrics (Beta-Binomial model) and average metrics (Normal model).
|
|
35003
35263
|
Supports multiple variants.
|
|
35004
35264
|
|
|
35265
|
+
Units: for rate metrics, raw values (posterior_mean, credible_interval, expected_loss, absolute lift) are
|
|
35266
|
+
proportions (0.05 = 5%), so a difference of 0.0006 is 0.06 percentage points. For average metrics they are
|
|
35267
|
+
in the metric's own units. Each variant also includes expected_loss_display in plain terms.
|
|
35268
|
+
Expected loss = the average amount you give up by shipping that variant if it is not actually the best.
|
|
35269
|
+
A common stopping rule: ship when the leader's expected loss is below the smallest difference you care about.
|
|
35270
|
+
|
|
35005
35271
|
Example: "Control: 5000 visitors, 250 conversions. Variant: 5000 visitors, 300 conversions."`,
|
|
35006
35272
|
{
|
|
35007
35273
|
metric_type: external_exports3.enum(["rate", "average"]).describe('Metric type: "rate" for proportions, "average" for continuous'),
|
|
@@ -35029,18 +35295,32 @@ Example: "Control: 5000 visitors, 250 conversions. Variant: 5000 visitors, 300 c
|
|
|
35029
35295
|
});
|
|
35030
35296
|
result = bayesianMonteCarlo(posteriors, params.simulations, params.credibility);
|
|
35031
35297
|
} else {
|
|
35032
|
-
const
|
|
35033
|
-
|
|
35034
|
-
|
|
35035
|
-
|
|
35036
|
-
|
|
35037
|
-
|
|
35298
|
+
const missing = params.variants.find((v) => v.mean === void 0 || v.std_dev === void 0);
|
|
35299
|
+
if (missing) {
|
|
35300
|
+
return {
|
|
35301
|
+
content: [{
|
|
35302
|
+
type: "text",
|
|
35303
|
+
text: `Error: mean and std_dev are required for average metrics (variant: ${missing.name}).`
|
|
35304
|
+
}],
|
|
35305
|
+
isError: true
|
|
35306
|
+
};
|
|
35307
|
+
}
|
|
35308
|
+
const posteriors = params.variants.map((v) => normalPosterior(v.mean, v.std_dev, v.users));
|
|
35038
35309
|
result = bayesianMonteCarloNormal(posteriors, params.simulations, params.credibility);
|
|
35039
35310
|
}
|
|
35311
|
+
const isRate = params.metric_type === "rate";
|
|
35312
|
+
const controlMean = result.posteriorMeans[0];
|
|
35313
|
+
const lossDisplay = (loss) => {
|
|
35314
|
+
const relPct = controlMean !== 0 ? loss / Math.abs(controlMean) * 100 : null;
|
|
35315
|
+
const relText = relPct === null ? "" : relPct > 0 && relPct < 0.01 ? "< 0.01%" : `\u2248 ${round(relPct, 2)}%`;
|
|
35316
|
+
const relative = relPct === null ? "" : ` (${relText} of the control ${isRate ? "rate" : "mean"})`;
|
|
35317
|
+
return isRate ? `\u2248 ${round(loss * 100, 3)} percentage points${relative}` : `\u2248 ${round(loss, 4)}${relative}`;
|
|
35318
|
+
};
|
|
35040
35319
|
const variantResults = params.variants.map((v, i) => ({
|
|
35041
35320
|
name: v.name,
|
|
35042
35321
|
probability_to_be_best: toPercent(result.probabilityToBeBest[i]),
|
|
35043
35322
|
expected_loss: round(result.expectedLoss[i], 6),
|
|
35323
|
+
expected_loss_display: lossDisplay(result.expectedLoss[i]),
|
|
35044
35324
|
credible_interval: {
|
|
35045
35325
|
lower: round(result.credibleIntervals[i].lower, 6),
|
|
35046
35326
|
upper: round(result.credibleIntervals[i].upper, 6)
|
|
@@ -35048,27 +35328,43 @@ Example: "Control: 5000 visitors, 250 conversions. Variant: 5000 visitors, 300 c
|
|
|
35048
35328
|
posterior_mean: round(result.posteriorMeans[i], 6)
|
|
35049
35329
|
}));
|
|
35050
35330
|
const bestIdx = result.probabilityToBeBest.indexOf(Math.max(...result.probabilityToBeBest));
|
|
35051
|
-
const summary = result.probabilityToBeBest[bestIdx] > 0.95 ? `\u{1F3C6} ${params.variants[bestIdx].name} is the winner with ${toPercent(result.probabilityToBeBest[bestIdx])} probability of being best.` : result.probabilityToBeBest[bestIdx] > 0.8 ? `\u{1F4CA} ${params.variants[bestIdx].name} is leading with ${toPercent(result.probabilityToBeBest[bestIdx])} probability, but more data may be needed.` : `\u23F3 No clear winner yet. The leading variant has only ${toPercent(result.probabilityToBeBest[bestIdx])} probability of being best.`;
|
|
35331
|
+
const summary = result.probabilityToBeBest[bestIdx] > 0.95 ? `\u{1F3C6} ${params.variants[bestIdx].name} is the winner with ${toPercent(result.probabilityToBeBest[bestIdx])} probability of being best.` : result.probabilityToBeBest[bestIdx] > 0.8 ? `\u{1F4CA} ${params.variants[bestIdx].name} is leading with ${toPercent(result.probabilityToBeBest[bestIdx])} probability, but more data may be needed.` : `\u23F3 No clear winner yet. The leading variant (${params.variants[bestIdx].name}) has only ${toPercent(result.probabilityToBeBest[bestIdx])} probability of being best.`;
|
|
35332
|
+
const summaryWithLoss = `${summary} Expected loss if you ship ${params.variants[bestIdx].name}: ${lossDisplay(result.expectedLoss[bestIdx])}.`;
|
|
35333
|
+
const liftVsControl = result.variantLifts.map((l) => ({
|
|
35334
|
+
variant: params.variants[l.variantIndex].name,
|
|
35335
|
+
vs: params.variants[0].name,
|
|
35336
|
+
absolute_lift: isRate ? {
|
|
35337
|
+
mean: toPercentagePoints(l.mean),
|
|
35338
|
+
credible_interval: { lower: toPercentagePoints(l.ci[0]), upper: toPercentagePoints(l.ci[1]) }
|
|
35339
|
+
} : {
|
|
35340
|
+
mean: round(l.mean, 6),
|
|
35341
|
+
credible_interval: { lower: round(l.ci[0], 6), upper: round(l.ci[1], 6) }
|
|
35342
|
+
},
|
|
35343
|
+
relative_lift: {
|
|
35344
|
+
mean: toPercent(l.relativeMean),
|
|
35345
|
+
credible_interval: { lower: toPercent(l.relativeCi[0]), upper: toPercent(l.relativeCi[1]) }
|
|
35346
|
+
}
|
|
35347
|
+
}));
|
|
35348
|
+
const bayesPart = (v, withName) => {
|
|
35349
|
+
const rate = isRate && v.conversions !== void 0 && v.users ? (v.conversions / v.users * 100).toFixed(2) : "";
|
|
35350
|
+
return pipeJoin(isRate ? [v.users, v.conversions, rate, "", withName ? v.name : ""] : [v.users, v.mean, "", v.std_dev, withName ? v.name : ""]);
|
|
35351
|
+
};
|
|
35352
|
+
const viewUrl = buildDeeplink("/bayesian-calculator", {
|
|
35353
|
+
c: bayesPart(params.variants[0], false),
|
|
35354
|
+
v: params.variants.slice(1).map((v) => bayesPart(v, true)).join(";"),
|
|
35355
|
+
m: isRate ? "r" : "a",
|
|
35356
|
+
// alpha=beta=1 is the uninformative prior; anything else maps to "weak".
|
|
35357
|
+
p: params.prior_alpha === 1 && params.prior_beta === 1 ? "u" : "w",
|
|
35358
|
+
l: String(Math.round(params.credibility * 100))
|
|
35359
|
+
});
|
|
35052
35360
|
return {
|
|
35053
35361
|
content: [{
|
|
35054
35362
|
type: "text",
|
|
35055
35363
|
text: JSON.stringify({
|
|
35056
|
-
summary,
|
|
35364
|
+
summary: summaryWithLoss,
|
|
35365
|
+
view_url: viewUrl,
|
|
35057
35366
|
variants: variantResults,
|
|
35058
|
-
|
|
35059
|
-
mean: round(result.liftDistribution.mean, 6),
|
|
35060
|
-
credible_interval: {
|
|
35061
|
-
lower: round(result.liftDistribution.ci[0], 6),
|
|
35062
|
-
upper: round(result.liftDistribution.ci[1], 6)
|
|
35063
|
-
}
|
|
35064
|
-
},
|
|
35065
|
-
relative_lift_distribution: {
|
|
35066
|
-
mean: toPercent(result.relativeLiftDistribution.mean),
|
|
35067
|
-
credible_interval: {
|
|
35068
|
-
lower: toPercent(result.relativeLiftDistribution.ci[0]),
|
|
35069
|
-
upper: toPercent(result.relativeLiftDistribution.ci[1])
|
|
35070
|
-
}
|
|
35071
|
-
},
|
|
35367
|
+
lift_vs_control: liftVsControl,
|
|
35072
35368
|
simulations: params.simulations,
|
|
35073
35369
|
credibility: toPercent(params.credibility)
|
|
35074
35370
|
}, null, 2)
|
|
@@ -35079,34 +35375,48 @@ Example: "Control: 5000 visitors, 250 conversions. Variant: 5000 visitors, 300 c
|
|
|
35079
35375
|
server.tool(
|
|
35080
35376
|
"check_srm",
|
|
35081
35377
|
`Check for Sample Ratio Mismatch (SRM) in an A/B test.
|
|
35082
|
-
SRM occurs when the traffic split between control and variant doesn't match the
|
|
35378
|
+
SRM occurs when the traffic split between control and variant doesn't match the planned ratio (50/50 unless you say otherwise).
|
|
35083
35379
|
SRM is a sign of a bug in the experiment setup and invalidates results.
|
|
35380
|
+
For intentionally unequal splits (holdouts, canary tests) pass the planned control share, e.g. 0.05 for a 5% holdout.
|
|
35084
35381
|
|
|
35085
35382
|
Example: "Control has 10,234 users, variant has 9,766 users"`,
|
|
35086
35383
|
{
|
|
35087
35384
|
control_users: external_exports3.number().int().positive().describe("Number of users in the control group"),
|
|
35088
|
-
variant_users: external_exports3.number().int().positive().describe("Number of users in the variant group")
|
|
35385
|
+
variant_users: external_exports3.number().int().positive().describe("Number of users in the variant group"),
|
|
35386
|
+
expected_control_share: external_exports3.number().gt(0).lt(1).default(0.5).describe("Planned share of users in control, as a decimal. 0.5 for 50/50, 0.1 for a 10% holdout, 0.95 for a 5% canary variant. Default: 0.5")
|
|
35089
35387
|
},
|
|
35090
35388
|
async (params) => {
|
|
35091
35389
|
const rl = checkRateLimit();
|
|
35092
35390
|
if (rl) return { content: rl, isError: true };
|
|
35093
35391
|
trackToolUsage("check_srm");
|
|
35094
|
-
const result = checkSRM(params.control_users, params.variant_users);
|
|
35392
|
+
const result = checkSRM(params.control_users, params.variant_users, params.expected_control_share);
|
|
35095
35393
|
const total = params.control_users + params.variant_users;
|
|
35096
35394
|
const actualRatio = params.control_users / total;
|
|
35097
|
-
const
|
|
35395
|
+
const expectedSplit = `${toPercent(params.expected_control_share)}/${toPercent(1 - params.expected_control_share)}`;
|
|
35396
|
+
const summary = result.hasMismatch ? `\u{1F6A8} SRM DETECTED (${formatPValueEq(result.pValue)}) \u2014 The traffic split is ${toPercent(actualRatio)}/${toPercent(1 - actualRatio)} instead of the expected ${expectedSplit}. Your experiment may have a bug. DO NOT trust the results.` : `\u2705 No SRM detected (${formatPValueEq(result.pValue)}) \u2014 The traffic split of ${toPercent(actualRatio)}/${toPercent(1 - actualRatio)} is consistent with the expected ${expectedSplit} split.`;
|
|
35397
|
+
const viewUrl = buildDeeplink("/sample-ratio-mismatch", {
|
|
35398
|
+
c: String(params.control_users),
|
|
35399
|
+
v: String(params.variant_users),
|
|
35400
|
+
e: params.expected_control_share !== 0.5 ? String(round(params.expected_control_share * 100, 4)) : ""
|
|
35401
|
+
});
|
|
35098
35402
|
return {
|
|
35099
35403
|
content: [{
|
|
35100
35404
|
type: "text",
|
|
35101
35405
|
text: JSON.stringify({
|
|
35102
35406
|
summary,
|
|
35407
|
+
view_url: viewUrl,
|
|
35103
35408
|
has_mismatch: result.hasMismatch,
|
|
35104
|
-
p_value:
|
|
35409
|
+
p_value: roundPValue(result.pValue),
|
|
35410
|
+
p_value_display: formatPValue(result.pValue),
|
|
35105
35411
|
chi_square: round(result.chiSquare, 4),
|
|
35106
35412
|
actual_split: {
|
|
35107
35413
|
control: toPercent(actualRatio),
|
|
35108
35414
|
variant: toPercent(1 - actualRatio)
|
|
35109
35415
|
},
|
|
35416
|
+
expected_split: {
|
|
35417
|
+
control: toPercent(params.expected_control_share),
|
|
35418
|
+
variant: toPercent(1 - params.expected_control_share)
|
|
35419
|
+
},
|
|
35110
35420
|
total_users: total
|
|
35111
35421
|
}, null, 2)
|
|
35112
35422
|
}]
|
|
@@ -35143,7 +35453,7 @@ Example: "Before: [10, 12, 8, 15, 11], After: [12, 14, 9, 18, 13]"`,
|
|
|
35143
35453
|
params.sample_after,
|
|
35144
35454
|
params.confidence_level
|
|
35145
35455
|
);
|
|
35146
|
-
const summary = result.isSignificant ? `\u2705 SIGNIFICANT (${result.testUsed}) \u2014 There is a statistically significant difference (
|
|
35456
|
+
const summary = result.isSignificant ? `\u2705 SIGNIFICANT (${result.testUsed}) \u2014 There is a statistically significant difference (${formatPValueEq(result.pValue)}).` : `\u23F3 NOT SIGNIFICANT (${result.testUsed}) \u2014 No statistically significant difference detected (${formatPValueEq(result.pValue)}).`;
|
|
35147
35457
|
return {
|
|
35148
35458
|
content: [{
|
|
35149
35459
|
type: "text",
|
|
@@ -35152,7 +35462,8 @@ Example: "Before: [10, 12, 8, 15, 11], After: [12, 14, 9, 18, 13]"`,
|
|
|
35152
35462
|
test_used: result.testUsed,
|
|
35153
35463
|
test_statistic_name: result.testStatisticName,
|
|
35154
35464
|
test_statistic: round(result.testStatistic, 4),
|
|
35155
|
-
p_value:
|
|
35465
|
+
p_value: roundPValue(result.pValue),
|
|
35466
|
+
p_value_display: formatPValue(result.pValue),
|
|
35156
35467
|
significant: result.isSignificant,
|
|
35157
35468
|
mean_before: round(result.mean1, 4),
|
|
35158
35469
|
mean_after: round(result.mean2, 4),
|
|
@@ -35174,13 +35485,15 @@ Example: "Before: [10, 12, 8, 15, 11], After: [12, 14, 9, 18, 13]"`,
|
|
|
35174
35485
|
);
|
|
35175
35486
|
server.tool(
|
|
35176
35487
|
"survey_sample_size",
|
|
35177
|
-
`Calculate the required
|
|
35488
|
+
`Calculate the required number of completed survey responses using Cochran's formula with finite population correction.
|
|
35489
|
+
Optionally pass expected_response_rate to get how many people to invite.
|
|
35178
35490
|
|
|
35179
35491
|
Example: "How many responses do I need from a population of 10,000 with 5% margin of error at 95% confidence?"`,
|
|
35180
35492
|
{
|
|
35181
35493
|
population: external_exports3.number().int().positive().describe("Total population size"),
|
|
35182
35494
|
margin_of_error: external_exports3.number().positive().max(50).describe("Desired margin of error as percentage (e.g. 5 for \xB15%)"),
|
|
35183
|
-
confidence_level: external_exports3.number().min(0.5).max(0.999).default(0.95).describe("Confidence level. Default: 0.95")
|
|
35495
|
+
confidence_level: external_exports3.number().min(0.5).max(0.999).default(0.95).describe("Confidence level. Default: 0.95"),
|
|
35496
|
+
expected_response_rate: external_exports3.number().positive().max(100).optional().describe("Optional: expected survey response rate as percentage (e.g. 20 for 20%). When given, returns how many people to invite to get the required completed responses.")
|
|
35184
35497
|
},
|
|
35185
35498
|
async (params) => {
|
|
35186
35499
|
const rl = checkRateLimit();
|
|
@@ -35191,15 +35504,26 @@ Example: "How many responses do I need from a population of 10,000 with 5% margi
|
|
|
35191
35504
|
params.margin_of_error / 100,
|
|
35192
35505
|
params.confidence_level
|
|
35193
35506
|
);
|
|
35507
|
+
const viewUrl = buildDeeplink("/sample-size-calculator", {
|
|
35508
|
+
m: "r",
|
|
35509
|
+
t: "s",
|
|
35510
|
+
// survey mode
|
|
35511
|
+
e: String(params.margin_of_error),
|
|
35512
|
+
c: String(Math.round(params.confidence_level * 100)),
|
|
35513
|
+
z: String(params.population)
|
|
35514
|
+
});
|
|
35194
35515
|
return {
|
|
35195
35516
|
content: [{
|
|
35196
35517
|
type: "text",
|
|
35197
35518
|
text: JSON.stringify({
|
|
35519
|
+
view_url: viewUrl,
|
|
35198
35520
|
required_sample_size: sampleSize,
|
|
35199
35521
|
population: params.population,
|
|
35200
35522
|
margin_of_error: `\xB1${params.margin_of_error}%`,
|
|
35201
35523
|
confidence_level: toPercent(params.confidence_level),
|
|
35202
|
-
|
|
35524
|
+
// Share of the population that must complete the survey — not a response rate.
|
|
35525
|
+
share_of_population_needed: toPercent(sampleSize / params.population),
|
|
35526
|
+
...invitationEstimate(sampleSize, params.population, params.expected_response_rate)
|
|
35203
35527
|
}, null, 2)
|
|
35204
35528
|
}]
|
|
35205
35529
|
};
|
|
@@ -35210,6 +35534,53 @@ function round(value, decimals) {
|
|
|
35210
35534
|
const factor = Math.pow(10, decimals);
|
|
35211
35535
|
return Math.round(value * factor) / factor;
|
|
35212
35536
|
}
|
|
35537
|
+
function superioritySummary(isSignificant, delta) {
|
|
35538
|
+
return isSignificant ? `\u2705 SIGNIFICANT \u2014 The variant ${delta > 0 ? "outperforms" : "underperforms"} the control.` : `\u23F3 NOT SIGNIFICANT \u2014 No statistically significant difference detected.`;
|
|
35539
|
+
}
|
|
35540
|
+
function nonInferioritySummary(isSignificant, marginText) {
|
|
35541
|
+
return isSignificant ? `\u2705 NON-INFERIORITY CONFIRMED \u2014 The variant is not worse than the control by more than ${marginText}.` : `\u23F3 NON-INFERIORITY NOT SHOWN \u2014 We cannot rule out that the variant is worse than the control by more than ${marginText}.`;
|
|
35542
|
+
}
|
|
35543
|
+
function toPercentagePoints(value) {
|
|
35544
|
+
if (!isFinite(value)) return "0 pp";
|
|
35545
|
+
const pts = round(value * 100, 3);
|
|
35546
|
+
return `${pts > 0 ? "+" : ""}${pts} pp`;
|
|
35547
|
+
}
|
|
35548
|
+
function invitationEstimate(sampleSize, population, responseRatePercent) {
|
|
35549
|
+
if (responseRatePercent === void 0) return {};
|
|
35550
|
+
const invitations = Math.ceil(sampleSize / (responseRatePercent / 100));
|
|
35551
|
+
const exceedsPopulation = invitations > population;
|
|
35552
|
+
return {
|
|
35553
|
+
expected_response_rate: `${responseRatePercent}%`,
|
|
35554
|
+
invitations_needed: Math.min(invitations, population),
|
|
35555
|
+
...exceedsPopulation && {
|
|
35556
|
+
warning: `At a ${responseRatePercent}% response rate you would need to invite ${invitations.toLocaleString("en-US")} people, more than the population of ${population.toLocaleString("en-US")}. Even inviting everyone, expect about ${Math.floor(population * responseRatePercent / 100).toLocaleString("en-US")} responses, short of the ${sampleSize.toLocaleString("en-US")} needed. Accept a wider margin of error or improve the response rate.`
|
|
35557
|
+
}
|
|
35558
|
+
};
|
|
35559
|
+
}
|
|
35560
|
+
function formatPValue(p) {
|
|
35561
|
+
if (!isFinite(p)) return "n/a";
|
|
35562
|
+
return p < 1e-4 ? "< 0.0001" : String(round(p, 4));
|
|
35563
|
+
}
|
|
35564
|
+
function formatPValueEq(p) {
|
|
35565
|
+
const display = formatPValue(p);
|
|
35566
|
+
return display.startsWith("<") ? `p ${display}` : `p = ${display}`;
|
|
35567
|
+
}
|
|
35568
|
+
function roundPValue(p) {
|
|
35569
|
+
if (!isFinite(p)) return 0;
|
|
35570
|
+
return p > 0 && p < 1e-6 ? Number(p.toPrecision(3)) : round(p, 6);
|
|
35571
|
+
}
|
|
35572
|
+
function plannedSplitSRM(controlUsers, variantUsers, expectedControlShare) {
|
|
35573
|
+
if (expectedControlShare === void 0) return {};
|
|
35574
|
+
const srm = checkSRM(controlUsers, variantUsers, expectedControlShare);
|
|
35575
|
+
return {
|
|
35576
|
+
sample_ratio_mismatch: {
|
|
35577
|
+
has_mismatch: srm.hasMismatch,
|
|
35578
|
+
p_value: roundPValue(srm.pValue),
|
|
35579
|
+
expected_split: { control: toPercent(expectedControlShare), variant: toPercent(1 - expectedControlShare) },
|
|
35580
|
+
actual_split: { control: toPercent(controlUsers / (controlUsers + variantUsers)), variant: toPercent(variantUsers / (controlUsers + variantUsers)) }
|
|
35581
|
+
}
|
|
35582
|
+
};
|
|
35583
|
+
}
|
|
35213
35584
|
function toPercent(value) {
|
|
35214
35585
|
if (!isFinite(value)) return "0%";
|
|
35215
35586
|
return `${round(value * 100, 2)}%`;
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "abtestresult-mcp",
|
|
3
|
-
"version": "1.0
|
|
3
|
+
"version": "1.3.0",
|
|
4
4
|
"description": "MCP server for A/B test statistical analysis — significance testing, sample size calculation, Bayesian analysis, and more",
|
|
5
5
|
"license": "SEE LICENSE IN LICENSE",
|
|
6
6
|
"type": "module",
|