abtestresult-mcp 1.0.8 → 1.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +11 -2
- package/dist/index.js +89 -1
- package/package.json +1 -1
package/README.md
CHANGED
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
|
|
3
3
|
Statistical tools for A/B testing, available as a [Model Context Protocol](https://modelcontextprotocol.io) (MCP) server. Powered by [abtestresult.com](https://abtestresult.com).
|
|
4
4
|
|
|
5
|
-
Give any AI assistant (Claude, Cursor, VS Code, etc.) the ability to analyze A/B tests, calculate sample sizes, run Bayesian analysis, and more. All calculations run locally on your machine — no API keys, your test data never leaves your device.
|
|
5
|
+
Give any AI assistant (Claude, Codex, Cursor, VS Code, etc.) the ability to analyze A/B tests, calculate sample sizes, run Bayesian analysis, and more. All calculations run locally on your machine — no API keys, your test data never leaves your device.
|
|
6
6
|
|
|
7
7
|
This package ships readable source. It is source-available for local personal use only and is not licensed for commercial use.
|
|
8
8
|
|
|
@@ -23,7 +23,8 @@ Add to your MCP configuration:
|
|
|
23
23
|
|
|
24
24
|
**Where to add this:**
|
|
25
25
|
|
|
26
|
-
- **Claude Desktop:**
|
|
26
|
+
- **Claude Desktop:** `~/Library/Application Support/Claude/claude_desktop_config.json`
|
|
27
|
+
- **Codex:** `~/.codex/config.toml`
|
|
27
28
|
- **Claude Code:** `.mcp.json` in your project root
|
|
28
29
|
- **Cursor:** `.cursor/mcp.json`
|
|
29
30
|
- **VS Code:** `.vscode/settings.json` (use `mcp.servers` instead of `mcpServers`)
|
|
@@ -43,6 +44,14 @@ That's it. No API keys, no authentication. The server starts automatically when
|
|
|
43
44
|
| `paired_test` | Paired T-test or Wilcoxon signed-rank for before/after data |
|
|
44
45
|
| `survey_sample_size` | Cochran's formula with finite population correction |
|
|
45
46
|
|
|
47
|
+
## Shareable result links
|
|
48
|
+
|
|
49
|
+
Every analysis tool (except `paired_test`) returns a `view_url` in its response — a
|
|
50
|
+
deeplink to [abtestresult.com](https://abtestresult.com) that pre-fills the matching
|
|
51
|
+
calculator with your inputs and auto-runs it. Open it to inspect the result visually
|
|
52
|
+
(charts, confidence intervals, recommendations) or share it with your team. No data is
|
|
53
|
+
stored server-side: the inputs are encoded directly into the URL.
|
|
54
|
+
|
|
46
55
|
## Usage Examples
|
|
47
56
|
|
|
48
57
|
Once connected, just ask your AI assistant in natural language:
|
package/dist/index.js
CHANGED
|
@@ -34719,9 +34719,29 @@ function trackToolUsage(toolName) {
|
|
|
34719
34719
|
}
|
|
34720
34720
|
var server = new McpServer({
|
|
34721
34721
|
name: "ABTestResult",
|
|
34722
|
-
version: "1.
|
|
34722
|
+
version: "1.1.0",
|
|
34723
34723
|
description: "Statistical tools for A/B testing \u2014 significance testing, sample size calculation, Bayesian analysis, and more. Powered by abtestresult.com"
|
|
34724
34724
|
});
|
|
34725
|
+
var SITE_URL = (process.env.ABTESTRESULT_BASE_URL || "https://abtestresult.com").replace(/\/+$/, "");
|
|
34726
|
+
function toBase64Url(str) {
|
|
34727
|
+
return Buffer.from(str, "utf-8").toString("base64url");
|
|
34728
|
+
}
|
|
34729
|
+
function buildDeeplink(path, compact) {
|
|
34730
|
+
const clean = {};
|
|
34731
|
+
for (const [k, v] of Object.entries(compact)) {
|
|
34732
|
+
if (v !== void 0 && v !== null && v !== "") clean[k] = v;
|
|
34733
|
+
}
|
|
34734
|
+
return `${SITE_URL}${path}?d=${toBase64Url(JSON.stringify(clean))}`;
|
|
34735
|
+
}
|
|
34736
|
+
function pipeJoin(parts) {
|
|
34737
|
+
const arr = parts.map((p) => p === void 0 || p === null ? "" : String(p));
|
|
34738
|
+
while (arr.length > 0 && arr[arr.length - 1] === "") arr.pop();
|
|
34739
|
+
return arr.join("|");
|
|
34740
|
+
}
|
|
34741
|
+
function tailCode(testType, niMargin) {
|
|
34742
|
+
if (niMargin > 0) return "n";
|
|
34743
|
+
return testType === "one-sided" ? "1" : "2";
|
|
34744
|
+
}
|
|
34725
34745
|
server.tool(
|
|
34726
34746
|
"analyze_ab_test",
|
|
34727
34747
|
`Analyze an A/B test with rate/proportion metrics (e.g. conversion rate, click-through rate).
|
|
@@ -34752,11 +34772,22 @@ Example: "Control had 5000 visitors with 250 conversions, variant had 5000 visit
|
|
|
34752
34772
|
params.non_inferiority_margin
|
|
34753
34773
|
);
|
|
34754
34774
|
const summary = result.isSignificant ? `\u2705 SIGNIFICANT \u2014 The variant ${result.delta > 0 ? "outperforms" : "underperforms"} the control.` : `\u23F3 NOT SIGNIFICANT \u2014 No statistically significant difference detected.`;
|
|
34775
|
+
const ctrlRate = (params.control_conversions / params.control_users * 100).toFixed(2);
|
|
34776
|
+
const varRate = (params.variant_conversions / params.variant_users * 100).toFixed(2);
|
|
34777
|
+
const viewUrl = buildDeeplink("/statistical-significance-calculator", {
|
|
34778
|
+
c: pipeJoin([params.control_users, params.control_conversions, ctrlRate]),
|
|
34779
|
+
v: pipeJoin([params.variant_users, params.variant_conversions, varRate]),
|
|
34780
|
+
m: "r",
|
|
34781
|
+
t: tailCode(params.test_type, params.non_inferiority_margin),
|
|
34782
|
+
l: String(Math.round(params.confidence_level * 100)),
|
|
34783
|
+
n: params.non_inferiority_margin > 0 ? String(params.non_inferiority_margin) : ""
|
|
34784
|
+
});
|
|
34755
34785
|
return {
|
|
34756
34786
|
content: [{
|
|
34757
34787
|
type: "text",
|
|
34758
34788
|
text: JSON.stringify({
|
|
34759
34789
|
summary,
|
|
34790
|
+
view_url: viewUrl,
|
|
34760
34791
|
significant: result.isSignificant,
|
|
34761
34792
|
p_value: round(result.pValue, 6),
|
|
34762
34793
|
z_score: round(result.zScore, 4),
|
|
@@ -34817,11 +34848,20 @@ Example: "Control had 1000 users with mean $45.20 (std $12.50), variant had 1000
|
|
|
34817
34848
|
params.non_inferiority_margin
|
|
34818
34849
|
);
|
|
34819
34850
|
const summary = result.isSignificant ? `\u2705 SIGNIFICANT \u2014 The variant ${result.delta > 0 ? "outperforms" : "underperforms"} the control.` : `\u23F3 NOT SIGNIFICANT \u2014 No statistically significant difference detected.`;
|
|
34851
|
+
const viewUrl = buildDeeplink("/statistical-significance-calculator", {
|
|
34852
|
+
c: pipeJoin([params.control_users, params.control_mean, "", params.control_std_dev]),
|
|
34853
|
+
v: pipeJoin([params.variant_users, params.variant_mean, "", params.variant_std_dev]),
|
|
34854
|
+
m: "a",
|
|
34855
|
+
t: tailCode(params.test_type, params.non_inferiority_margin),
|
|
34856
|
+
l: String(Math.round(params.confidence_level * 100)),
|
|
34857
|
+
n: params.non_inferiority_margin > 0 ? String(params.non_inferiority_margin) : ""
|
|
34858
|
+
});
|
|
34820
34859
|
return {
|
|
34821
34860
|
content: [{
|
|
34822
34861
|
type: "text",
|
|
34823
34862
|
text: JSON.stringify({
|
|
34824
34863
|
summary,
|
|
34864
|
+
view_url: viewUrl,
|
|
34825
34865
|
significant: result.isSignificant,
|
|
34826
34866
|
p_value: round(result.pValue, 6),
|
|
34827
34867
|
t_score: round(result.tScore, 4),
|
|
@@ -34910,10 +34950,21 @@ Example: "How many users do I need to detect a 10% relative lift on a 5% baselin
|
|
|
34910
34950
|
params.daily_traffic ?? null
|
|
34911
34951
|
);
|
|
34912
34952
|
}
|
|
34953
|
+
const viewUrl = buildDeeplink("/sample-size-calculator", {
|
|
34954
|
+
m: params.metric_type === "rate" ? "r" : "a",
|
|
34955
|
+
t: params.test_type === "one-sided" ? "1" : "2",
|
|
34956
|
+
b: String(params.baseline),
|
|
34957
|
+
d: String(params.mde),
|
|
34958
|
+
s: params.metric_type === "average" && params.std_dev ? String(params.std_dev) : "",
|
|
34959
|
+
c: String(Math.round(params.confidence_level * 100)),
|
|
34960
|
+
p: String(Math.round(params.power * 100)),
|
|
34961
|
+
v: params.num_variants !== 1 ? String(params.num_variants) : ""
|
|
34962
|
+
});
|
|
34913
34963
|
return {
|
|
34914
34964
|
content: [{
|
|
34915
34965
|
type: "text",
|
|
34916
34966
|
text: JSON.stringify({
|
|
34967
|
+
view_url: viewUrl,
|
|
34917
34968
|
samples_per_group: result.samplesEach,
|
|
34918
34969
|
total_samples: result.samplesTotal,
|
|
34919
34970
|
number_of_groups: params.num_variants + 1,
|
|
@@ -34977,10 +35028,19 @@ Example: "I have 50,000 total users and a 5% conversion rate. What's my MDE?"`,
|
|
|
34977
35028
|
);
|
|
34978
35029
|
const baselineValue = params.metric_type === "rate" ? params.baseline / 100 : params.baseline;
|
|
34979
35030
|
const mdeRelative = baselineValue !== 0 ? mdeAbsolute / baselineValue : 0;
|
|
35031
|
+
const viewUrl = buildDeeplink("/mde-calculator", {
|
|
35032
|
+
t: String(params.total_traffic),
|
|
35033
|
+
s: params.std_dev ? String(params.std_dev) : "",
|
|
35034
|
+
r: String(params.baseline),
|
|
35035
|
+
p: String(params.power),
|
|
35036
|
+
c: String(params.confidence_level),
|
|
35037
|
+
m: params.metric_type === "average" ? "c" : "r"
|
|
35038
|
+
});
|
|
34980
35039
|
return {
|
|
34981
35040
|
content: [{
|
|
34982
35041
|
type: "text",
|
|
34983
35042
|
text: JSON.stringify({
|
|
35043
|
+
view_url: viewUrl,
|
|
34984
35044
|
mde_absolute: round(mdeAbsolute, 6),
|
|
34985
35045
|
mde_relative_percent: toPercent(mdeRelative),
|
|
34986
35046
|
interpretation: params.metric_type === "rate" ? `You can detect an absolute change of ${toPercent(mdeAbsolute)} (${toPercent(mdeRelative)} relative lift) from the ${params.baseline}% baseline.` : `You can detect an absolute change of ${round(mdeAbsolute, 4)} (${toPercent(mdeRelative)} relative lift) from the ${params.baseline} baseline.`,
|
|
@@ -35049,11 +35109,25 @@ Example: "Control: 5000 visitors, 250 conversions. Variant: 5000 visitors, 300 c
|
|
|
35049
35109
|
}));
|
|
35050
35110
|
const bestIdx = result.probabilityToBeBest.indexOf(Math.max(...result.probabilityToBeBest));
|
|
35051
35111
|
const summary = result.probabilityToBeBest[bestIdx] > 0.95 ? `\u{1F3C6} ${params.variants[bestIdx].name} is the winner with ${toPercent(result.probabilityToBeBest[bestIdx])} probability of being best.` : result.probabilityToBeBest[bestIdx] > 0.8 ? `\u{1F4CA} ${params.variants[bestIdx].name} is leading with ${toPercent(result.probabilityToBeBest[bestIdx])} probability, but more data may be needed.` : `\u23F3 No clear winner yet. The leading variant has only ${toPercent(result.probabilityToBeBest[bestIdx])} probability of being best.`;
|
|
35112
|
+
const isRate = params.metric_type === "rate";
|
|
35113
|
+
const bayesPart = (v, withName) => {
|
|
35114
|
+
const rate = isRate && v.conversions !== void 0 && v.users ? (v.conversions / v.users * 100).toFixed(2) : "";
|
|
35115
|
+
return pipeJoin(isRate ? [v.users, v.conversions, rate, "", withName ? v.name : ""] : [v.users, v.mean, "", v.std_dev, withName ? v.name : ""]);
|
|
35116
|
+
};
|
|
35117
|
+
const viewUrl = buildDeeplink("/bayesian-calculator", {
|
|
35118
|
+
c: bayesPart(params.variants[0], false),
|
|
35119
|
+
v: params.variants.slice(1).map((v) => bayesPart(v, true)).join(";"),
|
|
35120
|
+
m: isRate ? "r" : "a",
|
|
35121
|
+
// alpha=beta=1 is the uninformative prior; anything else maps to "weak".
|
|
35122
|
+
p: params.prior_alpha === 1 && params.prior_beta === 1 ? "u" : "w",
|
|
35123
|
+
l: String(Math.round(params.credibility * 100))
|
|
35124
|
+
});
|
|
35052
35125
|
return {
|
|
35053
35126
|
content: [{
|
|
35054
35127
|
type: "text",
|
|
35055
35128
|
text: JSON.stringify({
|
|
35056
35129
|
summary,
|
|
35130
|
+
view_url: viewUrl,
|
|
35057
35131
|
variants: variantResults,
|
|
35058
35132
|
lift_distribution: {
|
|
35059
35133
|
mean: round(result.liftDistribution.mean, 6),
|
|
@@ -35095,11 +35169,16 @@ Example: "Control has 10,234 users, variant has 9,766 users"`,
|
|
|
35095
35169
|
const total = params.control_users + params.variant_users;
|
|
35096
35170
|
const actualRatio = params.control_users / total;
|
|
35097
35171
|
const summary = result.hasMismatch ? `\u{1F6A8} SRM DETECTED (p=${round(result.pValue, 6)}) \u2014 The traffic split is ${toPercent(actualRatio)}/${toPercent(1 - actualRatio)} instead of the expected 50/50. Your experiment may have a bug. DO NOT trust the results.` : `\u2705 No SRM detected (p=${round(result.pValue, 4)}) \u2014 The traffic split of ${toPercent(actualRatio)}/${toPercent(1 - actualRatio)} is consistent with a 50/50 split.`;
|
|
35172
|
+
const viewUrl = buildDeeplink("/sample-ratio-mismatch", {
|
|
35173
|
+
c: String(params.control_users),
|
|
35174
|
+
v: String(params.variant_users)
|
|
35175
|
+
});
|
|
35098
35176
|
return {
|
|
35099
35177
|
content: [{
|
|
35100
35178
|
type: "text",
|
|
35101
35179
|
text: JSON.stringify({
|
|
35102
35180
|
summary,
|
|
35181
|
+
view_url: viewUrl,
|
|
35103
35182
|
has_mismatch: result.hasMismatch,
|
|
35104
35183
|
p_value: round(result.pValue, 6),
|
|
35105
35184
|
chi_square: round(result.chiSquare, 4),
|
|
@@ -35191,10 +35270,19 @@ Example: "How many responses do I need from a population of 10,000 with 5% margi
|
|
|
35191
35270
|
params.margin_of_error / 100,
|
|
35192
35271
|
params.confidence_level
|
|
35193
35272
|
);
|
|
35273
|
+
const viewUrl = buildDeeplink("/sample-size-calculator", {
|
|
35274
|
+
m: "r",
|
|
35275
|
+
t: "s",
|
|
35276
|
+
// survey mode
|
|
35277
|
+
e: String(params.margin_of_error),
|
|
35278
|
+
c: String(Math.round(params.confidence_level * 100)),
|
|
35279
|
+
z: String(params.population)
|
|
35280
|
+
});
|
|
35194
35281
|
return {
|
|
35195
35282
|
content: [{
|
|
35196
35283
|
type: "text",
|
|
35197
35284
|
text: JSON.stringify({
|
|
35285
|
+
view_url: viewUrl,
|
|
35198
35286
|
required_sample_size: sampleSize,
|
|
35199
35287
|
population: params.population,
|
|
35200
35288
|
margin_of_error: `\xB1${params.margin_of_error}%`,
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "abtestresult-mcp",
|
|
3
|
-
"version": "1.0
|
|
3
|
+
"version": "1.1.0",
|
|
4
4
|
"description": "MCP server for A/B test statistical analysis — significance testing, sample size calculation, Bayesian analysis, and more",
|
|
5
5
|
"license": "SEE LICENSE IN LICENSE",
|
|
6
6
|
"type": "module",
|