argonx 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (41) hide show
  1. argonx-0.1.0/LICENSE +21 -0
  2. argonx-0.1.0/MANIFEST.in +6 -0
  3. argonx-0.1.0/PKG-INFO +291 -0
  4. argonx-0.1.0/README.md +268 -0
  5. argonx-0.1.0/argonx/__init__.py +11 -0
  6. argonx-0.1.0/argonx/decision_rules/__init__.py +99 -0
  7. argonx-0.1.0/argonx/decision_rules/composite.py +203 -0
  8. argonx-0.1.0/argonx/decision_rules/engine.py +335 -0
  9. argonx-0.1.0/argonx/decision_rules/guardrails.py +301 -0
  10. argonx-0.1.0/argonx/decision_rules/joint.py +264 -0
  11. argonx-0.1.0/argonx/decision_rules/metrics.py +500 -0
  12. argonx-0.1.0/argonx/experiment.py +790 -0
  13. argonx-0.1.0/argonx/models/__init__.py +86 -0
  14. argonx-0.1.0/argonx/models/base_model.py +91 -0
  15. argonx-0.1.0/argonx/models/binary_model.py +503 -0
  16. argonx-0.1.0/argonx/models/count_model.py +461 -0
  17. argonx-0.1.0/argonx/models/gaussian_model.py +546 -0
  18. argonx-0.1.0/argonx/models/lognormal_model.py +470 -0
  19. argonx-0.1.0/argonx/results/__init__.py +27 -0
  20. argonx-0.1.0/argonx/results/plots.py +781 -0
  21. argonx-0.1.0/argonx/results/result.py +668 -0
  22. argonx-0.1.0/argonx/sequential/__init__.py +26 -0
  23. argonx-0.1.0/argonx/sequential/stopping.py +1051 -0
  24. argonx-0.1.0/argonx.egg-info/PKG-INFO +291 -0
  25. argonx-0.1.0/argonx.egg-info/SOURCES.txt +39 -0
  26. argonx-0.1.0/argonx.egg-info/dependency_links.txt +1 -0
  27. argonx-0.1.0/argonx.egg-info/requires.txt +13 -0
  28. argonx-0.1.0/argonx.egg-info/top_level.txt +2 -0
  29. argonx-0.1.0/pyproject.toml +36 -0
  30. argonx-0.1.0/setup.cfg +4 -0
  31. argonx-0.1.0/tests/integration/test_experiment.py +588 -0
  32. argonx-0.1.0/tests/integration/test_models.py +1197 -0
  33. argonx-0.1.0/tests/test_math.py +250 -0
  34. argonx-0.1.0/tests/unit/test_composite.py +296 -0
  35. argonx-0.1.0/tests/unit/test_engine.py +173 -0
  36. argonx-0.1.0/tests/unit/test_guardrails.py +190 -0
  37. argonx-0.1.0/tests/unit/test_joints.py +201 -0
  38. argonx-0.1.0/tests/unit/test_metrics.py +186 -0
  39. argonx-0.1.0/tests/unit/test_plots.py +571 -0
  40. argonx-0.1.0/tests/unit/test_result.py +498 -0
  41. argonx-0.1.0/tests/unit/test_stopping.py +1384 -0
argonx-0.1.0/LICENSE ADDED
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 souro26
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,6 @@
1
+ prune venv
2
+ prune build
3
+ prune dist
4
+ prune scratch
5
+ prune .pytest_cache
6
+ prune .ruff_cache
argonx-0.1.0/PKG-INFO ADDED
@@ -0,0 +1,291 @@
1
+ Metadata-Version: 2.4
2
+ Name: argonx
3
+ Version: 0.1.0
4
+ Summary: Bayesian decision engine for A/B testing
5
+ Author: Souradeep Roy
6
+ License-Expression: MIT
7
+ Requires-Python: >=3.10
8
+ Description-Content-Type: text/markdown
9
+ License-File: LICENSE
10
+ Requires-Dist: numpy>=1.24
11
+ Requires-Dist: pandas>=1.5
12
+ Requires-Dist: scipy>=1.10
13
+ Requires-Dist: matplotlib>=3.7
14
+ Requires-Dist: pymc>=5.0
15
+ Requires-Dist: arviz>=0.15
16
+ Provides-Extra: dev
17
+ Requires-Dist: pytest; extra == "dev"
18
+ Requires-Dist: ruff; extra == "dev"
19
+ Requires-Dist: pre-commit; extra == "dev"
20
+ Requires-Dist: notebook; extra == "dev"
21
+ Requires-Dist: ipywidgets; extra == "dev"
22
+ Dynamic: license-file
23
+
24
+ # argonx
25
+
26
+ <!-- CI badge — live once GitHub Actions is connected -->
27
+ ![CI](https://github.com/souro26/bayesian-a-b-testing/actions/workflows/ci.yml/badge.svg)
28
+ ![License](https://img.shields.io/badge/license-MIT-blue.svg)
29
+ ![Python](https://img.shields.io/badge/python-3.10%2B-blue.svg)
30
+ <!-- PyPI badges — add after upload -->
31
+ <!-- ![PyPI](https://img.shields.io/pypi/v/argonx.svg) -->
32
+ <!-- ![Downloads](https://img.shields.io/pypi/dm/argonx.svg) -->
33
+
34
+ **argonx** is a decision-support system for A/B experiments. It handles Bayesian inference, multi-metric risk management, hierarchical segment-aware analysis, and sequential stopping — and surfaces a complete evidential picture so the right decision is obvious.
35
+
36
+ Most testing frameworks answer *"is there an effect?"* argonx answers *"what should you do about it, and how much do you lose if you're wrong?"*
37
+
38
+ ---
39
+
40
+ ## Install
41
+
42
+ ```bash
43
+ pip install git+https://github.com/souro26/bayesian-a-b-testing.git
44
+ ```
45
+
46
+ ```bash
47
+ # or, for local development
48
+ git clone https://github.com/souro26/bayesian-a-b-testing.git
49
+ cd bayesian-a-b-testing
50
+ pip install -e .
51
+ ```
52
+
53
+ ---
54
+
55
+ ## Quick Example
56
+
57
+ ```python
58
+ from argonx import Experiment
59
+
60
+ experiment = Experiment(
61
+ data=df,
62
+ variant_col='variant',
63
+ primary_metric='revenue',
64
+ guardrails=['page_load_ms'],
65
+ lower_is_better={'page_load_ms': True},
66
+ model='lognormal',
67
+ guardrail_models={'page_load_ms': 'gaussian'},
68
+ control='control',
69
+ )
70
+
71
+ result = experiment.run()
72
+ result.summary()
73
+ result.plot()
74
+ ```
75
+
76
+ For ratio metrics, pass a callable directly — no class system needed:
77
+
78
+ ```python
79
+ experiment = Experiment(
80
+ data=df,
81
+ variant_col='variant',
82
+ primary_metric=lambda df: df['clicks'] / df['impressions'],
83
+ model='lognormal',
84
+ control='control',
85
+ )
86
+ ```
87
+
88
+ For segment-aware hierarchical inference, add one argument:
89
+
90
+ ```python
91
+ experiment = Experiment(
92
+ data=df,
93
+ variant_col='variant',
94
+ segment_col='device_type', # triggers hierarchical model automatically
95
+ primary_metric='revenue',
96
+ model='lognormal',
97
+ control='control',
98
+ )
99
+
100
+ result = experiment.run()
101
+ result.summary() # aggregate, population-level
102
+ result.segment_summary() # per-segment decisions + cross-segment conflict detection
103
+ ```
104
+
105
+ ---
106
+
107
+ ## What It Computes
108
+
109
+ A p-value tells you the probability of seeing data this extreme if the null is true. It does not tell you what to do. argonx computes the quantities that actually drive decisions:
110
+
111
+ | Metric | What it answers |
112
+ |---|---|
113
+ | **P(variant is best)** | Which variant has the highest posterior probability of being the true winner — computed via simultaneous argmax across all N variants, not pairwise comparison |
114
+ | **Expected loss** | How much you lose on average if you ship the wrong variant — integrated over the full posterior, not a point estimate |
115
+ | **CVaR** | Expected loss in the worst-case tail — catches cases where the average loss looks fine but catastrophic outcomes are possible |
116
+ | **ROPE** | Is the effect large enough to matter in practice? An effect can be statistically real and business-irrelevant. ROPE separates these |
117
+ | **HDI** | The actual probability interval — not a frequentist confidence interval. The lift is inside this range with 95% posterior probability |
118
+ | **Joint probability** | P(all business conditions satisfied simultaneously) — not per-metric checks that miss correlations |
119
+ | **Composite score** | Weighted multi-metric business impact, computed draw-by-draw from posteriors |
120
+ | **Guardrail conflict** | When the primary metric improves and a guardrail degrades, the framework surfaces the conflict clearly rather than resolving it arbitrarily |
121
+ | **Sequential stopping** | Evidence-based stopping signal. Stop when expected loss drops below threshold — not when a fixed sample size is hit |
122
+
123
+ ### Why not just use a t-test?
124
+
125
+ A t-test answers one question: is the observed difference unlikely under the null? It cannot tell you:
126
+
127
+ - How much you lose if you ship and you're wrong
128
+ - Whether the effect is large enough to change user behaviour
129
+ - What to do when conversion improves but latency degrades
130
+ - Whether it's safe to stop the experiment early
131
+ - How thin-segment estimates should borrow strength from larger segments
132
+
133
+ argonx answers all of these. The decision engine is the project — the models are plumbing.
134
+
135
+ ### Why not just use PyMC directly?
136
+
137
+ PyMC gives you posteriors and stops there. It has no concept of which variant to ship, what your business risk tolerance is, or whether your full policy is satisfied simultaneously. argonx is a genuine layer on top of PyMC — not a wrapper, not a replacement.
138
+
139
+ ---
140
+
141
+ ## Models
142
+
143
+ | Model | Use case | Data type |
144
+ |---|---|---|
145
+ | `binary` | Conversion rate, click-through, churn | 0/1 outcomes |
146
+ | `lognormal` | Revenue, order value, session duration | Right-skewed positive continuous |
147
+ | `gaussian` | Latency, load time, scores | Symmetric continuous |
148
+ | `studentt` | Same as gaussian but robust to outliers | Symmetric continuous with heavy tails |
149
+ | `poisson` | Events per user, purchases per session | Count data |
150
+
151
+ Every model has a flat and hierarchical variant. Flat is selected by default. Hierarchical is selected automatically when `segment_col` is provided — no additional configuration required.
152
+
153
+ Guardrail metrics can use a different model than the primary metric:
154
+
155
+ ```python
156
+ experiment = Experiment(
157
+ ...
158
+ model='binary', # primary: conversion rate
159
+ guardrail_models={'page_load_ms': 'lognormal'}, # guardrail: load time
160
+ )
161
+ ```
162
+
163
+ ---
164
+
165
+ ## What `result.summary()` Looks Like
166
+
167
+ ```
168
+ ============================================================
169
+ EXPERIMENT RESULTS
170
+ ============================================================
171
+
172
+ PRIMARY METRIC
173
+ ----------------------------------------
174
+ Best Variant: variant_b
175
+ Expected lift: +4.3% (95% HDI: +1.0% to +7.0%)
176
+ P(best) across all variants: 0.971
177
+
178
+ RISK
179
+ ----------------------------------------
180
+ Expected loss if wrong: 0.0009
181
+ CVaR (95th percentile loss): 0.0021
182
+ Risk level: low
183
+
184
+ PRACTICAL SIGNIFICANCE (ROPE)
185
+ ----------------------------------------
186
+ Effect is OUTSIDE ROPE — practically meaningful.
187
+ P(practical effect): 0.941
188
+
189
+ GUARDRAILS
190
+ ----------------------------------------
191
+ page_load_ms [FAIL] variant=variant_b P(degraded)=0.912 threshold=0.100
192
+
193
+ GUARDRAIL CONFLICTS DETECTED
194
+ ----------------------------------------
195
+ Strong evidence for variant_b on primary metric.
196
+ Guardrail violation on page_load_ms with 91.2% probability.
197
+ Framework cannot resolve this tradeoff. Human review required.
198
+
199
+ ============================================================
200
+ DECISION
201
+ ----------------------------------------
202
+ State: conflict
203
+ Recommendation: REVIEW REQUIRED
204
+ Confidence: low
205
+
206
+ Reasoning:
207
+ - P(best) exceeds strong threshold
208
+ - Expected loss below configured maximum
209
+ - Guardrail violation: page_load_ms cannot be resolved automatically
210
+ ============================================================
211
+ ```
212
+
213
+ The framework does not make the decision. It makes the right decision obvious.
214
+
215
+ ---
216
+
217
+ ## Sequential Stopping
218
+
219
+ ```python
220
+ from argonx.sequential import StoppingChecker
221
+
222
+ checker = StoppingChecker(
223
+ loss_threshold=0.01,
224
+ prob_best_min=0.95,
225
+ min_sample_size=1000,
226
+ )
227
+
228
+ # called at each checkpoint as data accumulates
229
+ status = checker.update(
230
+ samples=result.samples,
231
+ variant_names=['control', 'variant_b'],
232
+ control='control',
233
+ n_users_per_variant=n_counts,
234
+ )
235
+
236
+ print(status.safe_to_stop)
237
+ print(status.users_needed) # approximate additional users if not safe to stop
238
+
239
+ checker.plot_trajectory() # evidence accumulation over time
240
+ ```
241
+
242
+ Bayesian sequential testing is valid at any checkpoint. Frequentist peeking inflates false positive rates — Bayesian expected-loss stopping does not. argonx tells you when evidence is strong enough, not when a predetermined sample size is reached.
243
+
244
+ ---
245
+
246
+ ## Examples
247
+
248
+ Five real-world worked examples in [`examples/`](examples/):
249
+
250
+ | Notebook | Scenario | Key feature |
251
+ |---|---|---|
252
+ | `01_ecommerce_checkout.ipynb` | Checkout redesign | Guardrail conflict: conversion vs. load time |
253
+ | `02_saas_revenue_sequential.ipynb` | SaaS pricing page | Sequential stopping fires early |
254
+ | `03_clinical_trial.ipynb` | Drug dosage protocol | StudentT vs Gaussian: outlier robustness |
255
+ | `04_gaming_matchmaking.ipynb` | Matchmaking algorithm | 3-way multivariant, simultaneous argmax |
256
+ | `05_mobile_personalisation.ipynb` | Fintech personalisation | Hierarchical: iOS wins, Android neutral, thin tablet segment |
257
+
258
+ ---
259
+
260
+ ## Running Tests
261
+
262
+ ```bash
263
+ # Fast — unit tests only, no MCMC (~60 seconds)
264
+ pytest tests/unit/
265
+
266
+ # Statistical property verification — no MCMC
267
+ pytest tests/math/
268
+
269
+ # Full suite including MCMC integration tests (slow)
270
+ pytest tests/
271
+ ```
272
+
273
+ The test suite has three tiers matching the CI pipeline. Unit tests run on every push. Math tests run on every PR. Integration tests run on merge to main.
274
+
275
+ ---
276
+
277
+ ## Contributing
278
+
279
+ Bug reports and PRs are welcome. Before opening a PR:
280
+
281
+ - Run `pytest tests/unit/ tests/math/` and confirm everything passes
282
+ - For changes to the decision engine, add a test to `tests/math/test_decision_sims.py` that verifies the statistical property you're changing
283
+ - For new model variants, add corresponding tests to `tests/integration/test_models.py`
284
+
285
+ Open an issue first for anything beyond bug fixes — architectural changes to the decision engine or new model types are worth discussing before implementation.
286
+
287
+ ---
288
+
289
+ ## License
290
+
291
+ MIT
argonx-0.1.0/README.md ADDED
@@ -0,0 +1,268 @@
1
+ # argonx
2
+
3
+ <!-- CI badge — live once GitHub Actions is connected -->
4
+ ![CI](https://github.com/souro26/bayesian-a-b-testing/actions/workflows/ci.yml/badge.svg)
5
+ ![License](https://img.shields.io/badge/license-MIT-blue.svg)
6
+ ![Python](https://img.shields.io/badge/python-3.10%2B-blue.svg)
7
+ <!-- PyPI badges — add after upload -->
8
+ <!-- ![PyPI](https://img.shields.io/pypi/v/argonx.svg) -->
9
+ <!-- ![Downloads](https://img.shields.io/pypi/dm/argonx.svg) -->
10
+
11
+ **argonx** is a decision-support system for A/B experiments. It handles Bayesian inference, multi-metric risk management, hierarchical segment-aware analysis, and sequential stopping — and surfaces a complete evidential picture so the right decision is obvious.
12
+
13
+ Most testing frameworks answer *"is there an effect?"* argonx answers *"what should you do about it, and how much do you lose if you're wrong?"*
14
+
15
+ ---
16
+
17
+ ## Install
18
+
19
+ ```bash
20
+ pip install git+https://github.com/souro26/bayesian-a-b-testing.git
21
+ ```
22
+
23
+ ```bash
24
+ # or, for local development
25
+ git clone https://github.com/souro26/bayesian-a-b-testing.git
26
+ cd bayesian-a-b-testing
27
+ pip install -e .
28
+ ```
29
+
30
+ ---
31
+
32
+ ## Quick Example
33
+
34
+ ```python
35
+ from argonx import Experiment
36
+
37
+ experiment = Experiment(
38
+ data=df,
39
+ variant_col='variant',
40
+ primary_metric='revenue',
41
+ guardrails=['page_load_ms'],
42
+ lower_is_better={'page_load_ms': True},
43
+ model='lognormal',
44
+ guardrail_models={'page_load_ms': 'gaussian'},
45
+ control='control',
46
+ )
47
+
48
+ result = experiment.run()
49
+ result.summary()
50
+ result.plot()
51
+ ```
52
+
53
+ For ratio metrics, pass a callable directly — no class system needed:
54
+
55
+ ```python
56
+ experiment = Experiment(
57
+ data=df,
58
+ variant_col='variant',
59
+ primary_metric=lambda df: df['clicks'] / df['impressions'],
60
+ model='lognormal',
61
+ control='control',
62
+ )
63
+ ```
64
+
65
+ For segment-aware hierarchical inference, add one argument:
66
+
67
+ ```python
68
+ experiment = Experiment(
69
+ data=df,
70
+ variant_col='variant',
71
+ segment_col='device_type', # triggers hierarchical model automatically
72
+ primary_metric='revenue',
73
+ model='lognormal',
74
+ control='control',
75
+ )
76
+
77
+ result = experiment.run()
78
+ result.summary() # aggregate, population-level
79
+ result.segment_summary() # per-segment decisions + cross-segment conflict detection
80
+ ```
81
+
82
+ ---
83
+
84
+ ## What It Computes
85
+
86
+ A p-value tells you the probability of seeing data this extreme if the null is true. It does not tell you what to do. argonx computes the quantities that actually drive decisions:
87
+
88
+ | Metric | What it answers |
89
+ |---|---|
90
+ | **P(variant is best)** | Which variant has the highest posterior probability of being the true winner — computed via simultaneous argmax across all N variants, not pairwise comparison |
91
+ | **Expected loss** | How much you lose on average if you ship the wrong variant — integrated over the full posterior, not a point estimate |
92
+ | **CVaR** | Expected loss in the worst-case tail — catches cases where the average loss looks fine but catastrophic outcomes are possible |
93
+ | **ROPE** | Is the effect large enough to matter in practice? An effect can be statistically real and business-irrelevant. ROPE separates these |
94
+ | **HDI** | The actual probability interval — not a frequentist confidence interval. The lift is inside this range with 95% posterior probability |
95
+ | **Joint probability** | P(all business conditions satisfied simultaneously) — not per-metric checks that miss correlations |
96
+ | **Composite score** | Weighted multi-metric business impact, computed draw-by-draw from posteriors |
97
+ | **Guardrail conflict** | When the primary metric improves and a guardrail degrades, the framework surfaces the conflict clearly rather than resolving it arbitrarily |
98
+ | **Sequential stopping** | Evidence-based stopping signal. Stop when expected loss drops below threshold — not when a fixed sample size is hit |
99
+
100
+ ### Why not just use a t-test?
101
+
102
+ A t-test answers one question: is the observed difference unlikely under the null? It cannot tell you:
103
+
104
+ - How much you lose if you ship and you're wrong
105
+ - Whether the effect is large enough to change user behaviour
106
+ - What to do when conversion improves but latency degrades
107
+ - Whether it's safe to stop the experiment early
108
+ - How thin-segment estimates should borrow strength from larger segments
109
+
110
+ argonx answers all of these. The decision engine is the project — the models are plumbing.
111
+
112
+ ### Why not just use PyMC directly?
113
+
114
+ PyMC gives you posteriors and stops there. It has no concept of which variant to ship, what your business risk tolerance is, or whether your full policy is satisfied simultaneously. argonx is a genuine layer on top of PyMC — not a wrapper, not a replacement.
115
+
116
+ ---
117
+
118
+ ## Models
119
+
120
+ | Model | Use case | Data type |
121
+ |---|---|---|
122
+ | `binary` | Conversion rate, click-through, churn | 0/1 outcomes |
123
+ | `lognormal` | Revenue, order value, session duration | Right-skewed positive continuous |
124
+ | `gaussian` | Latency, load time, scores | Symmetric continuous |
125
+ | `studentt` | Same as gaussian but robust to outliers | Symmetric continuous with heavy tails |
126
+ | `poisson` | Events per user, purchases per session | Count data |
127
+
128
+ Every model has a flat and hierarchical variant. Flat is selected by default. Hierarchical is selected automatically when `segment_col` is provided — no additional configuration required.
129
+
130
+ Guardrail metrics can use a different model than the primary metric:
131
+
132
+ ```python
133
+ experiment = Experiment(
134
+ ...
135
+ model='binary', # primary: conversion rate
136
+ guardrail_models={'page_load_ms': 'lognormal'}, # guardrail: load time
137
+ )
138
+ ```
139
+
140
+ ---
141
+
142
+ ## What `result.summary()` Looks Like
143
+
144
+ ```
145
+ ============================================================
146
+ EXPERIMENT RESULTS
147
+ ============================================================
148
+
149
+ PRIMARY METRIC
150
+ ----------------------------------------
151
+ Best Variant: variant_b
152
+ Expected lift: +4.3% (95% HDI: +1.0% to +7.0%)
153
+ P(best) across all variants: 0.971
154
+
155
+ RISK
156
+ ----------------------------------------
157
+ Expected loss if wrong: 0.0009
158
+ CVaR (95th percentile loss): 0.0021
159
+ Risk level: low
160
+
161
+ PRACTICAL SIGNIFICANCE (ROPE)
162
+ ----------------------------------------
163
+ Effect is OUTSIDE ROPE — practically meaningful.
164
+ P(practical effect): 0.941
165
+
166
+ GUARDRAILS
167
+ ----------------------------------------
168
+ page_load_ms [FAIL] variant=variant_b P(degraded)=0.912 threshold=0.100
169
+
170
+ GUARDRAIL CONFLICTS DETECTED
171
+ ----------------------------------------
172
+ Strong evidence for variant_b on primary metric.
173
+ Guardrail violation on page_load_ms with 91.2% probability.
174
+ Framework cannot resolve this tradeoff. Human review required.
175
+
176
+ ============================================================
177
+ DECISION
178
+ ----------------------------------------
179
+ State: conflict
180
+ Recommendation: REVIEW REQUIRED
181
+ Confidence: low
182
+
183
+ Reasoning:
184
+ - P(best) exceeds strong threshold
185
+ - Expected loss below configured maximum
186
+ - Guardrail violation: page_load_ms cannot be resolved automatically
187
+ ============================================================
188
+ ```
189
+
190
+ The framework does not make the decision. It makes the right decision obvious.
191
+
192
+ ---
193
+
194
+ ## Sequential Stopping
195
+
196
+ ```python
197
+ from argonx.sequential import StoppingChecker
198
+
199
+ checker = StoppingChecker(
200
+ loss_threshold=0.01,
201
+ prob_best_min=0.95,
202
+ min_sample_size=1000,
203
+ )
204
+
205
+ # called at each checkpoint as data accumulates
206
+ status = checker.update(
207
+ samples=result.samples,
208
+ variant_names=['control', 'variant_b'],
209
+ control='control',
210
+ n_users_per_variant=n_counts,
211
+ )
212
+
213
+ print(status.safe_to_stop)
214
+ print(status.users_needed) # approximate additional users if not safe to stop
215
+
216
+ checker.plot_trajectory() # evidence accumulation over time
217
+ ```
218
+
219
+ Bayesian sequential testing is valid at any checkpoint. Frequentist peeking inflates false positive rates — Bayesian expected-loss stopping does not. argonx tells you when evidence is strong enough, not when a predetermined sample size is reached.
220
+
221
+ ---
222
+
223
+ ## Examples
224
+
225
+ Five real-world worked examples in [`examples/`](examples/):
226
+
227
+ | Notebook | Scenario | Key feature |
228
+ |---|---|---|
229
+ | `01_ecommerce_checkout.ipynb` | Checkout redesign | Guardrail conflict: conversion vs. load time |
230
+ | `02_saas_revenue_sequential.ipynb` | SaaS pricing page | Sequential stopping fires early |
231
+ | `03_clinical_trial.ipynb` | Drug dosage protocol | StudentT vs Gaussian: outlier robustness |
232
+ | `04_gaming_matchmaking.ipynb` | Matchmaking algorithm | 3-way multivariant, simultaneous argmax |
233
+ | `05_mobile_personalisation.ipynb` | Fintech personalisation | Hierarchical: iOS wins, Android neutral, thin tablet segment |
234
+
235
+ ---
236
+
237
+ ## Running Tests
238
+
239
+ ```bash
240
+ # Fast — unit tests only, no MCMC (~60 seconds)
241
+ pytest tests/unit/
242
+
243
+ # Statistical property verification — no MCMC
244
+ pytest tests/math/
245
+
246
+ # Full suite including MCMC integration tests (slow)
247
+ pytest tests/
248
+ ```
249
+
250
+ The test suite has three tiers matching the CI pipeline. Unit tests run on every push. Math tests run on every PR. Integration tests run on merge to main.
251
+
252
+ ---
253
+
254
+ ## Contributing
255
+
256
+ Bug reports and PRs are welcome. Before opening a PR:
257
+
258
+ - Run `pytest tests/unit/ tests/math/` and confirm everything passes
259
+ - For changes to the decision engine, add a test to `tests/math/test_decision_sims.py` that verifies the statistical property you're changing
260
+ - For new model variants, add corresponding tests to `tests/integration/test_models.py`
261
+
262
+ Open an issue first for anything beyond bug fixes — architectural changes to the decision engine or new model types are worth discussing before implementation.
263
+
264
+ ---
265
+
266
+ ## License
267
+
268
+ MIT
@@ -0,0 +1,11 @@
1
+ """
2
+ argonx: Bayesian decision engine for robust A/B testing.
3
+
4
+ This package provides a comprehensive framework for Bayesian experimentation,
5
+ integrating primary metrics with guardrails, sequential stopping rules, and
6
+ hierarchical partial pooling.
7
+ """
8
+
9
+ from .experiment import Experiment
10
+
11
+ __all__ = ["Experiment"]
@@ -0,0 +1,99 @@
1
+ """
2
+ Decision rules and analysis engines for the Bayesian A/B testing framework.
3
+
4
+ This subpackage implements the full analytical pipeline that transforms raw posterior
5
+ samples into actionable experiment outcomes. It covers five distinct concerns:
6
+
7
+ - **Metrics** (`metrics.py`): Core Bayesian decision metrics — P(best), expected loss,
8
+ CVaR, ROPE analysis, and HDI-bounded lift. All metrics are derived from the same
9
+ posterior draws to ensure coherence.
10
+
11
+ - **Guardrails** (`guardrails.py`): Safety constraint evaluation. Computes P(degraded)
12
+ per metric per variant and surfaces conflicts when the primary metric clears its bar
13
+ but a secondary metric fails.
14
+
15
+ - **Joint** (`joint.py`): Joint probability of simultaneously satisfying both the primary
16
+ metric and all selected guardrails, with correlation diagnostics to expose when
17
+ metric co-movement helps or hurts the compound decision.
18
+
19
+ - **Composite** (`composite.py`): Weighted scoring of multiple metrics into a single
20
+ posterior distribution of business value, with asymmetric deterioration weights and
21
+ optional guardrail penalties.
22
+
23
+ - **Engine** (`engine.py`): Orchestration layer. Calls all of the above in dependency
24
+ order and assembles a structured `DecisionResult` with a plain-English recommendation.
25
+ """
26
+
27
+ from .composite import (
28
+ CompositeResult,
29
+ compute_composite_score,
30
+ )
31
+
32
+ from .engine import (
33
+ DecisionResult,
34
+ run_engine,
35
+ )
36
+
37
+ from .guardrails import (
38
+ ConflictResult,
39
+ GuardrailBundle,
40
+ GuardrailResult,
41
+ compute_all_guardrails,
42
+ compute_guardrail,
43
+ )
44
+
45
+ from .joint import (
46
+ JointResult,
47
+ compute_joint_probability,
48
+ )
49
+
50
+ from .metrics import (
51
+ CVaRResult,
52
+ LiftResult,
53
+ LossResult,
54
+ MetricsBundle,
55
+ PBestResult,
56
+ ROPEResult,
57
+ compute_all_metrics,
58
+ compute_cvar,
59
+ compute_expected_loss,
60
+ compute_lift_hdi,
61
+ compute_prob_best,
62
+ compute_rope,
63
+ )
64
+
65
+ __all__ = [
66
+ # Dataclasses — composite
67
+ "CompositeResult",
68
+ # Dataclasses — engine
69
+ "DecisionResult",
70
+ # Dataclasses — guardrails
71
+ "ConflictResult",
72
+ "GuardrailBundle",
73
+ "GuardrailResult",
74
+ # Dataclasses — joint
75
+ "JointResult",
76
+ # Dataclasses — metrics
77
+ "CVaRResult",
78
+ "LiftResult",
79
+ "LossResult",
80
+ "MetricsBundle",
81
+ "PBestResult",
82
+ "ROPEResult",
83
+ # Functions — composite
84
+ "compute_composite_score",
85
+ # Functions — engine
86
+ "run_engine",
87
+ # Functions — guardrails
88
+ "compute_all_guardrails",
89
+ "compute_guardrail",
90
+ # Functions — joint
91
+ "compute_joint_probability",
92
+ # Functions — metrics
93
+ "compute_all_metrics",
94
+ "compute_cvar",
95
+ "compute_expected_loss",
96
+ "compute_lift_hdi",
97
+ "compute_prob_best",
98
+ "compute_rope",
99
+ ]