vigipy 3.0.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- vigipy-3.0.0/LICENSE +21 -0
- vigipy-3.0.0/PKG-INFO +352 -0
- vigipy-3.0.0/README.md +328 -0
- vigipy-3.0.0/pyproject.toml +64 -0
- vigipy-3.0.0/setup.cfg +4 -0
- vigipy-3.0.0/setup.py +5 -0
- vigipy-3.0.0/src/vigipy/BCPNN/BCPNN.py +197 -0
- vigipy-3.0.0/src/vigipy/BCPNN/__init__.py +1 -0
- vigipy-3.0.0/src/vigipy/GPS/GPS.py +346 -0
- vigipy-3.0.0/src/vigipy/GPS/__init__.py +1 -0
- vigipy-3.0.0/src/vigipy/LASSO/LASSO.py +169 -0
- vigipy-3.0.0/src/vigipy/LASSO/__init__.py +1 -0
- vigipy-3.0.0/src/vigipy/LongitudinalModel/LongitudinalModel.py +165 -0
- vigipy-3.0.0/src/vigipy/LongitudinalModel/__init__.py +0 -0
- vigipy-3.0.0/src/vigipy/PRR/PRR.py +81 -0
- vigipy-3.0.0/src/vigipy/PRR/__init__.py +1 -0
- vigipy-3.0.0/src/vigipy/RFET/RFET.py +87 -0
- vigipy-3.0.0/src/vigipy/RFET/__init__.py +1 -0
- vigipy-3.0.0/src/vigipy/ROR/ROR.py +81 -0
- vigipy-3.0.0/src/vigipy/ROR/__init__.py +1 -0
- vigipy-3.0.0/src/vigipy/__init__.py +27 -0
- vigipy-3.0.0/src/vigipy/analyze.py +104 -0
- vigipy-3.0.0/src/vigipy/config.py +211 -0
- vigipy-3.0.0/src/vigipy/utils/Container.py +108 -0
- vigipy-3.0.0/src/vigipy/utils/__init__.py +3 -0
- vigipy-3.0.0/src/vigipy/utils/common.py +281 -0
- vigipy-3.0.0/src/vigipy/utils/data_prep.py +243 -0
- vigipy-3.0.0/src/vigipy/utils/distribution_funcs/__init__.py +0 -0
- vigipy-3.0.0/src/vigipy/utils/distribution_funcs/quantile_funcs.py +37 -0
- vigipy-3.0.0/src/vigipy/utils/expectations.py +163 -0
- vigipy-3.0.0/src/vigipy/utils/lbe.py +113 -0
- vigipy-3.0.0/src/vigipy/utils/types.py +14 -0
- vigipy-3.0.0/src/vigipy.egg-info/PKG-INFO +352 -0
- vigipy-3.0.0/src/vigipy.egg-info/SOURCES.txt +37 -0
- vigipy-3.0.0/src/vigipy.egg-info/dependency_links.txt +1 -0
- vigipy-3.0.0/src/vigipy.egg-info/requires.txt +14 -0
- vigipy-3.0.0/src/vigipy.egg-info/top_level.txt +1 -0
- vigipy-3.0.0/test/test_data_prep.py +79 -0
- vigipy-3.0.0/test/test_methods.py +608 -0
vigipy-3.0.0/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 David Beery
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
vigipy-3.0.0/PKG-INFO
ADDED
|
@@ -0,0 +1,352 @@
|
|
|
1
|
+
Metadata-Version: 2.1
|
|
2
|
+
Name: vigipy
|
|
3
|
+
Version: 3.0.0
|
|
4
|
+
Summary: A Python library for disproportionality analyses
|
|
5
|
+
Author-email: David Beery <shakesbeery@gmail.com>
|
|
6
|
+
License: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/Shakesbeery/vigipy
|
|
8
|
+
Project-URL: Bug Tracker, https://github.com/Shakesbeery/vigipy/issues
|
|
9
|
+
Classifier: Programming Language :: Python :: 3
|
|
10
|
+
Classifier: Programming Language :: Python :: 3.9
|
|
11
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
12
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
13
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
14
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
15
|
+
Classifier: Operating System :: OS Independent
|
|
16
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
17
|
+
Classifier: Intended Audience :: Science/Research
|
|
18
|
+
Classifier: Topic :: Scientific/Engineering :: Medical Science Apps.
|
|
19
|
+
Requires-Python: >=3.9
|
|
20
|
+
Description-Content-Type: text/markdown
|
|
21
|
+
Provides-Extra: excel
|
|
22
|
+
Provides-Extra: dev
|
|
23
|
+
License-File: LICENSE
|
|
24
|
+
|
|
25
|
+
# vigipy
|
|
26
|
+
|
|
27
|
+
> [!IMPORTANT]
|
|
28
|
+
> **Major Release — `vigipy` v3.0 is live!**
|
|
29
|
+
> This major release introduces an extensive modernization of the library:
|
|
30
|
+
> - **Unified Interface**: Typed execution via `analyze()`, `analyze_all()`, and configuration dataclasses (`PRRConfig`, `RORConfig`, `RFETConfig`, `BCPNNConfig`, `GPSConfig`, `LASSOConfig`).
|
|
31
|
+
> - **Statistical & Mathematical Rigor**: Corrected FDR step-up monotonicity, decision-theoretic Bayesian metrics (FDR, FNR, FOR, Se, Sp), RFET mid-$p$ parameter order, and Haldane-Anscombe (+0.5) zero-cell continuity corrections.
|
|
32
|
+
> - **Vectorized Performance**: Fully vectorized core operations, replacing SymPy with native SciPy C-routines for >50% speedups.
|
|
33
|
+
> - **Permissive MIT License**: Formally relicensed under the MIT License.
|
|
34
|
+
>
|
|
35
|
+
> *See the updated API documentation and examples below.*
|
|
36
|
+
|
|
37
|
+
`vigipy` is a Python library bringing modern disproportionality analyses and pharmacovigilance techniques into the Python ecosystem with a clean, intuitive, and type-safe interface. Core disproportionality methods are adapted and extended from Ismail Ahmed and Antoine Poncet's [PhViD](https://cran.r-project.org/web/packages/PhViD/index.html) package, fully vectorized with native NumPy and SciPy routines.
|
|
38
|
+
|
|
39
|
+
### Top-level Functions & Classes:
|
|
40
|
+
|
|
41
|
+
* **Unified Interface**:
|
|
42
|
+
* `analyze()` - Execute any disproportionality analysis via typed configuration dataclasses
|
|
43
|
+
* `analyze_all()` - Run multiple analysis methods in a single call with shared or distinct parameters
|
|
44
|
+
* `PRRConfig`, `RORConfig`, `RFETConfig`, `BCPNNConfig`, `GPSConfig`, `LASSOConfig` - Type-safe configuration dataclasses
|
|
45
|
+
* **Disproportionality Methods**:
|
|
46
|
+
* `prr()` - Proportional Reporting Ratio (frequentist log-normal approximation)
|
|
47
|
+
* `ror()` - Reporting Odds Ratio (Woolf log-odds approximation)
|
|
48
|
+
* `rfet()` - Reporting Fisher's Exact Test (exact hypergeometric p-values with optional mid-p correction)
|
|
49
|
+
* `bcpnn()` - Bayesian Confidence Propagation Neural Network (analytical or Dirichlet Monte Carlo Information Component)
|
|
50
|
+
* `gps()` - Multi-item Gamma Poisson Shrinker (Empirical Bayes bivariate mixture model)
|
|
51
|
+
* `lasso()` - LASSO regression for multivariate signal detection and confounding adjustment
|
|
52
|
+
* **Longitudinal Modeling**:
|
|
53
|
+
* `LongitudinalModel()` - Apply any analysis method over time to evaluate cumulative or disjoint signal evolution
|
|
54
|
+
* **Data Preparation**:
|
|
55
|
+
* `convert()` - Convert adverse event and product count tables into a structured `DataContainer`
|
|
56
|
+
* `convert_binary()` - Generate binary product feature matrices and event outcome matrices for LASSO
|
|
57
|
+
* `convert_multi_item()` - Aggregate co-occurring product columns into multi-item interaction tables
|
|
58
|
+
* **Result & Data Containers**:
|
|
59
|
+
* `AnalysisResult` - Structured container for `signals`, `all_signals`, `num_signals`, and model `params`, with `.export()` to Excel or CSV
|
|
60
|
+
* `DataContainer` - Typed container holding contingency, event, and product matrices
|
|
61
|
+
|
|
62
|
+
---
|
|
63
|
+
|
|
64
|
+
## Getting Started
|
|
65
|
+
|
|
66
|
+
### Dependencies
|
|
67
|
+
|
|
68
|
+
`vigipy` requires Python 3.9+ and modern scientific computing libraries:
|
|
69
|
+
|
|
70
|
+
* `pandas>=2.0`
|
|
71
|
+
* `numpy>=1.24,<3`
|
|
72
|
+
* `scipy>=1.10`
|
|
73
|
+
* `scikit-learn>=1.3`
|
|
74
|
+
* `statsmodels>=0.14`
|
|
75
|
+
|
|
76
|
+
Optional dependencies:
|
|
77
|
+
* `openpyxl>=3.0.0` (required for exporting results directly to Excel `.xlsx` spreadsheets)
|
|
78
|
+
|
|
79
|
+
### Installation
|
|
80
|
+
|
|
81
|
+
Install `vigipy` from source or local checkout:
|
|
82
|
+
|
|
83
|
+
```bash
|
|
84
|
+
pip install .
|
|
85
|
+
```
|
|
86
|
+
|
|
87
|
+
To include Excel export capabilities:
|
|
88
|
+
|
|
89
|
+
```bash
|
|
90
|
+
pip install ".[excel]"
|
|
91
|
+
```
|
|
92
|
+
|
|
93
|
+
For development (includes test and lint tools):
|
|
94
|
+
|
|
95
|
+
```bash
|
|
96
|
+
pip install -e ".[dev]"
|
|
97
|
+
```
|
|
98
|
+
|
|
99
|
+
### Running Tests
|
|
100
|
+
|
|
101
|
+
Run the full test suite using `pytest`:
|
|
102
|
+
|
|
103
|
+
```bash
|
|
104
|
+
pytest test/ -v
|
|
105
|
+
```
|
|
106
|
+
|
|
107
|
+
---
|
|
108
|
+
|
|
109
|
+
## Usage
|
|
110
|
+
|
|
111
|
+
### Unified API (Recommended)
|
|
112
|
+
|
|
113
|
+
The unified interface provides type safety, autocompletion, and consistent result structures across all methods:
|
|
114
|
+
|
|
115
|
+
```python
|
|
116
|
+
import pandas as pd
|
|
117
|
+
from vigipy import convert, analyze, analyze_all, PRRConfig, BCPNNConfig, GPSConfig
|
|
118
|
+
|
|
119
|
+
# 1. Load data and convert to a DataContainer
|
|
120
|
+
df = pd.read_csv("AE_count_data.csv")
|
|
121
|
+
data = convert(df, product_label="name", ae_label="AE", count_label="count")
|
|
122
|
+
|
|
123
|
+
# 2. Run a single method with typed configuration
|
|
124
|
+
result = analyze(data, PRRConfig(min_events=3, decision_metric="fdr", fdr_threshold=0.05))
|
|
125
|
+
|
|
126
|
+
print(f"Detected {result.num_signals} signals")
|
|
127
|
+
print(result.signals.head())
|
|
128
|
+
|
|
129
|
+
# Export both significant signals and full dataset to Excel or CSV
|
|
130
|
+
result.export("prr_signals.xlsx") # Creates 'Signals' and 'all_data' sheets
|
|
131
|
+
result.export("prr_signals.csv") # Exports detected signals to CSV
|
|
132
|
+
|
|
133
|
+
# 3. Batch comparison across all methods in one call
|
|
134
|
+
batch_results = analyze_all(data, min_events=3, decision_metric="rank")
|
|
135
|
+
for method_name, res in batch_results.items():
|
|
136
|
+
print(f"{method_name.upper()}: {res.num_signals} signals detected")
|
|
137
|
+
|
|
138
|
+
# 4. Iterate over custom configurations
|
|
139
|
+
configs = [
|
|
140
|
+
PRRConfig(min_events=5, ranking_statistic="CI"),
|
|
141
|
+
BCPNNConfig(min_events=5, ranking_statistic="quantile"),
|
|
142
|
+
GPSConfig(min_events=5, ranking_statistic="log2"),
|
|
143
|
+
]
|
|
144
|
+
for cfg in configs:
|
|
145
|
+
res = analyze(data, cfg)
|
|
146
|
+
print(f"{cfg.method}: {res.num_signals} signals")
|
|
147
|
+
```
|
|
148
|
+
|
|
149
|
+
### Classic Function API
|
|
150
|
+
|
|
151
|
+
Direct function calls are fully supported with identical return structures:
|
|
152
|
+
|
|
153
|
+
```python
|
|
154
|
+
import pandas as pd
|
|
155
|
+
from vigipy import convert, prr, ror, rfet, bcpnn, gps
|
|
156
|
+
|
|
157
|
+
df = pd.read_csv("AE_count_data.csv")
|
|
158
|
+
data = convert(df)
|
|
159
|
+
|
|
160
|
+
# Frequentist: Reporting Odds Ratio with Haldane-Anscombe continuity correction
|
|
161
|
+
ror_res = ror(data, min_events=3, decision_metric="fdr", fdr_threshold=0.05)
|
|
162
|
+
|
|
163
|
+
# Fisher's Exact Test with Lancaster mid-p adjustment
|
|
164
|
+
rfet_res = rfet(data, min_events=3, mid_pval=True)
|
|
165
|
+
|
|
166
|
+
# Bayesian Confidence Propagation Neural Network
|
|
167
|
+
bcpnn_res = bcpnn(data, min_events=3, ranking_statistic="quantile")
|
|
168
|
+
|
|
169
|
+
# Empirical Bayes: Gamma Poisson Shrinker
|
|
170
|
+
gps_res = gps(data, min_events=5, decision_metric="rank", ranking_statistic="log2")
|
|
171
|
+
|
|
172
|
+
# Access results
|
|
173
|
+
print(gps_res.signals[["Product", "Adverse Event", "Count", "quantile", "fdr"]].head())
|
|
174
|
+
gps_res.export("gps_signals.xlsx")
|
|
175
|
+
```
|
|
176
|
+
|
|
177
|
+
---
|
|
178
|
+
|
|
179
|
+
## Decision Rules & Ranking Statistics
|
|
180
|
+
|
|
181
|
+
`vigipy` standardizes signal identification across frequentist and Bayesian methods:
|
|
182
|
+
|
|
183
|
+
### Decision Metrics (`decision_metric`)
|
|
184
|
+
* `"fdr"` - Controls the False Discovery Rate at `decision_thres` (default: `0.05`) using Local Bayes Estimation (LBE) or cumulative posterior null probabilities.
|
|
185
|
+
* `"rank"` - Retains signals where the ranking statistic meets `decision_thres`. For p-values, selects values $\le \text{threshold}$; for confidence/credible bounds, selects values $\ge \text{threshold}$.
|
|
186
|
+
* `"signals"` - Selects the top $N$ ranked candidates up to `decision_thres`.
|
|
187
|
+
|
|
188
|
+
### Ranking Statistics (`ranking_statistic`)
|
|
189
|
+
| Method | Supported Statistics | Notes |
|
|
190
|
+
| :--- | :--- | :--- |
|
|
191
|
+
| **PRR / ROR** | `"p_value"`, `"CI"` | `"CI"` ranks by the lower bound of the 95% confidence interval. |
|
|
192
|
+
| **RFET** | `"p_value"` | Exact hypergeometric p-value (supports `mid_pval=True`). |
|
|
193
|
+
| **BCPNN** | `"quantile"`, `"p_value"` | `"quantile"` ranks by $IC_{025}$ (lower 95% credible bound). |
|
|
194
|
+
| **GPS** | `"log2"`, `"quantile"`, `"p_value"` | `"log2"` ranks by $EB_{05}$ of $\log_2(\lambda)$, shrinked towards expected counts. |
|
|
195
|
+
|
|
196
|
+
---
|
|
197
|
+
|
|
198
|
+
## Expected Count Calculations & Dispersion Testing
|
|
199
|
+
|
|
200
|
+
Expected counts ($E$) model the baseline event frequency under the null hypothesis of no association. `vigipy` supports three expectation models:
|
|
201
|
+
|
|
202
|
+
1. `"mantel-haentzel"` (default): Standard independence assumption, $E_{ij} = \frac{n_{i\cdot} n_{\cdot j}}{N}$.
|
|
203
|
+
2. `"poisson"`: Generalized Linear Model using Poisson log-linear regression.
|
|
204
|
+
3. `"negative-binomial"`: Generalized Linear Model with negative binomial dispersion parameter `method_alpha`.
|
|
205
|
+
|
|
206
|
+
When event data exhibits overdispersion (variance significantly exceeds the mean), Poisson estimates may underestimate variance. You can test for overdispersion using Cameron and Trivedi's auxiliary regression test:
|
|
207
|
+
|
|
208
|
+
```python
|
|
209
|
+
import pandas as pd
|
|
210
|
+
from vigipy import convert, bcpnn
|
|
211
|
+
from vigipy.utils import test_dispersion
|
|
212
|
+
|
|
213
|
+
df = pd.read_csv("AE_count_data.csv")
|
|
214
|
+
data = convert(df)
|
|
215
|
+
|
|
216
|
+
# Test for overdispersion
|
|
217
|
+
dispersion_info = test_dispersion(data)
|
|
218
|
+
print(f"Dispersion ratio: {dispersion_info['dispersion']:.2f}")
|
|
219
|
+
|
|
220
|
+
# If overdispersed (> 2), use the estimated alpha in Negative Binomial regression
|
|
221
|
+
alpha = dispersion_info["alpha"] if dispersion_info["dispersion"] > 2 else 1.0
|
|
222
|
+
res = bcpnn(data, expected_method="negative-binomial", method_alpha=alpha, min_events=3)
|
|
223
|
+
```
|
|
224
|
+
|
|
225
|
+
---
|
|
226
|
+
|
|
227
|
+
## Longitudinal Modeling
|
|
228
|
+
|
|
229
|
+
The `LongitudinalModel` class evaluates disproportionality over time to monitor signal emergence, stability, and trajectory.
|
|
230
|
+
|
|
231
|
+
You can run models in two modes:
|
|
232
|
+
* **Cumulative (`run`)**: Progressively incorporates historical data up to each resampled time boundary, tracking accumulating evidence.
|
|
233
|
+
* **Disjoint (`run_disjoint`)**: Evaluates each time window independently without historical accumulation.
|
|
234
|
+
|
|
235
|
+
```python
|
|
236
|
+
import pandas as pd
|
|
237
|
+
from vigipy import LongitudinalModel, gps
|
|
238
|
+
|
|
239
|
+
df = pd.read_csv("AE_time_series.csv")
|
|
240
|
+
# Must contain: 'date', 'name', 'AE', and 'count' (or custom count_col)
|
|
241
|
+
|
|
242
|
+
# Initialize grouped by calendar year ('YE', 'QE', 'ME')
|
|
243
|
+
lm = LongitudinalModel(df, time_unit="YE", count_col="count")
|
|
244
|
+
|
|
245
|
+
# Run GPS cumulatively over time
|
|
246
|
+
lm.run(gps, include_gaps=False, decision_metric="rank", ranking_statistic="log2")
|
|
247
|
+
|
|
248
|
+
# Regroup to quarterly slices and evaluate
|
|
249
|
+
lm.regroup_dates("QE")
|
|
250
|
+
lm.run_disjoint(gps, include_gaps=False, decision_metric="rank", ranking_statistic="log2")
|
|
251
|
+
|
|
252
|
+
# Access results chronologically: (timestamp, AnalysisResult)
|
|
253
|
+
for timestamp, result in lm.results:
|
|
254
|
+
if result is not None:
|
|
255
|
+
print(f"Slice ending {timestamp.date()}: {result.num_signals} signals")
|
|
256
|
+
print(result.signals.head(2))
|
|
257
|
+
```
|
|
258
|
+
|
|
259
|
+
---
|
|
260
|
+
|
|
261
|
+
## LASSO Signal Detection
|
|
262
|
+
|
|
263
|
+
LASSO regression models multiple products simultaneously, adjusting for co-prescriptions, confounding by indication, and polypharmacy.
|
|
264
|
+
|
|
265
|
+
### 1. Pure Binary Matrix
|
|
266
|
+
Each row represents an individual report or subject:
|
|
267
|
+
|
|
268
|
+
```python
|
|
269
|
+
import pandas as pd
|
|
270
|
+
from vigipy import convert_binary, lasso
|
|
271
|
+
|
|
272
|
+
df = pd.read_csv("patient_reports.csv")
|
|
273
|
+
bin_data = convert_binary(df, product_label="name", ae_label="AE", use_counts=False)
|
|
274
|
+
|
|
275
|
+
# Linear LASSO with Information Criterion model selection
|
|
276
|
+
result = lasso(bin_data, use_IC=True, IC_criterion="bic", min_events=3)
|
|
277
|
+
print(result.signals[["Product", "Adverse Event", "LASSO Coefficient", "CI Lower", "CI Upper"]])
|
|
278
|
+
result.export("lasso_signals.xlsx")
|
|
279
|
+
```
|
|
280
|
+
|
|
281
|
+
### 2. Count Outcomes & GLM LASSO
|
|
282
|
+
When aggregate event counts are used:
|
|
283
|
+
|
|
284
|
+
```python
|
|
285
|
+
bin_data = convert_binary(df, product_label="name", ae_label="AE", use_counts=True)
|
|
286
|
+
|
|
287
|
+
# Fit Negative Binomial GLM with L1 regularization
|
|
288
|
+
result = lasso(bin_data, use_glm=True, lasso_thresh=0.2, nb_alpha=1.0)
|
|
289
|
+
result.export("lasso_glm_signals.csv")
|
|
290
|
+
```
|
|
291
|
+
|
|
292
|
+
---
|
|
293
|
+
|
|
294
|
+
## Multi-Item Interaction Conversion
|
|
295
|
+
|
|
296
|
+
To analyze interactions between co-occurring drugs or devices:
|
|
297
|
+
|
|
298
|
+
```python
|
|
299
|
+
from vigipy.utils.data_prep import convert_multi_item
|
|
300
|
+
|
|
301
|
+
# Aggregate co-administered products
|
|
302
|
+
multi_data = convert_multi_item(
|
|
303
|
+
df,
|
|
304
|
+
product_label=["suspect_drug_1", "suspect_drug_2"],
|
|
305
|
+
ae_label="AE",
|
|
306
|
+
count_label="count",
|
|
307
|
+
min_threshold=3,
|
|
308
|
+
)
|
|
309
|
+
```
|
|
310
|
+
|
|
311
|
+
The returned `DataContainer` is fully compatible with `prr`, `ror`, `rfet`, `bcpnn`, and `gps`.
|
|
312
|
+
|
|
313
|
+
---
|
|
314
|
+
|
|
315
|
+
## Result Inspection & Export
|
|
316
|
+
|
|
317
|
+
All analysis methods return an `AnalysisResult` object:
|
|
318
|
+
|
|
319
|
+
```python
|
|
320
|
+
result = analyze(data, PRRConfig(min_events=3))
|
|
321
|
+
|
|
322
|
+
# Filtered signals meeting decision criteria
|
|
323
|
+
signals_df = result.signals
|
|
324
|
+
|
|
325
|
+
# All evaluated candidate pairs with computed statistics
|
|
326
|
+
all_df = result.all_signals
|
|
327
|
+
|
|
328
|
+
# Number of identified signals
|
|
329
|
+
count = result.num_signals
|
|
330
|
+
|
|
331
|
+
# Input parameters and model metadata
|
|
332
|
+
params_dict = result.params
|
|
333
|
+
|
|
334
|
+
# Export to Excel (.xlsx) or CSV (.csv)
|
|
335
|
+
result.export("output.xlsx") # Writes 'Signals' and 'all_data' sheets
|
|
336
|
+
result.export("output.csv") # Writes signals DataFrame
|
|
337
|
+
```
|
|
338
|
+
|
|
339
|
+
---
|
|
340
|
+
|
|
341
|
+
## Authors
|
|
342
|
+
|
|
343
|
+
* **David Beery** ([@Shakesbeery](https://github.com/Shakesbeery))
|
|
344
|
+
|
|
345
|
+
## License
|
|
346
|
+
|
|
347
|
+
This project is licensed under the [MIT License](LICENSE).
|
|
348
|
+
|
|
349
|
+
## Acknowledgements
|
|
350
|
+
|
|
351
|
+
* **Ismail Ahmed and Antoine Poncet** for the foundational design of the [PhViD](https://cran.r-project.org/web/packages/PhViD/index.html) package in R.
|
|
352
|
+
* **Ross Ihaka** and **Catherine Loader** for early mathematical formulations of deviance and log-gamma approximations.
|
vigipy-3.0.0/README.md
ADDED
|
@@ -0,0 +1,328 @@
|
|
|
1
|
+
# vigipy
|
|
2
|
+
|
|
3
|
+
> [!IMPORTANT]
|
|
4
|
+
> **Major Release — `vigipy` v3.0 is live!**
|
|
5
|
+
> This major release introduces an extensive modernization of the library:
|
|
6
|
+
> - **Unified Interface**: Typed execution via `analyze()`, `analyze_all()`, and configuration dataclasses (`PRRConfig`, `RORConfig`, `RFETConfig`, `BCPNNConfig`, `GPSConfig`, `LASSOConfig`).
|
|
7
|
+
> - **Statistical & Mathematical Rigor**: Corrected FDR step-up monotonicity, decision-theoretic Bayesian metrics (FDR, FNR, FOR, Se, Sp), RFET mid-$p$ parameter order, and Haldane-Anscombe (+0.5) zero-cell continuity corrections.
|
|
8
|
+
> - **Vectorized Performance**: Fully vectorized core operations, replacing SymPy with native SciPy C-routines for >50% speedups.
|
|
9
|
+
> - **Permissive MIT License**: Formally relicensed under the MIT License.
|
|
10
|
+
>
|
|
11
|
+
> *See the updated API documentation and examples below.*
|
|
12
|
+
|
|
13
|
+
`vigipy` is a Python library bringing modern disproportionality analyses and pharmacovigilance techniques into the Python ecosystem with a clean, intuitive, and type-safe interface. Core disproportionality methods are adapted and extended from Ismail Ahmed and Antoine Poncet's [PhViD](https://cran.r-project.org/web/packages/PhViD/index.html) package, fully vectorized with native NumPy and SciPy routines.
|
|
14
|
+
|
|
15
|
+
### Top-level Functions & Classes:
|
|
16
|
+
|
|
17
|
+
* **Unified Interface**:
|
|
18
|
+
* `analyze()` - Execute any disproportionality analysis via typed configuration dataclasses
|
|
19
|
+
* `analyze_all()` - Run multiple analysis methods in a single call with shared or distinct parameters
|
|
20
|
+
* `PRRConfig`, `RORConfig`, `RFETConfig`, `BCPNNConfig`, `GPSConfig`, `LASSOConfig` - Type-safe configuration dataclasses
|
|
21
|
+
* **Disproportionality Methods**:
|
|
22
|
+
* `prr()` - Proportional Reporting Ratio (frequentist log-normal approximation)
|
|
23
|
+
* `ror()` - Reporting Odds Ratio (Woolf log-odds approximation)
|
|
24
|
+
* `rfet()` - Reporting Fisher's Exact Test (exact hypergeometric p-values with optional mid-p correction)
|
|
25
|
+
* `bcpnn()` - Bayesian Confidence Propagation Neural Network (analytical or Dirichlet Monte Carlo Information Component)
|
|
26
|
+
* `gps()` - Multi-item Gamma Poisson Shrinker (Empirical Bayes bivariate mixture model)
|
|
27
|
+
* `lasso()` - LASSO regression for multivariate signal detection and confounding adjustment
|
|
28
|
+
* **Longitudinal Modeling**:
|
|
29
|
+
* `LongitudinalModel()` - Apply any analysis method over time to evaluate cumulative or disjoint signal evolution
|
|
30
|
+
* **Data Preparation**:
|
|
31
|
+
* `convert()` - Convert adverse event and product count tables into a structured `DataContainer`
|
|
32
|
+
* `convert_binary()` - Generate binary product feature matrices and event outcome matrices for LASSO
|
|
33
|
+
* `convert_multi_item()` - Aggregate co-occurring product columns into multi-item interaction tables
|
|
34
|
+
* **Result & Data Containers**:
|
|
35
|
+
* `AnalysisResult` - Structured container for `signals`, `all_signals`, `num_signals`, and model `params`, with `.export()` to Excel or CSV
|
|
36
|
+
* `DataContainer` - Typed container holding contingency, event, and product matrices
|
|
37
|
+
|
|
38
|
+
---
|
|
39
|
+
|
|
40
|
+
## Getting Started
|
|
41
|
+
|
|
42
|
+
### Dependencies
|
|
43
|
+
|
|
44
|
+
`vigipy` requires Python 3.9+ and modern scientific computing libraries:
|
|
45
|
+
|
|
46
|
+
* `pandas>=2.0`
|
|
47
|
+
* `numpy>=1.24,<3`
|
|
48
|
+
* `scipy>=1.10`
|
|
49
|
+
* `scikit-learn>=1.3`
|
|
50
|
+
* `statsmodels>=0.14`
|
|
51
|
+
|
|
52
|
+
Optional dependencies:
|
|
53
|
+
* `openpyxl>=3.0.0` (required for exporting results directly to Excel `.xlsx` spreadsheets)
|
|
54
|
+
|
|
55
|
+
### Installation
|
|
56
|
+
|
|
57
|
+
Install `vigipy` from source or local checkout:
|
|
58
|
+
|
|
59
|
+
```bash
|
|
60
|
+
pip install .
|
|
61
|
+
```
|
|
62
|
+
|
|
63
|
+
To include Excel export capabilities:
|
|
64
|
+
|
|
65
|
+
```bash
|
|
66
|
+
pip install ".[excel]"
|
|
67
|
+
```
|
|
68
|
+
|
|
69
|
+
For development (includes test and lint tools):
|
|
70
|
+
|
|
71
|
+
```bash
|
|
72
|
+
pip install -e ".[dev]"
|
|
73
|
+
```
|
|
74
|
+
|
|
75
|
+
### Running Tests
|
|
76
|
+
|
|
77
|
+
Run the full test suite using `pytest`:
|
|
78
|
+
|
|
79
|
+
```bash
|
|
80
|
+
pytest test/ -v
|
|
81
|
+
```
|
|
82
|
+
|
|
83
|
+
---
|
|
84
|
+
|
|
85
|
+
## Usage
|
|
86
|
+
|
|
87
|
+
### Unified API (Recommended)
|
|
88
|
+
|
|
89
|
+
The unified interface provides type safety, autocompletion, and consistent result structures across all methods:
|
|
90
|
+
|
|
91
|
+
```python
|
|
92
|
+
import pandas as pd
|
|
93
|
+
from vigipy import convert, analyze, analyze_all, PRRConfig, BCPNNConfig, GPSConfig
|
|
94
|
+
|
|
95
|
+
# 1. Load data and convert to a DataContainer
|
|
96
|
+
df = pd.read_csv("AE_count_data.csv")
|
|
97
|
+
data = convert(df, product_label="name", ae_label="AE", count_label="count")
|
|
98
|
+
|
|
99
|
+
# 2. Run a single method with typed configuration
|
|
100
|
+
result = analyze(data, PRRConfig(min_events=3, decision_metric="fdr", fdr_threshold=0.05))
|
|
101
|
+
|
|
102
|
+
print(f"Detected {result.num_signals} signals")
|
|
103
|
+
print(result.signals.head())
|
|
104
|
+
|
|
105
|
+
# Export both significant signals and full dataset to Excel or CSV
|
|
106
|
+
result.export("prr_signals.xlsx") # Creates 'Signals' and 'all_data' sheets
|
|
107
|
+
result.export("prr_signals.csv") # Exports detected signals to CSV
|
|
108
|
+
|
|
109
|
+
# 3. Batch comparison across all methods in one call
|
|
110
|
+
batch_results = analyze_all(data, min_events=3, decision_metric="rank")
|
|
111
|
+
for method_name, res in batch_results.items():
|
|
112
|
+
print(f"{method_name.upper()}: {res.num_signals} signals detected")
|
|
113
|
+
|
|
114
|
+
# 4. Iterate over custom configurations
|
|
115
|
+
configs = [
|
|
116
|
+
PRRConfig(min_events=5, ranking_statistic="CI"),
|
|
117
|
+
BCPNNConfig(min_events=5, ranking_statistic="quantile"),
|
|
118
|
+
GPSConfig(min_events=5, ranking_statistic="log2"),
|
|
119
|
+
]
|
|
120
|
+
for cfg in configs:
|
|
121
|
+
res = analyze(data, cfg)
|
|
122
|
+
print(f"{cfg.method}: {res.num_signals} signals")
|
|
123
|
+
```
|
|
124
|
+
|
|
125
|
+
### Classic Function API
|
|
126
|
+
|
|
127
|
+
Direct function calls are fully supported with identical return structures:
|
|
128
|
+
|
|
129
|
+
```python
|
|
130
|
+
import pandas as pd
|
|
131
|
+
from vigipy import convert, prr, ror, rfet, bcpnn, gps
|
|
132
|
+
|
|
133
|
+
df = pd.read_csv("AE_count_data.csv")
|
|
134
|
+
data = convert(df)
|
|
135
|
+
|
|
136
|
+
# Frequentist: Reporting Odds Ratio with Haldane-Anscombe continuity correction
|
|
137
|
+
ror_res = ror(data, min_events=3, decision_metric="fdr", fdr_threshold=0.05)
|
|
138
|
+
|
|
139
|
+
# Fisher's Exact Test with Lancaster mid-p adjustment
|
|
140
|
+
rfet_res = rfet(data, min_events=3, mid_pval=True)
|
|
141
|
+
|
|
142
|
+
# Bayesian Confidence Propagation Neural Network
|
|
143
|
+
bcpnn_res = bcpnn(data, min_events=3, ranking_statistic="quantile")
|
|
144
|
+
|
|
145
|
+
# Empirical Bayes: Gamma Poisson Shrinker
|
|
146
|
+
gps_res = gps(data, min_events=5, decision_metric="rank", ranking_statistic="log2")
|
|
147
|
+
|
|
148
|
+
# Access results
|
|
149
|
+
print(gps_res.signals[["Product", "Adverse Event", "Count", "quantile", "fdr"]].head())
|
|
150
|
+
gps_res.export("gps_signals.xlsx")
|
|
151
|
+
```
|
|
152
|
+
|
|
153
|
+
---
|
|
154
|
+
|
|
155
|
+
## Decision Rules & Ranking Statistics
|
|
156
|
+
|
|
157
|
+
`vigipy` standardizes signal identification across frequentist and Bayesian methods:
|
|
158
|
+
|
|
159
|
+
### Decision Metrics (`decision_metric`)
|
|
160
|
+
* `"fdr"` - Controls the False Discovery Rate at `decision_thres` (default: `0.05`) using Local Bayes Estimation (LBE) or cumulative posterior null probabilities.
|
|
161
|
+
* `"rank"` - Retains signals where the ranking statistic meets `decision_thres`. For p-values, selects values $\le \text{threshold}$; for confidence/credible bounds, selects values $\ge \text{threshold}$.
|
|
162
|
+
* `"signals"` - Selects the top $N$ ranked candidates up to `decision_thres`.
|
|
163
|
+
|
|
164
|
+
### Ranking Statistics (`ranking_statistic`)
|
|
165
|
+
| Method | Supported Statistics | Notes |
|
|
166
|
+
| :--- | :--- | :--- |
|
|
167
|
+
| **PRR / ROR** | `"p_value"`, `"CI"` | `"CI"` ranks by the lower bound of the 95% confidence interval. |
|
|
168
|
+
| **RFET** | `"p_value"` | Exact hypergeometric p-value (supports `mid_pval=True`). |
|
|
169
|
+
| **BCPNN** | `"quantile"`, `"p_value"` | `"quantile"` ranks by $IC_{025}$ (lower 95% credible bound). |
|
|
170
|
+
| **GPS** | `"log2"`, `"quantile"`, `"p_value"` | `"log2"` ranks by $EB_{05}$ of $\log_2(\lambda)$, shrinked towards expected counts. |
|
|
171
|
+
|
|
172
|
+
---
|
|
173
|
+
|
|
174
|
+
## Expected Count Calculations & Dispersion Testing
|
|
175
|
+
|
|
176
|
+
Expected counts ($E$) model the baseline event frequency under the null hypothesis of no association. `vigipy` supports three expectation models:
|
|
177
|
+
|
|
178
|
+
1. `"mantel-haentzel"` (default): Standard independence assumption, $E_{ij} = \frac{n_{i\cdot} n_{\cdot j}}{N}$.
|
|
179
|
+
2. `"poisson"`: Generalized Linear Model using Poisson log-linear regression.
|
|
180
|
+
3. `"negative-binomial"`: Generalized Linear Model with negative binomial dispersion parameter `method_alpha`.
|
|
181
|
+
|
|
182
|
+
When event data exhibits overdispersion (variance significantly exceeds the mean), Poisson estimates may underestimate variance. You can test for overdispersion using Cameron and Trivedi's auxiliary regression test:
|
|
183
|
+
|
|
184
|
+
```python
|
|
185
|
+
import pandas as pd
|
|
186
|
+
from vigipy import convert, bcpnn
|
|
187
|
+
from vigipy.utils import test_dispersion
|
|
188
|
+
|
|
189
|
+
df = pd.read_csv("AE_count_data.csv")
|
|
190
|
+
data = convert(df)
|
|
191
|
+
|
|
192
|
+
# Test for overdispersion
|
|
193
|
+
dispersion_info = test_dispersion(data)
|
|
194
|
+
print(f"Dispersion ratio: {dispersion_info['dispersion']:.2f}")
|
|
195
|
+
|
|
196
|
+
# If overdispersed (> 2), use the estimated alpha in Negative Binomial regression
|
|
197
|
+
alpha = dispersion_info["alpha"] if dispersion_info["dispersion"] > 2 else 1.0
|
|
198
|
+
res = bcpnn(data, expected_method="negative-binomial", method_alpha=alpha, min_events=3)
|
|
199
|
+
```
|
|
200
|
+
|
|
201
|
+
---
|
|
202
|
+
|
|
203
|
+
## Longitudinal Modeling
|
|
204
|
+
|
|
205
|
+
The `LongitudinalModel` class evaluates disproportionality over time to monitor signal emergence, stability, and trajectory.
|
|
206
|
+
|
|
207
|
+
You can run models in two modes:
|
|
208
|
+
* **Cumulative (`run`)**: Progressively incorporates historical data up to each resampled time boundary, tracking accumulating evidence.
|
|
209
|
+
* **Disjoint (`run_disjoint`)**: Evaluates each time window independently without historical accumulation.
|
|
210
|
+
|
|
211
|
+
```python
|
|
212
|
+
import pandas as pd
|
|
213
|
+
from vigipy import LongitudinalModel, gps
|
|
214
|
+
|
|
215
|
+
df = pd.read_csv("AE_time_series.csv")
|
|
216
|
+
# Must contain: 'date', 'name', 'AE', and 'count' (or custom count_col)
|
|
217
|
+
|
|
218
|
+
# Initialize grouped by calendar year ('YE', 'QE', 'ME')
|
|
219
|
+
lm = LongitudinalModel(df, time_unit="YE", count_col="count")
|
|
220
|
+
|
|
221
|
+
# Run GPS cumulatively over time
|
|
222
|
+
lm.run(gps, include_gaps=False, decision_metric="rank", ranking_statistic="log2")
|
|
223
|
+
|
|
224
|
+
# Regroup to quarterly slices and evaluate
|
|
225
|
+
lm.regroup_dates("QE")
|
|
226
|
+
lm.run_disjoint(gps, include_gaps=False, decision_metric="rank", ranking_statistic="log2")
|
|
227
|
+
|
|
228
|
+
# Access results chronologically: (timestamp, AnalysisResult)
|
|
229
|
+
for timestamp, result in lm.results:
|
|
230
|
+
if result is not None:
|
|
231
|
+
print(f"Slice ending {timestamp.date()}: {result.num_signals} signals")
|
|
232
|
+
print(result.signals.head(2))
|
|
233
|
+
```
|
|
234
|
+
|
|
235
|
+
---
|
|
236
|
+
|
|
237
|
+
## LASSO Signal Detection
|
|
238
|
+
|
|
239
|
+
LASSO regression models multiple products simultaneously, adjusting for co-prescriptions, confounding by indication, and polypharmacy.
|
|
240
|
+
|
|
241
|
+
### 1. Pure Binary Matrix
|
|
242
|
+
Each row represents an individual report or subject:
|
|
243
|
+
|
|
244
|
+
```python
|
|
245
|
+
import pandas as pd
|
|
246
|
+
from vigipy import convert_binary, lasso
|
|
247
|
+
|
|
248
|
+
df = pd.read_csv("patient_reports.csv")
|
|
249
|
+
bin_data = convert_binary(df, product_label="name", ae_label="AE", use_counts=False)
|
|
250
|
+
|
|
251
|
+
# Linear LASSO with Information Criterion model selection
|
|
252
|
+
result = lasso(bin_data, use_IC=True, IC_criterion="bic", min_events=3)
|
|
253
|
+
print(result.signals[["Product", "Adverse Event", "LASSO Coefficient", "CI Lower", "CI Upper"]])
|
|
254
|
+
result.export("lasso_signals.xlsx")
|
|
255
|
+
```
|
|
256
|
+
|
|
257
|
+
### 2. Count Outcomes & GLM LASSO
|
|
258
|
+
When aggregate event counts are used:
|
|
259
|
+
|
|
260
|
+
```python
|
|
261
|
+
bin_data = convert_binary(df, product_label="name", ae_label="AE", use_counts=True)
|
|
262
|
+
|
|
263
|
+
# Fit Negative Binomial GLM with L1 regularization
|
|
264
|
+
result = lasso(bin_data, use_glm=True, lasso_thresh=0.2, nb_alpha=1.0)
|
|
265
|
+
result.export("lasso_glm_signals.csv")
|
|
266
|
+
```
|
|
267
|
+
|
|
268
|
+
---
|
|
269
|
+
|
|
270
|
+
## Multi-Item Interaction Conversion
|
|
271
|
+
|
|
272
|
+
To analyze interactions between co-occurring drugs or devices:
|
|
273
|
+
|
|
274
|
+
```python
|
|
275
|
+
from vigipy.utils.data_prep import convert_multi_item
|
|
276
|
+
|
|
277
|
+
# Aggregate co-administered products
|
|
278
|
+
multi_data = convert_multi_item(
|
|
279
|
+
df,
|
|
280
|
+
product_label=["suspect_drug_1", "suspect_drug_2"],
|
|
281
|
+
ae_label="AE",
|
|
282
|
+
count_label="count",
|
|
283
|
+
min_threshold=3,
|
|
284
|
+
)
|
|
285
|
+
```
|
|
286
|
+
|
|
287
|
+
The returned `DataContainer` is fully compatible with `prr`, `ror`, `rfet`, `bcpnn`, and `gps`.
|
|
288
|
+
|
|
289
|
+
---
|
|
290
|
+
|
|
291
|
+
## Result Inspection & Export
|
|
292
|
+
|
|
293
|
+
All analysis methods return an `AnalysisResult` object:
|
|
294
|
+
|
|
295
|
+
```python
|
|
296
|
+
result = analyze(data, PRRConfig(min_events=3))
|
|
297
|
+
|
|
298
|
+
# Filtered signals meeting decision criteria
|
|
299
|
+
signals_df = result.signals
|
|
300
|
+
|
|
301
|
+
# All evaluated candidate pairs with computed statistics
|
|
302
|
+
all_df = result.all_signals
|
|
303
|
+
|
|
304
|
+
# Number of identified signals
|
|
305
|
+
count = result.num_signals
|
|
306
|
+
|
|
307
|
+
# Input parameters and model metadata
|
|
308
|
+
params_dict = result.params
|
|
309
|
+
|
|
310
|
+
# Export to Excel (.xlsx) or CSV (.csv)
|
|
311
|
+
result.export("output.xlsx") # Writes 'Signals' and 'all_data' sheets
|
|
312
|
+
result.export("output.csv") # Writes signals DataFrame
|
|
313
|
+
```
|
|
314
|
+
|
|
315
|
+
---
|
|
316
|
+
|
|
317
|
+
## Authors
|
|
318
|
+
|
|
319
|
+
* **David Beery** ([@Shakesbeery](https://github.com/Shakesbeery))
|
|
320
|
+
|
|
321
|
+
## License
|
|
322
|
+
|
|
323
|
+
This project is licensed under the [MIT License](LICENSE).
|
|
324
|
+
|
|
325
|
+
## Acknowledgements
|
|
326
|
+
|
|
327
|
+
* **Ismail Ahmed and Antoine Poncet** for the foundational design of the [PhViD](https://cran.r-project.org/web/packages/PhViD/index.html) package in R.
|
|
328
|
+
* **Ross Ihaka** and **Catherine Loader** for early mathematical formulations of deviance and log-gamma approximations.
|