pymaude 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- pymaude/__init__.py +14 -0
- pymaude/database.py +1100 -0
- pymaude/metadata.py +54 -0
- pymaude-0.2.0.dist-info/METADATA +21 -0
- pymaude-0.2.0.dist-info/RECORD +8 -0
- pymaude-0.2.0.dist-info/WHEEL +5 -0
- pymaude-0.2.0.dist-info/licenses/LICENSE +21 -0
- pymaude-0.2.0.dist-info/top_level.txt +1 -0
pymaude/__init__.py
ADDED
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
# PyMAUDE - FDA MAUDE Database Interface (DuckDB backend)
|
|
2
|
+
# Copyright (C) 2026 Jacob Schwartz <jaschwa@umich.edu>
|
|
3
|
+
# MIT License
|
|
4
|
+
|
|
5
|
+
from .database import MaudeDatabase
|
|
6
|
+
from .metadata import TABLE_METADATA, FDA_BASE_URL
|
|
7
|
+
|
|
8
|
+
__version__ = '0.2.0'
|
|
9
|
+
__author__ = 'Jacob Schwartz <jaschwa@umich.edu>'
|
|
10
|
+
__all__ = [
|
|
11
|
+
'MaudeDatabase',
|
|
12
|
+
'TABLE_METADATA',
|
|
13
|
+
'FDA_BASE_URL',
|
|
14
|
+
]
|
pymaude/database.py
ADDED
|
@@ -0,0 +1,1100 @@
|
|
|
1
|
+
# database.py - FDA MAUDE Database Interface (DuckDB backend)
|
|
2
|
+
# Copyright (C) 2026 Jacob Schwartz <jaschwa@umich.edu>
|
|
3
|
+
# MIT License
|
|
4
|
+
|
|
5
|
+
"""
|
|
6
|
+
MaudeDatabase: download, load, and query FDA MAUDE adverse event data.
|
|
7
|
+
|
|
8
|
+
Usage:
|
|
9
|
+
db = MaudeDatabase('maude.duckdb', data_dir='./maude_data')
|
|
10
|
+
db.add_years('2019-2024', tables=['master', 'device', 'text'], download=True)
|
|
11
|
+
results = db.search_by_device_names([['argon', 'cleaner'], 'angiojet'])
|
|
12
|
+
narratives = db.get_narratives(results['MDR_REPORT_KEY'])
|
|
13
|
+
db.close()
|
|
14
|
+
"""
|
|
15
|
+
|
|
16
|
+
import os
|
|
17
|
+
import json
|
|
18
|
+
import shutil
|
|
19
|
+
import zipfile
|
|
20
|
+
import hashlib
|
|
21
|
+
from collections import defaultdict
|
|
22
|
+
from datetime import datetime
|
|
23
|
+
|
|
24
|
+
import duckdb
|
|
25
|
+
import pandas as pd
|
|
26
|
+
import requests
|
|
27
|
+
|
|
28
|
+
from .metadata import TABLE_METADATA, FDA_BASE_URL
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
class MaudeDatabase:
|
|
32
|
+
"""
|
|
33
|
+
Interface to FDA MAUDE database via DuckDB.
|
|
34
|
+
|
|
35
|
+
Data is stored in a persistent DuckDB file. Raw downloaded CSVs are kept
|
|
36
|
+
in data_dir and can be deleted after loading if disk space is a concern.
|
|
37
|
+
|
|
38
|
+
Args:
|
|
39
|
+
db_path: Path to DuckDB database file (created if it doesn't exist).
|
|
40
|
+
data_dir: Directory for downloaded/extracted MAUDE files.
|
|
41
|
+
verbose: Print progress messages (default True).
|
|
42
|
+
memory_limit: DuckDB memory cap (e.g. '4GB', '512MB'). When set,
|
|
43
|
+
DuckDB spills intermediate data to disk instead of OOMing during
|
|
44
|
+
large CSV loads. Defaults to None (DuckDB manages its own limit).
|
|
45
|
+
"""
|
|
46
|
+
|
|
47
|
+
def __init__(self, db_path, data_dir='./maude_data', verbose=True, memory_limit=None):
|
|
48
|
+
self.db_path = db_path
|
|
49
|
+
self.data_dir = data_dir
|
|
50
|
+
self.verbose = verbose
|
|
51
|
+
self._download_cache = set()
|
|
52
|
+
|
|
53
|
+
os.makedirs(data_dir, exist_ok=True)
|
|
54
|
+
self.conn = duckdb.connect(db_path)
|
|
55
|
+
if memory_limit is not None:
|
|
56
|
+
self.conn.execute(f"SET memory_limit='{memory_limit}'")
|
|
57
|
+
self._init_metadata_table()
|
|
58
|
+
|
|
59
|
+
def __enter__(self):
|
|
60
|
+
return self
|
|
61
|
+
|
|
62
|
+
def __exit__(self, *args):
|
|
63
|
+
self.close()
|
|
64
|
+
|
|
65
|
+
# ── Public: data management ───────────────────────────────────────────────
|
|
66
|
+
|
|
67
|
+
def add_years(self, years, tables=None, download=False,
|
|
68
|
+
force_download=False, force_reload=False, force_partial=False):
|
|
69
|
+
"""
|
|
70
|
+
Load MAUDE data for the specified years into the database.
|
|
71
|
+
|
|
72
|
+
Uses checksum tracking to skip files that haven't changed since last load.
|
|
73
|
+
|
|
74
|
+
Args:
|
|
75
|
+
years: Years to load. One of:
|
|
76
|
+
- int: 2024
|
|
77
|
+
- list: [2020, 2021, 2022]
|
|
78
|
+
- str range: '2019-2024'
|
|
79
|
+
- 'all', 'latest', 'current'
|
|
80
|
+
tables: List of tables to load (default: ['master', 'device', 'text', 'patient']).
|
|
81
|
+
download: If True, download files from FDA before loading.
|
|
82
|
+
force_download: If True, re-download even if zip exists locally.
|
|
83
|
+
force_reload: If True, reload into DB even if checksum is unchanged.
|
|
84
|
+
force_partial: Cumulative tables (master, patient, problem) are fully
|
|
85
|
+
replaced on every reload, so a request that doesn't cover years
|
|
86
|
+
already loaded for one of them would silently drop those years.
|
|
87
|
+
That's rejected by default — pass True to proceed anyway. Doesn't
|
|
88
|
+
apply to yearly tables (device, text), which can't lose data this
|
|
89
|
+
way since each year is an independent file.
|
|
90
|
+
"""
|
|
91
|
+
years_list = self._parse_year_range(years)
|
|
92
|
+
if tables is None:
|
|
93
|
+
tables = ['master', 'device', 'text', 'patient']
|
|
94
|
+
|
|
95
|
+
valid = self._validate(years_list, tables)
|
|
96
|
+
if not valid:
|
|
97
|
+
return
|
|
98
|
+
|
|
99
|
+
years_by_table = defaultdict(list)
|
|
100
|
+
for year, table in valid:
|
|
101
|
+
years_by_table[table].append(year)
|
|
102
|
+
|
|
103
|
+
if not force_partial:
|
|
104
|
+
for table in sorted(years_by_table):
|
|
105
|
+
meta = TABLE_METADATA[table]
|
|
106
|
+
if meta['pattern_type'] != 'cumulative':
|
|
107
|
+
continue
|
|
108
|
+
years_for_table = sorted(years_by_table[table])
|
|
109
|
+
implied = self._implied_years(table, years_for_table, meta)
|
|
110
|
+
missing = self._covered_years(table) - implied
|
|
111
|
+
if missing:
|
|
112
|
+
raise ValueError(
|
|
113
|
+
f"add_years({years_for_table}, tables=['{table}']) would not "
|
|
114
|
+
f"cover years {sorted(missing)} that '{table}' already has "
|
|
115
|
+
f"loaded. '{table}' is fully replaced on every reload, so this "
|
|
116
|
+
f"would silently drop that data. Call update() to refresh "
|
|
117
|
+
f"everything '{table}' has, or pass force_partial=True to "
|
|
118
|
+
f"proceed and accept the loss."
|
|
119
|
+
)
|
|
120
|
+
|
|
121
|
+
for table in sorted(years_by_table):
|
|
122
|
+
meta = TABLE_METADATA[table]
|
|
123
|
+
years_for_table = sorted(years_by_table[table])
|
|
124
|
+
|
|
125
|
+
if meta['pattern_type'] == 'yearly':
|
|
126
|
+
self._process_yearly_table(
|
|
127
|
+
table, years_for_table, download, force_download, force_reload
|
|
128
|
+
)
|
|
129
|
+
else:
|
|
130
|
+
self._process_cumulative_table(
|
|
131
|
+
table, years_for_table, meta, download, force_download, force_reload
|
|
132
|
+
)
|
|
133
|
+
|
|
134
|
+
self._create_indexes()
|
|
135
|
+
|
|
136
|
+
def update(self, download=True, force_download=True):
|
|
137
|
+
"""
|
|
138
|
+
Refresh all currently-loaded years and add any new years since the last load.
|
|
139
|
+
|
|
140
|
+
Args:
|
|
141
|
+
download: If True, download updated files from FDA.
|
|
142
|
+
force_download: If True, re-download even if zip files exist locally.
|
|
143
|
+
"""
|
|
144
|
+
years = self._get_years_in_db()
|
|
145
|
+
if not years:
|
|
146
|
+
if self.verbose:
|
|
147
|
+
print('Database is empty. Use add_years() to populate.')
|
|
148
|
+
return
|
|
149
|
+
|
|
150
|
+
all_years = list(range(min(years), datetime.now().year + 1))
|
|
151
|
+
loaded_tables = self._get_loaded_tables()
|
|
152
|
+
if self.verbose:
|
|
153
|
+
print(f'Refreshing {len(loaded_tables)} tables, years {min(years)}–{datetime.now().year}')
|
|
154
|
+
self.add_years(all_years, tables=loaded_tables, download=download,
|
|
155
|
+
force_download=force_download)
|
|
156
|
+
|
|
157
|
+
# ── Public: queries ───────────────────────────────────────────────────────
|
|
158
|
+
|
|
159
|
+
def query_device(self, brand_name=None, generic_name=None,
|
|
160
|
+
manufacturer_name=None, product_code=None,
|
|
161
|
+
start_date=None, end_date=None):
|
|
162
|
+
"""
|
|
163
|
+
Query device events by exact field matching (case-insensitive).
|
|
164
|
+
|
|
165
|
+
All provided parameters are combined with AND logic. At least one
|
|
166
|
+
device field (brand_name, generic_name, manufacturer_name, product_code)
|
|
167
|
+
is required.
|
|
168
|
+
|
|
169
|
+
For partial/substring matching or boolean logic, use search_by_device_names().
|
|
170
|
+
|
|
171
|
+
Args:
|
|
172
|
+
brand_name: Exact BRAND_NAME match (e.g., 'Venovo').
|
|
173
|
+
generic_name: Exact GENERIC_NAME match (e.g., 'Venous Stent').
|
|
174
|
+
manufacturer_name: Exact MANUFACTURER_D_NAME match.
|
|
175
|
+
product_code: Exact DEVICE_REPORT_PRODUCT_CODE (e.g., 'NIQ').
|
|
176
|
+
start_date: Earliest DATE_RECEIVED to include (YYYY-MM-DD).
|
|
177
|
+
end_date: Latest DATE_RECEIVED to include (YYYY-MM-DD).
|
|
178
|
+
|
|
179
|
+
Returns:
|
|
180
|
+
DataFrame joining master and device tables.
|
|
181
|
+
"""
|
|
182
|
+
conditions, params = [], []
|
|
183
|
+
|
|
184
|
+
if brand_name is not None:
|
|
185
|
+
conditions.append("d.BRAND_NAME ILIKE ?")
|
|
186
|
+
params.append(brand_name)
|
|
187
|
+
if generic_name is not None:
|
|
188
|
+
conditions.append("d.GENERIC_NAME ILIKE ?")
|
|
189
|
+
params.append(generic_name)
|
|
190
|
+
if manufacturer_name is not None:
|
|
191
|
+
conditions.append("d.MANUFACTURER_D_NAME ILIKE ?")
|
|
192
|
+
params.append(manufacturer_name)
|
|
193
|
+
if product_code is not None:
|
|
194
|
+
conditions.append("d.DEVICE_REPORT_PRODUCT_CODE = ?")
|
|
195
|
+
params.append(product_code)
|
|
196
|
+
|
|
197
|
+
if not conditions:
|
|
198
|
+
raise ValueError(
|
|
199
|
+
"At least one device field required: brand_name, generic_name, "
|
|
200
|
+
"manufacturer_name, or product_code."
|
|
201
|
+
)
|
|
202
|
+
|
|
203
|
+
if start_date:
|
|
204
|
+
conditions.append("m.DATE_RECEIVED >= ?::DATE")
|
|
205
|
+
params.append(start_date)
|
|
206
|
+
if end_date:
|
|
207
|
+
conditions.append("m.DATE_RECEIVED <= ?::DATE")
|
|
208
|
+
params.append(end_date)
|
|
209
|
+
|
|
210
|
+
where = " AND ".join(conditions)
|
|
211
|
+
sql = f"""
|
|
212
|
+
SELECT m.*, d.* EXCLUDE (MDR_REPORT_KEY)
|
|
213
|
+
FROM device d
|
|
214
|
+
JOIN master m USING (MDR_REPORT_KEY)
|
|
215
|
+
WHERE {where}
|
|
216
|
+
"""
|
|
217
|
+
return self.conn.execute(sql, params).df()
|
|
218
|
+
|
|
219
|
+
def search_by_device_names(self, criteria, start_date=None, end_date=None,
|
|
220
|
+
group_column='search_group'):
|
|
221
|
+
"""
|
|
222
|
+
Substring search across DEVICE_NAME_CONCAT (BRAND_NAME|GENERIC_NAME|MANUFACTURER_D_NAME).
|
|
223
|
+
|
|
224
|
+
Criteria formats:
|
|
225
|
+
'term' — any record where any name field contains 'term'
|
|
226
|
+
['a', 'b'] — contains 'a' OR 'b'
|
|
227
|
+
[['a', 'b'], 'c'] — (contains 'a' AND 'b') OR (contains 'c')
|
|
228
|
+
{'group1': [...], ...} — grouped search; result includes search_group column.
|
|
229
|
+
Use None as criteria to skip a group.
|
|
230
|
+
|
|
231
|
+
All matching is case-insensitive substring matching.
|
|
232
|
+
|
|
233
|
+
Args:
|
|
234
|
+
criteria: Search terms in one of the formats above.
|
|
235
|
+
start_date: Earliest DATE_RECEIVED to include (YYYY-MM-DD).
|
|
236
|
+
end_date: Latest DATE_RECEIVED to include (YYYY-MM-DD).
|
|
237
|
+
group_column: Column name for group label when using dict criteria.
|
|
238
|
+
|
|
239
|
+
Returns:
|
|
240
|
+
DataFrame joining master and device tables (plus group_column if dict input).
|
|
241
|
+
"""
|
|
242
|
+
if isinstance(criteria, dict):
|
|
243
|
+
parts = []
|
|
244
|
+
for group_name, group_criteria in criteria.items():
|
|
245
|
+
if group_criteria is None:
|
|
246
|
+
continue
|
|
247
|
+
df = self.search_by_device_names(group_criteria, start_date, end_date)
|
|
248
|
+
df[group_column] = group_name
|
|
249
|
+
parts.append(df)
|
|
250
|
+
if not parts:
|
|
251
|
+
return pd.DataFrame()
|
|
252
|
+
combined = pd.concat(parts, ignore_index=True)
|
|
253
|
+
# Events matching multiple groups: keep first group assignment (dict order).
|
|
254
|
+
return combined.drop_duplicates(subset=['MDR_REPORT_KEY'])
|
|
255
|
+
|
|
256
|
+
normalized = self._normalize_criteria(criteria)
|
|
257
|
+
params = []
|
|
258
|
+
or_groups = []
|
|
259
|
+
|
|
260
|
+
for and_group in normalized:
|
|
261
|
+
and_parts = []
|
|
262
|
+
for term in and_group:
|
|
263
|
+
and_parts.append("d.DEVICE_NAME_CONCAT ILIKE ?")
|
|
264
|
+
params.append(f'%{term}%')
|
|
265
|
+
or_groups.append("(" + " AND ".join(and_parts) + ")")
|
|
266
|
+
|
|
267
|
+
where = "(" + " OR ".join(or_groups) + ")"
|
|
268
|
+
|
|
269
|
+
if start_date:
|
|
270
|
+
where += " AND m.DATE_RECEIVED >= ?::DATE"
|
|
271
|
+
params.append(start_date)
|
|
272
|
+
if end_date:
|
|
273
|
+
where += " AND m.DATE_RECEIVED <= ?::DATE"
|
|
274
|
+
params.append(end_date)
|
|
275
|
+
|
|
276
|
+
sql = f"""
|
|
277
|
+
SELECT m.*, d.* EXCLUDE (MDR_REPORT_KEY)
|
|
278
|
+
FROM device d
|
|
279
|
+
JOIN master m USING (MDR_REPORT_KEY)
|
|
280
|
+
WHERE {where}
|
|
281
|
+
"""
|
|
282
|
+
return self.conn.execute(sql, params).df()
|
|
283
|
+
|
|
284
|
+
def get_narratives(self, mdr_report_keys):
|
|
285
|
+
"""
|
|
286
|
+
Fetch FOI_TEXT narratives for the given MDR report keys.
|
|
287
|
+
|
|
288
|
+
Args:
|
|
289
|
+
mdr_report_keys: List or Series of MDR_REPORT_KEY values.
|
|
290
|
+
|
|
291
|
+
Returns:
|
|
292
|
+
DataFrame with MDR_REPORT_KEY and FOI_TEXT columns.
|
|
293
|
+
"""
|
|
294
|
+
keys = list(mdr_report_keys)
|
|
295
|
+
if not keys:
|
|
296
|
+
return pd.DataFrame(columns=['MDR_REPORT_KEY', 'FOI_TEXT'])
|
|
297
|
+
placeholders = ', '.join(['?'] * len(keys))
|
|
298
|
+
sql = f"SELECT MDR_REPORT_KEY, FOI_TEXT FROM text WHERE MDR_REPORT_KEY IN ({placeholders})"
|
|
299
|
+
return self.conn.execute(sql, keys).df()
|
|
300
|
+
|
|
301
|
+
def get_trends_by_year(self, results_df):
|
|
302
|
+
"""
|
|
303
|
+
Count events per year from a results DataFrame.
|
|
304
|
+
|
|
305
|
+
If results_df has a 'search_group' column (from grouped search), counts
|
|
306
|
+
are broken out per group.
|
|
307
|
+
|
|
308
|
+
Args:
|
|
309
|
+
results_df: DataFrame with at least DATE_RECEIVED and MDR_REPORT_KEY columns.
|
|
310
|
+
|
|
311
|
+
Returns:
|
|
312
|
+
DataFrame with year, event_count (and search_group if present).
|
|
313
|
+
"""
|
|
314
|
+
if not isinstance(results_df, pd.DataFrame):
|
|
315
|
+
raise TypeError("results_df must be a pandas DataFrame")
|
|
316
|
+
if len(results_df) == 0:
|
|
317
|
+
cols = ['year', 'event_count']
|
|
318
|
+
if 'search_group' in results_df.columns:
|
|
319
|
+
cols.insert(0, 'search_group')
|
|
320
|
+
return pd.DataFrame(columns=cols)
|
|
321
|
+
if 'DATE_RECEIVED' not in results_df.columns:
|
|
322
|
+
raise ValueError("results_df must contain DATE_RECEIVED column")
|
|
323
|
+
|
|
324
|
+
df = results_df.copy()
|
|
325
|
+
df['year'] = pd.to_datetime(df['DATE_RECEIVED'], errors='coerce').dt.year
|
|
326
|
+
|
|
327
|
+
group_cols = ['year']
|
|
328
|
+
if 'search_group' in df.columns:
|
|
329
|
+
group_cols.insert(0, 'search_group')
|
|
330
|
+
|
|
331
|
+
trends = df.groupby(group_cols, as_index=False).size()
|
|
332
|
+
trends.rename(columns={'size': 'event_count'}, inplace=True)
|
|
333
|
+
return trends.sort_values(group_cols)
|
|
334
|
+
|
|
335
|
+
def enrich_with_patient_data(self, results_df):
|
|
336
|
+
"""
|
|
337
|
+
Left-join patient outcome data onto a results DataFrame.
|
|
338
|
+
|
|
339
|
+
Args:
|
|
340
|
+
results_df: DataFrame with MDR_REPORT_KEY column.
|
|
341
|
+
|
|
342
|
+
Returns:
|
|
343
|
+
results_df with patient columns appended (suffixed '_patient' on collision).
|
|
344
|
+
"""
|
|
345
|
+
keys = results_df['MDR_REPORT_KEY'].tolist()
|
|
346
|
+
if not keys:
|
|
347
|
+
return results_df
|
|
348
|
+
placeholders = ', '.join(['?'] * len(keys))
|
|
349
|
+
patient = self.conn.execute(
|
|
350
|
+
f"SELECT * FROM patient WHERE MDR_REPORT_KEY IN ({placeholders})", keys
|
|
351
|
+
).df()
|
|
352
|
+
return results_df.merge(patient, on='MDR_REPORT_KEY', how='left',
|
|
353
|
+
suffixes=('', '_patient'))
|
|
354
|
+
|
|
355
|
+
def enrich_with_problems(self, results_df):
|
|
356
|
+
"""
|
|
357
|
+
Left-join device problem codes onto a results DataFrame.
|
|
358
|
+
|
|
359
|
+
Args:
|
|
360
|
+
results_df: DataFrame with MDR_REPORT_KEY column.
|
|
361
|
+
|
|
362
|
+
Returns:
|
|
363
|
+
results_df with problem columns appended (suffixed '_problem' on collision).
|
|
364
|
+
"""
|
|
365
|
+
keys = results_df['MDR_REPORT_KEY'].tolist()
|
|
366
|
+
if not keys:
|
|
367
|
+
return results_df
|
|
368
|
+
placeholders = ', '.join(['?'] * len(keys))
|
|
369
|
+
problems = self.conn.execute(
|
|
370
|
+
f"SELECT * FROM problem WHERE MDR_REPORT_KEY IN ({placeholders})", keys
|
|
371
|
+
).df()
|
|
372
|
+
return results_df.merge(problems, on='MDR_REPORT_KEY', how='left',
|
|
373
|
+
suffixes=('', '_problem'))
|
|
374
|
+
|
|
375
|
+
def query(self, sql, params=None):
|
|
376
|
+
"""Execute a raw DuckDB SQL query and return a DataFrame."""
|
|
377
|
+
return self.conn.execute(sql, params or []).df()
|
|
378
|
+
|
|
379
|
+
def info(self):
|
|
380
|
+
"""Print a summary of loaded tables."""
|
|
381
|
+
print(f"Database : {self.db_path}")
|
|
382
|
+
print(f"Data dir : {self.data_dir}")
|
|
383
|
+
for t in ['master', 'device', 'text', 'patient', 'problem']:
|
|
384
|
+
if not self._table_exists(t):
|
|
385
|
+
print(f" {t:10s}: not loaded")
|
|
386
|
+
continue
|
|
387
|
+
count = self.conn.execute(f"SELECT COUNT(*) FROM {t}").fetchone()[0]
|
|
388
|
+
if t in ('master', 'device'):
|
|
389
|
+
try:
|
|
390
|
+
yr = self.conn.execute(
|
|
391
|
+
f"SELECT MIN(year(DATE_RECEIVED)), MAX(year(DATE_RECEIVED)) FROM {t}"
|
|
392
|
+
).fetchone()
|
|
393
|
+
print(f" {t:10s}: {count:>10,} rows ({yr[0]}–{yr[1]})")
|
|
394
|
+
continue
|
|
395
|
+
except Exception:
|
|
396
|
+
pass
|
|
397
|
+
print(f" {t:10s}: {count:>10,} rows")
|
|
398
|
+
|
|
399
|
+
def close(self):
|
|
400
|
+
"""Close the DuckDB connection."""
|
|
401
|
+
self.conn.close()
|
|
402
|
+
|
|
403
|
+
# ── Public: archiving ───────────────────────────────────────────────────────
|
|
404
|
+
|
|
405
|
+
def archive(self, output_dir, include_raw=False):
|
|
406
|
+
"""
|
|
407
|
+
Prepare a citable snapshot of this database (e.g. for Zenodo upload).
|
|
408
|
+
|
|
409
|
+
Checkpoints and copies the DuckDB file, and writes a manifest.json
|
|
410
|
+
recording, per loaded table/year: source file, SHA-256 checksum, row
|
|
411
|
+
count, and load timestamp (from _load_metadata) — plus the DuckDB and
|
|
412
|
+
pymaude versions used to build it, so the snapshot can be reproduced
|
|
413
|
+
or verified later.
|
|
414
|
+
|
|
415
|
+
Args:
|
|
416
|
+
output_dir: Directory to write the archive into (created if missing).
|
|
417
|
+
include_raw: If True, also copy the raw MAUDE source files referenced
|
|
418
|
+
in _load_metadata (from data_dir) into an output_dir/raw/ subfolder.
|
|
419
|
+
|
|
420
|
+
Returns:
|
|
421
|
+
Path to the written manifest.json.
|
|
422
|
+
"""
|
|
423
|
+
import pymaude
|
|
424
|
+
|
|
425
|
+
os.makedirs(output_dir, exist_ok=True)
|
|
426
|
+
self.conn.execute("CHECKPOINT")
|
|
427
|
+
|
|
428
|
+
db_filename = os.path.basename(self.db_path)
|
|
429
|
+
db_dest = os.path.join(output_dir, db_filename)
|
|
430
|
+
shutil.copy2(self.db_path, db_dest)
|
|
431
|
+
|
|
432
|
+
rows = self.conn.execute(
|
|
433
|
+
"SELECT table_name, year, source_file, checksum, row_count, loaded_at "
|
|
434
|
+
"FROM _load_metadata ORDER BY table_name, year"
|
|
435
|
+
).fetchall()
|
|
436
|
+
|
|
437
|
+
tables = [
|
|
438
|
+
{
|
|
439
|
+
'table': r[0],
|
|
440
|
+
'year': r[1],
|
|
441
|
+
'source_file': r[2],
|
|
442
|
+
'sha256': r[3],
|
|
443
|
+
'row_count': r[4],
|
|
444
|
+
'loaded_at': r[5].isoformat(),
|
|
445
|
+
}
|
|
446
|
+
for r in rows
|
|
447
|
+
]
|
|
448
|
+
|
|
449
|
+
manifest = {
|
|
450
|
+
'generated_at': datetime.now().isoformat(),
|
|
451
|
+
'pymaude_version': pymaude.__version__,
|
|
452
|
+
'duckdb_version': duckdb.__version__,
|
|
453
|
+
'checksum_algorithm': 'sha256',
|
|
454
|
+
'database': {
|
|
455
|
+
'filename': db_filename,
|
|
456
|
+
'sha256': self._checksum(db_dest),
|
|
457
|
+
'size_bytes': os.path.getsize(db_dest),
|
|
458
|
+
},
|
|
459
|
+
'tables': tables,
|
|
460
|
+
}
|
|
461
|
+
|
|
462
|
+
if include_raw:
|
|
463
|
+
raw_dir = os.path.join(output_dir, 'raw')
|
|
464
|
+
os.makedirs(raw_dir, exist_ok=True)
|
|
465
|
+
raw_files = []
|
|
466
|
+
for source_file in sorted({r[2] for r in rows if r[2]}):
|
|
467
|
+
src = os.path.join(self.data_dir, source_file)
|
|
468
|
+
if not os.path.exists(src):
|
|
469
|
+
if self.verbose:
|
|
470
|
+
print(f' Skipping raw file (not found): {source_file}')
|
|
471
|
+
continue
|
|
472
|
+
dest = os.path.join(raw_dir, source_file)
|
|
473
|
+
shutil.copy2(src, dest)
|
|
474
|
+
raw_files.append({'filename': source_file, 'sha256': self._checksum(dest)})
|
|
475
|
+
manifest['raw_files'] = raw_files
|
|
476
|
+
|
|
477
|
+
manifest_path = os.path.join(output_dir, 'manifest.json')
|
|
478
|
+
with open(manifest_path, 'w') as f:
|
|
479
|
+
json.dump(manifest, f, indent=2)
|
|
480
|
+
|
|
481
|
+
if self.verbose:
|
|
482
|
+
print(f'Archive written to {output_dir}')
|
|
483
|
+
print(f' Database : {db_filename} ({manifest["database"]["size_bytes"]:,} bytes)')
|
|
484
|
+
print(f' Tables : {len(tables)} entries')
|
|
485
|
+
if include_raw:
|
|
486
|
+
print(f' Raw files: {len(manifest["raw_files"])}')
|
|
487
|
+
|
|
488
|
+
return manifest_path
|
|
489
|
+
|
|
490
|
+
# ── Private: loading orchestration ───────────────────────────────────────
|
|
491
|
+
|
|
492
|
+
def _process_yearly_table(self, table, years, download, force_download, force_reload):
|
|
493
|
+
for year in years:
|
|
494
|
+
if download:
|
|
495
|
+
self._download_file(table, year, force_download)
|
|
496
|
+
fp = self._make_file_path(table, year)
|
|
497
|
+
if not fp:
|
|
498
|
+
if self.verbose:
|
|
499
|
+
print(f' Skipping {table} {year}: file not found in {self.data_dir}')
|
|
500
|
+
continue
|
|
501
|
+
cksum = self._checksum(fp)
|
|
502
|
+
if not force_reload and self._get_stored_checksum(table, year) == cksum:
|
|
503
|
+
if self.verbose:
|
|
504
|
+
print(f' {table} {year}: up to date, skipping')
|
|
505
|
+
continue
|
|
506
|
+
rows = self._load_yearly(table, year, fp)
|
|
507
|
+
self._record_load(table, year, os.path.basename(fp), cksum, rows)
|
|
508
|
+
|
|
509
|
+
def _process_cumulative_table(self, table, years, meta, download, force_download, force_reload):
|
|
510
|
+
"""
|
|
511
|
+
Load a cumulative table (master, patient, problem) from its two source
|
|
512
|
+
files: a historical "thru{N}" file and the small current-year file.
|
|
513
|
+
|
|
514
|
+
There's no way to know which specific rows inside a changed cumulative
|
|
515
|
+
file were actually modified — FDA ships one monolithic file per group,
|
|
516
|
+
not a diff. So rather than trying to scope a delete, any checksum
|
|
517
|
+
change in either file triggers a full delete-and-reload of the whole
|
|
518
|
+
table from both currently available source files.
|
|
519
|
+
"""
|
|
520
|
+
current_year = datetime.now().year
|
|
521
|
+
prior_years = sorted(y for y in years if y < current_year)
|
|
522
|
+
curr_years = sorted(y for y in years if y == current_year)
|
|
523
|
+
date_column = meta.get('date_column')
|
|
524
|
+
|
|
525
|
+
groups = []
|
|
526
|
+
if prior_years:
|
|
527
|
+
groups.append((max(prior_years), False))
|
|
528
|
+
if curr_years:
|
|
529
|
+
groups.append((current_year, True))
|
|
530
|
+
if not groups:
|
|
531
|
+
return
|
|
532
|
+
|
|
533
|
+
fetched = []
|
|
534
|
+
for anchor, is_current in groups:
|
|
535
|
+
if download:
|
|
536
|
+
self._download_file(table, anchor, force_download)
|
|
537
|
+
fp = self._make_file_path(table, anchor)
|
|
538
|
+
if not fp:
|
|
539
|
+
if self.verbose:
|
|
540
|
+
label = 'current year' if is_current else f'thru {anchor}'
|
|
541
|
+
print(f' Skipping {table} ({label}): file not found in {self.data_dir}')
|
|
542
|
+
continue
|
|
543
|
+
fetched.append((anchor, is_current, fp, self._checksum(fp)))
|
|
544
|
+
|
|
545
|
+
if len(fetched) != len(groups):
|
|
546
|
+
# Can't safely rebuild the whole table without every expected
|
|
547
|
+
# source file — a partial reload could permanently drop the group
|
|
548
|
+
# we couldn't fetch, since there's no scoped delete to fall back on.
|
|
549
|
+
if self.verbose:
|
|
550
|
+
print(f' Skipping {table}: not all source files available, leaving table untouched')
|
|
551
|
+
return
|
|
552
|
+
|
|
553
|
+
any_changed = force_reload or any(
|
|
554
|
+
self._get_stored_checksum(table, current_year if is_current else anchor) != cksum
|
|
555
|
+
for anchor, is_current, fp, cksum in fetched
|
|
556
|
+
)
|
|
557
|
+
if not any_changed:
|
|
558
|
+
if self.verbose:
|
|
559
|
+
print(f' {table}: up to date, skipping')
|
|
560
|
+
return
|
|
561
|
+
|
|
562
|
+
if self._table_exists(table):
|
|
563
|
+
self.conn.execute(f"DELETE FROM {table}")
|
|
564
|
+
self.conn.execute("DELETE FROM _load_metadata WHERE table_name = ?", [table])
|
|
565
|
+
|
|
566
|
+
for i, (anchor, is_current, fp, cksum) in enumerate(fetched):
|
|
567
|
+
rows = self._load_all(table, fp, date_column=date_column, dedup=(i > 0))
|
|
568
|
+
source_file = os.path.basename(fp)
|
|
569
|
+
if is_current:
|
|
570
|
+
self._record_load(table, current_year, source_file, cksum, rows)
|
|
571
|
+
else:
|
|
572
|
+
if date_column:
|
|
573
|
+
# Record the years this file *actually* contains, rather
|
|
574
|
+
# than assuming it only covers up to the requested anchor —
|
|
575
|
+
# thru-file selection chases whatever's latest-available
|
|
576
|
+
# regardless of the specific year requested, so the real
|
|
577
|
+
# content can cover more than that (see _implied_years).
|
|
578
|
+
covered_rows = self.conn.execute(
|
|
579
|
+
f'SELECT year("{date_column}") AS y, COUNT(*) FROM {table} '
|
|
580
|
+
f"WHERE source_file = '{source_file}' GROUP BY y"
|
|
581
|
+
).fetchall()
|
|
582
|
+
for y, c in covered_rows:
|
|
583
|
+
if y is not None:
|
|
584
|
+
self._record_load(table, y, source_file, cksum, c)
|
|
585
|
+
else:
|
|
586
|
+
# No date column to verify against (patient, problem) — the
|
|
587
|
+
# same latest-available fetch behavior applies, so assume
|
|
588
|
+
# the same full range _implied_years does.
|
|
589
|
+
for y in range(meta['start_year'], current_year):
|
|
590
|
+
self._record_load(table, y, source_file, cksum, rows)
|
|
591
|
+
|
|
592
|
+
if self.verbose:
|
|
593
|
+
total = self.conn.execute(f"SELECT COUNT(*) FROM {table}").fetchone()[0]
|
|
594
|
+
print(f' {table}: {total:,} total rows')
|
|
595
|
+
|
|
596
|
+
# ── Private: DuckDB loading ───────────────────────────────────────────────
|
|
597
|
+
|
|
598
|
+
# DuckDB CSV read options shared across all load methods.
|
|
599
|
+
# all_varchar=true prevents DuckDB from auto-inferring types, which would
|
|
600
|
+
# cause TRY_STRPTIME to fail (it expects VARCHAR, not auto-detected DATE).
|
|
601
|
+
# Types for key columns are handled explicitly in each SELECT.
|
|
602
|
+
_CSV_OPTS = "sep='|', encoding='latin-1', quote='', ignore_errors=true, all_varchar=true, strict_mode=false"
|
|
603
|
+
|
|
604
|
+
def _parse_date_expr(self, col):
|
|
605
|
+
"""Return a SQL expression that parses a VARCHAR date column to DATE."""
|
|
606
|
+
return (
|
|
607
|
+
f"coalesce("
|
|
608
|
+
f"TRY_STRPTIME({col}, '%m/%d/%Y'), "
|
|
609
|
+
f"TRY_STRPTIME({col}, '%Y-%m-%d'), "
|
|
610
|
+
f"TRY_STRPTIME({col}, '%Y/%m/%d'))::DATE"
|
|
611
|
+
)
|
|
612
|
+
|
|
613
|
+
def _load_yearly(self, table, year, filepath):
|
|
614
|
+
"""Load one year's file into its table. Returns row count."""
|
|
615
|
+
if self.verbose:
|
|
616
|
+
print(f' Loading {table} {year}...')
|
|
617
|
+
|
|
618
|
+
fp = filepath.replace("'", "''")
|
|
619
|
+
source_file = os.path.basename(filepath).replace("'", "''")
|
|
620
|
+
|
|
621
|
+
if table == 'device':
|
|
622
|
+
# Parse DATE_RECEIVED and add DEVICE_NAME_CONCAT in one CTE pass.
|
|
623
|
+
date_expr = self._parse_date_expr('DATE_RECEIVED')
|
|
624
|
+
select_sql = f"""
|
|
625
|
+
WITH raw AS (
|
|
626
|
+
SELECT * FROM read_csv('{fp}', {self._CSV_OPTS})
|
|
627
|
+
)
|
|
628
|
+
SELECT * REPLACE ({date_expr} AS DATE_RECEIVED),
|
|
629
|
+
upper(coalesce(BRAND_NAME, '') || '|' ||
|
|
630
|
+
coalesce(GENERIC_NAME, '') || '|' ||
|
|
631
|
+
coalesce(MANUFACTURER_D_NAME, '')) AS DEVICE_NAME_CONCAT,
|
|
632
|
+
'{source_file}' AS source_file
|
|
633
|
+
FROM raw
|
|
634
|
+
"""
|
|
635
|
+
else:
|
|
636
|
+
# text/problem: no date column that needs parsing.
|
|
637
|
+
select_sql = f"""
|
|
638
|
+
SELECT *, '{source_file}' AS source_file
|
|
639
|
+
FROM read_csv('{fp}', {self._CSV_OPTS})
|
|
640
|
+
"""
|
|
641
|
+
|
|
642
|
+
if self._table_exists(table):
|
|
643
|
+
# Scope the delete to exactly this file's prior contribution.
|
|
644
|
+
# Filename <-> year is fixed for life for yearly tables
|
|
645
|
+
# (device2020.zip always means 2020, never anything else), so this
|
|
646
|
+
# is precise and doesn't depend on a per-row date — unlike text,
|
|
647
|
+
# which has no DATE_RECEIVED column at all, or device, where a
|
|
648
|
+
# row's date can fail to parse and never match a year() filter.
|
|
649
|
+
self.conn.execute(
|
|
650
|
+
f"DELETE FROM {table} WHERE source_file = '{source_file}'"
|
|
651
|
+
)
|
|
652
|
+
self._ensure_new_columns(table, filepath)
|
|
653
|
+
self.conn.execute(f"INSERT INTO {table} BY NAME {select_sql}")
|
|
654
|
+
else:
|
|
655
|
+
self.conn.execute(f"CREATE TABLE {table} AS {select_sql}")
|
|
656
|
+
|
|
657
|
+
rows = self.conn.execute(
|
|
658
|
+
f"SELECT COUNT(*) FROM {table} WHERE source_file = '{source_file}'"
|
|
659
|
+
).fetchone()[0]
|
|
660
|
+
|
|
661
|
+
if self.verbose:
|
|
662
|
+
print(f' {rows:,} rows')
|
|
663
|
+
return rows
|
|
664
|
+
|
|
665
|
+
def _load_all(self, table, filepath, date_column=None, dedup=False):
|
|
666
|
+
"""
|
|
667
|
+
Load a cumulative file's full content into table, creating it if it
|
|
668
|
+
doesn't exist yet. Used for master, patient, problem — the caller
|
|
669
|
+
(_process_cumulative_table) is responsible for wiping the table first
|
|
670
|
+
when a fresh load is needed; this just inserts/creates.
|
|
671
|
+
|
|
672
|
+
date_column: if given, that column is parsed from VARCHAR to DATE
|
|
673
|
+
(master has one; patient/problem don't).
|
|
674
|
+
dedup: the current-year file's rows can overlap with what the thru-file
|
|
675
|
+
already loaded — pass True (for every group after the first in a given
|
|
676
|
+
table-processing pass) to only insert rows not already present.
|
|
677
|
+
Compared on the data columns only (EXCLUDE source_file), since a row
|
|
678
|
+
that's identical except for which file it came from should still count
|
|
679
|
+
as a duplicate; source_file is attached only after dedup so it doesn't
|
|
680
|
+
interfere with the comparison.
|
|
681
|
+
"""
|
|
682
|
+
if self.verbose:
|
|
683
|
+
print(f' Loading {table} ({os.path.basename(filepath)})...')
|
|
684
|
+
|
|
685
|
+
fp = filepath.replace("'", "''")
|
|
686
|
+
source_file = os.path.basename(filepath).replace("'", "''")
|
|
687
|
+
|
|
688
|
+
if table == 'problem':
|
|
689
|
+
# foidevproblem has no header row — name columns explicitly so DuckDB
|
|
690
|
+
# doesn't fall back to column0/column1/column2 naming.
|
|
691
|
+
# The file has shipped with 2 or 3 columns depending on the release year.
|
|
692
|
+
_opts = (f"sep='|', encoding='latin-1', quote='', ignore_errors=true, "
|
|
693
|
+
f"all_varchar=true, strict_mode=false, header=false")
|
|
694
|
+
with open(filepath, 'r', encoding='latin1') as _f:
|
|
695
|
+
_ncols = len(_f.readline().split('|'))
|
|
696
|
+
if _ncols >= 3:
|
|
697
|
+
_col_select = ("column0 AS MDR_REPORT_KEY, "
|
|
698
|
+
"column1 AS DEVICE_PROBLEM_CODE, "
|
|
699
|
+
"column2 AS DATE_ADDED_FLAG")
|
|
700
|
+
else:
|
|
701
|
+
_col_select = ("column0 AS MDR_REPORT_KEY, "
|
|
702
|
+
"column1 AS DEVICE_PROBLEM_CODE")
|
|
703
|
+
raw_select_sql = f"""
|
|
704
|
+
SELECT {_col_select}
|
|
705
|
+
FROM read_csv('{fp}', {_opts})
|
|
706
|
+
"""
|
|
707
|
+
if self._table_exists(table):
|
|
708
|
+
# Migrate column names if db was created before explicit-naming fix.
|
|
709
|
+
existing = {r[0] for r in self.conn.execute("DESCRIBE problem").fetchall()}
|
|
710
|
+
if 'column0' in existing and 'MDR_REPORT_KEY' not in existing:
|
|
711
|
+
for i, name in enumerate(['MDR_REPORT_KEY', 'DEVICE_PROBLEM_CODE', 'DATE_ADDED_FLAG']):
|
|
712
|
+
if f'column{i}' in existing:
|
|
713
|
+
self.conn.execute(f'ALTER TABLE problem RENAME COLUMN "column{i}" TO "{name}"')
|
|
714
|
+
else:
|
|
715
|
+
if date_column:
|
|
716
|
+
date_expr = self._parse_date_expr(f'"{date_column}"')
|
|
717
|
+
raw_select_sql = f"""
|
|
718
|
+
SELECT * REPLACE ({date_expr} AS "{date_column}")
|
|
719
|
+
FROM read_csv('{fp}', {self._CSV_OPTS})
|
|
720
|
+
"""
|
|
721
|
+
else:
|
|
722
|
+
raw_select_sql = f"""
|
|
723
|
+
SELECT * FROM read_csv('{fp}', {self._CSV_OPTS})
|
|
724
|
+
"""
|
|
725
|
+
if self._table_exists(table):
|
|
726
|
+
self._ensure_new_columns(table, filepath)
|
|
727
|
+
|
|
728
|
+
if self._table_exists(table):
|
|
729
|
+
if not dedup:
|
|
730
|
+
self.conn.execute(f"""
|
|
731
|
+
INSERT INTO {table} BY NAME
|
|
732
|
+
SELECT *, '{source_file}' AS source_file FROM ({raw_select_sql})
|
|
733
|
+
""")
|
|
734
|
+
else:
|
|
735
|
+
self.conn.execute(f"""
|
|
736
|
+
WITH truly_new AS (
|
|
737
|
+
SELECT * FROM ({raw_select_sql})
|
|
738
|
+
EXCEPT
|
|
739
|
+
SELECT * EXCLUDE (source_file) FROM {table}
|
|
740
|
+
)
|
|
741
|
+
INSERT INTO {table} BY NAME
|
|
742
|
+
SELECT *, '{source_file}' AS source_file FROM truly_new
|
|
743
|
+
""")
|
|
744
|
+
else:
|
|
745
|
+
self.conn.execute(f"""
|
|
746
|
+
CREATE TABLE {table} AS
|
|
747
|
+
SELECT *, '{source_file}' AS source_file FROM ({raw_select_sql})
|
|
748
|
+
""")
|
|
749
|
+
|
|
750
|
+
rows = self.conn.execute(
|
|
751
|
+
f"SELECT COUNT(*) FROM {table} WHERE source_file = '{source_file}'"
|
|
752
|
+
).fetchone()[0]
|
|
753
|
+
if self.verbose:
|
|
754
|
+
print(f' {rows:,} rows')
|
|
755
|
+
return rows
|
|
756
|
+
|
|
757
|
+
def _ensure_new_columns(self, table, filepath):
|
|
758
|
+
"""
|
|
759
|
+
Add any columns present in the CSV that are missing from the DuckDB table.
|
|
760
|
+
Handles MAUDE's schema variations across years (e.g., new fields added in later files).
|
|
761
|
+
"""
|
|
762
|
+
with open(filepath, 'r', encoding='latin1') as f:
|
|
763
|
+
csv_cols = set(f.readline().strip().split('|'))
|
|
764
|
+
|
|
765
|
+
existing = {row[0] for row in self.conn.execute(f"DESCRIBE {table}").fetchall()}
|
|
766
|
+
|
|
767
|
+
for col in csv_cols:
|
|
768
|
+
col = col.strip()
|
|
769
|
+
if col and col not in existing:
|
|
770
|
+
try:
|
|
771
|
+
self.conn.execute(f'ALTER TABLE "{table}" ADD COLUMN "{col}" VARCHAR')
|
|
772
|
+
except Exception:
|
|
773
|
+
pass
|
|
774
|
+
|
|
775
|
+
def _create_indexes(self):
|
|
776
|
+
"""Create indexes on join keys. DuckDB ART indexes help point lookups and joins."""
|
|
777
|
+
for table, col in [
|
|
778
|
+
('master', 'MDR_REPORT_KEY'),
|
|
779
|
+
('master', 'DATE_RECEIVED'),
|
|
780
|
+
('device', 'MDR_REPORT_KEY'),
|
|
781
|
+
('device', 'DEVICE_REPORT_PRODUCT_CODE'),
|
|
782
|
+
('text', 'MDR_REPORT_KEY'),
|
|
783
|
+
('patient', 'MDR_REPORT_KEY'),
|
|
784
|
+
('problem', 'MDR_REPORT_KEY'),
|
|
785
|
+
]:
|
|
786
|
+
if self._table_exists(table):
|
|
787
|
+
idx = f"idx_{table}_{col.lower()}"
|
|
788
|
+
try:
|
|
789
|
+
self.conn.execute(
|
|
790
|
+
f'CREATE INDEX IF NOT EXISTS {idx} ON "{table}"("{col}")'
|
|
791
|
+
)
|
|
792
|
+
except Exception:
|
|
793
|
+
pass
|
|
794
|
+
|
|
795
|
+
# ── Private: downloads ────────────────────────────────────────────────────
|
|
796
|
+
|
|
797
|
+
def _download_file(self, table, year, force_download=False):
|
|
798
|
+
"""Download and extract a MAUDE zip from the FDA FTP area."""
|
|
799
|
+
url, filename = self._construct_url(table, year)
|
|
800
|
+
if not url:
|
|
801
|
+
return False
|
|
802
|
+
|
|
803
|
+
cache_key = (table, filename)
|
|
804
|
+
if not force_download and cache_key in self._download_cache:
|
|
805
|
+
return True
|
|
806
|
+
|
|
807
|
+
zip_path = os.path.join(self.data_dir, filename)
|
|
808
|
+
|
|
809
|
+
if not force_download and os.path.exists(zip_path):
|
|
810
|
+
if self.verbose:
|
|
811
|
+
print(f' Using cached {filename}')
|
|
812
|
+
try:
|
|
813
|
+
with zipfile.ZipFile(zip_path, 'r') as z:
|
|
814
|
+
z.extractall(self.data_dir)
|
|
815
|
+
self._download_cache.add(cache_key)
|
|
816
|
+
return True
|
|
817
|
+
except Exception:
|
|
818
|
+
os.remove(zip_path)
|
|
819
|
+
|
|
820
|
+
try:
|
|
821
|
+
if self.verbose:
|
|
822
|
+
print(f' Downloading {filename}...')
|
|
823
|
+
headers = {'User-Agent': 'Mozilla/5.0'}
|
|
824
|
+
r = requests.get(url, headers=headers, timeout=60)
|
|
825
|
+
r.raise_for_status()
|
|
826
|
+
with open(zip_path, 'wb') as f:
|
|
827
|
+
f.write(r.content)
|
|
828
|
+
with zipfile.ZipFile(zip_path, 'r') as z:
|
|
829
|
+
z.extractall(self.data_dir)
|
|
830
|
+
self._download_cache.add(cache_key)
|
|
831
|
+
return True
|
|
832
|
+
except Exception as e:
|
|
833
|
+
if self.verbose:
|
|
834
|
+
print(f' Error downloading {filename}: {e}')
|
|
835
|
+
return False
|
|
836
|
+
|
|
837
|
+
def _construct_url(self, table, year):
|
|
838
|
+
"""
|
|
839
|
+
Return (url, filename) for a given table and year.
|
|
840
|
+
|
|
841
|
+
Yearly tables: device{year}.zip / {prefix}{year}.zip
|
|
842
|
+
Cumulative: {prefix}thru{N}.zip (N = most recent available year)
|
|
843
|
+
Current year: {current_year_prefix}.zip
|
|
844
|
+
"""
|
|
845
|
+
if table not in TABLE_METADATA:
|
|
846
|
+
return None, None
|
|
847
|
+
|
|
848
|
+
meta = TABLE_METADATA[table]
|
|
849
|
+
prefix = meta['file_prefix']
|
|
850
|
+
current_year = datetime.now().year
|
|
851
|
+
|
|
852
|
+
if year == current_year:
|
|
853
|
+
filename = f"{meta['current_year_prefix']}.zip"
|
|
854
|
+
return f"{FDA_BASE_URL}/{filename}", filename
|
|
855
|
+
|
|
856
|
+
if meta['pattern_type'] == 'yearly':
|
|
857
|
+
# Device table uses 'device{year}.zip', others use '{prefix}{year}.zip'.
|
|
858
|
+
filename = f"device{year}.zip" if table == 'device' else f"{prefix}{year}.zip"
|
|
859
|
+
return f"{FDA_BASE_URL}/{filename}", filename
|
|
860
|
+
|
|
861
|
+
# Cumulative: FDA releases thru{prev_year} files; probe for the latest available.
|
|
862
|
+
sep = meta.get('thru_separator', '')
|
|
863
|
+
for offset in [1, 2, 3]:
|
|
864
|
+
thru_year = current_year - offset
|
|
865
|
+
filename = f"{prefix}{sep}thru{thru_year}.zip"
|
|
866
|
+
url = f"{FDA_BASE_URL}/{filename}"
|
|
867
|
+
if self._url_exists(url):
|
|
868
|
+
if offset > 1 and self.verbose:
|
|
869
|
+
print(f' Note: using {filename} (expected {prefix}{sep}thru{current_year - 1} not available)')
|
|
870
|
+
return url, filename
|
|
871
|
+
|
|
872
|
+
filename = f"{prefix}{sep}thru{current_year - 1}.zip"
|
|
873
|
+
return f"{FDA_BASE_URL}/{filename}", filename
|
|
874
|
+
|
|
875
|
+
def _url_exists(self, url):
|
|
876
|
+
try:
|
|
877
|
+
r = requests.head(url, headers={'User-Agent': 'Mozilla/5.0'},
|
|
878
|
+
timeout=5, allow_redirects=True)
|
|
879
|
+
return 200 <= r.status_code < 300
|
|
880
|
+
except Exception:
|
|
881
|
+
return False
|
|
882
|
+
|
|
883
|
+
def _make_file_path(self, table, year):
|
|
884
|
+
"""
|
|
885
|
+
Find the extracted .txt file for a table/year in data_dir.
|
|
886
|
+
Returns full path or None if not found.
|
|
887
|
+
"""
|
|
888
|
+
if table not in TABLE_METADATA:
|
|
889
|
+
return None
|
|
890
|
+
|
|
891
|
+
meta = TABLE_METADATA[table]
|
|
892
|
+
prefix = meta['file_prefix']
|
|
893
|
+
current_year = datetime.now().year
|
|
894
|
+
|
|
895
|
+
try:
|
|
896
|
+
files = set(os.listdir(self.data_dir))
|
|
897
|
+
except FileNotFoundError:
|
|
898
|
+
return None
|
|
899
|
+
|
|
900
|
+
candidates = []
|
|
901
|
+
|
|
902
|
+
if year == current_year:
|
|
903
|
+
cp = meta['current_year_prefix']
|
|
904
|
+
candidates += [f"{cp}.txt", f"{cp.upper()}.txt"]
|
|
905
|
+
|
|
906
|
+
if meta['pattern_type'] == 'yearly':
|
|
907
|
+
if table == 'device':
|
|
908
|
+
candidates += [f"device{year}.txt", f"DEVICE{year}.txt"]
|
|
909
|
+
else:
|
|
910
|
+
candidates += [f"{prefix}{year}.txt", f"{prefix.upper()}{year}.txt"]
|
|
911
|
+
elif meta['pattern_type'] == 'cumulative':
|
|
912
|
+
# Cumulative: check for thru files from most recent backwards.
|
|
913
|
+
sep = meta.get('thru_separator', '')
|
|
914
|
+
for offset in [1, 2, 3]:
|
|
915
|
+
thru_year = current_year - offset
|
|
916
|
+
candidates += [
|
|
917
|
+
f"{prefix}{sep}thru{thru_year}.txt",
|
|
918
|
+
f"{prefix.upper()}{sep}thru{thru_year}.txt",
|
|
919
|
+
]
|
|
920
|
+
# Fallback: any file matching the cumulative pattern.
|
|
921
|
+
for fn in sorted(files):
|
|
922
|
+
if (fn.lower().startswith(prefix.lower())
|
|
923
|
+
and 'thru' in fn.lower()
|
|
924
|
+
and fn.endswith('.txt')):
|
|
925
|
+
candidates.append(fn)
|
|
926
|
+
|
|
927
|
+
for c in candidates:
|
|
928
|
+
if c in files:
|
|
929
|
+
return os.path.join(self.data_dir, c)
|
|
930
|
+
return None
|
|
931
|
+
|
|
932
|
+
# ── Private: checksum / metadata ─────────────────────────────────────────
|
|
933
|
+
|
|
934
|
+
def _init_metadata_table(self):
|
|
935
|
+
self.conn.execute("""
|
|
936
|
+
CREATE TABLE IF NOT EXISTS _load_metadata (
|
|
937
|
+
table_name VARCHAR,
|
|
938
|
+
year INTEGER,
|
|
939
|
+
source_file VARCHAR,
|
|
940
|
+
checksum VARCHAR,
|
|
941
|
+
row_count INTEGER,
|
|
942
|
+
loaded_at TIMESTAMP
|
|
943
|
+
)
|
|
944
|
+
""")
|
|
945
|
+
|
|
946
|
+
def _get_stored_checksum(self, table, year):
|
|
947
|
+
row = self.conn.execute(
|
|
948
|
+
"SELECT checksum FROM _load_metadata WHERE table_name = ? AND year = ?",
|
|
949
|
+
[table, year]
|
|
950
|
+
).fetchone()
|
|
951
|
+
return row[0] if row else None
|
|
952
|
+
|
|
953
|
+
def _record_load(self, table, year, source_file, checksum, rows):
|
|
954
|
+
self.conn.execute(
|
|
955
|
+
"DELETE FROM _load_metadata WHERE table_name = ? AND year = ?",
|
|
956
|
+
[table, year]
|
|
957
|
+
)
|
|
958
|
+
self.conn.execute(
|
|
959
|
+
"INSERT INTO _load_metadata VALUES (?, ?, ?, ?, ?, current_timestamp)",
|
|
960
|
+
[table, year, source_file, checksum, rows]
|
|
961
|
+
)
|
|
962
|
+
|
|
963
|
+
def _checksum(self, filepath):
|
|
964
|
+
h = hashlib.sha256()
|
|
965
|
+
with open(filepath, 'rb') as f:
|
|
966
|
+
for chunk in iter(lambda: f.read(65536), b''):
|
|
967
|
+
h.update(chunk)
|
|
968
|
+
return h.hexdigest()
|
|
969
|
+
|
|
970
|
+
def _table_exists(self, table):
|
|
971
|
+
return self.conn.execute(
|
|
972
|
+
"SELECT COUNT(*) FROM information_schema.tables "
|
|
973
|
+
"WHERE table_name = ? AND table_schema = 'main'",
|
|
974
|
+
[table]
|
|
975
|
+
).fetchone()[0] > 0
|
|
976
|
+
|
|
977
|
+
def _get_years_in_db(self):
|
|
978
|
+
try:
|
|
979
|
+
rows = self.conn.execute(
|
|
980
|
+
"SELECT DISTINCT year FROM _load_metadata"
|
|
981
|
+
).fetchall()
|
|
982
|
+
return sorted(r[0] for r in rows)
|
|
983
|
+
except Exception:
|
|
984
|
+
return []
|
|
985
|
+
|
|
986
|
+
def _get_loaded_tables(self):
|
|
987
|
+
try:
|
|
988
|
+
rows = self.conn.execute(
|
|
989
|
+
"SELECT DISTINCT table_name FROM _load_metadata"
|
|
990
|
+
).fetchall()
|
|
991
|
+
return [r[0] for r in rows]
|
|
992
|
+
except Exception:
|
|
993
|
+
return []
|
|
994
|
+
|
|
995
|
+
# ── Private: year parsing / validation ───────────────────────────────────
|
|
996
|
+
|
|
997
|
+
def _parse_year_range(self, years):
|
|
998
|
+
if isinstance(years, int):
|
|
999
|
+
return [years]
|
|
1000
|
+
if isinstance(years, list):
|
|
1001
|
+
return years
|
|
1002
|
+
s = str(years)
|
|
1003
|
+
if s == 'all':
|
|
1004
|
+
return list(range(1991, datetime.now().year + 1))
|
|
1005
|
+
if s == 'latest':
|
|
1006
|
+
return [datetime.now().year - 1]
|
|
1007
|
+
if s == 'current':
|
|
1008
|
+
return [datetime.now().year]
|
|
1009
|
+
if '-' in s:
|
|
1010
|
+
a, b = s.split('-', 1)
|
|
1011
|
+
return list(range(int(a), int(b) + 1))
|
|
1012
|
+
return [int(s)]
|
|
1013
|
+
|
|
1014
|
+
def _validate(self, years, tables):
|
|
1015
|
+
"""Return list of (year, table) pairs that are valid to load."""
|
|
1016
|
+
valid = []
|
|
1017
|
+
current_year = datetime.now().year
|
|
1018
|
+
for table in tables:
|
|
1019
|
+
if table not in TABLE_METADATA:
|
|
1020
|
+
if self.verbose:
|
|
1021
|
+
print(f" Unknown table '{table}', skipping")
|
|
1022
|
+
continue
|
|
1023
|
+
start_year = TABLE_METADATA[table]['start_year']
|
|
1024
|
+
for year in years:
|
|
1025
|
+
if year < start_year:
|
|
1026
|
+
if self.verbose:
|
|
1027
|
+
print(f" Skipping {table} {year}: available from {start_year} onwards")
|
|
1028
|
+
continue
|
|
1029
|
+
if year > current_year:
|
|
1030
|
+
if self.verbose:
|
|
1031
|
+
print(f" Skipping {table} {year}: future year")
|
|
1032
|
+
continue
|
|
1033
|
+
valid.append((year, table))
|
|
1034
|
+
return valid
|
|
1035
|
+
|
|
1036
|
+
def _covered_years(self, table):
|
|
1037
|
+
"""Years currently recorded as loaded for `table`, from _load_metadata."""
|
|
1038
|
+
rows = self.conn.execute(
|
|
1039
|
+
"SELECT DISTINCT year FROM _load_metadata WHERE table_name = ?", [table]
|
|
1040
|
+
).fetchall()
|
|
1041
|
+
return {r[0] for r in rows}
|
|
1042
|
+
|
|
1043
|
+
def _implied_years(self, table, years_for_table, meta):
|
|
1044
|
+
"""
|
|
1045
|
+
Years `table` would end up covering after loading `years_for_table`.
|
|
1046
|
+
|
|
1047
|
+
Yearly tables (device, text): each year is an independent file, so this
|
|
1048
|
+
is just the requested years themselves.
|
|
1049
|
+
|
|
1050
|
+
Cumulative tables (master, patient, problem): the "thru{N}" file fetch
|
|
1051
|
+
(_construct_url/_make_file_path) always resolves to whichever historical
|
|
1052
|
+
dump is actually latest-available, essentially ignoring the specific
|
|
1053
|
+
prior year requested — so requesting any prior year is treated as
|
|
1054
|
+
implying coverage through last year, not just up to the requested year.
|
|
1055
|
+
The current year, if requested, comes from a separate, current-year-only
|
|
1056
|
+
file and only implies itself.
|
|
1057
|
+
"""
|
|
1058
|
+
if meta['pattern_type'] == 'yearly':
|
|
1059
|
+
return set(years_for_table)
|
|
1060
|
+
|
|
1061
|
+
current_year = datetime.now().year
|
|
1062
|
+
prior = [y for y in years_for_table if y < current_year]
|
|
1063
|
+
curr = [y for y in years_for_table if y == current_year]
|
|
1064
|
+
implied = set()
|
|
1065
|
+
if prior:
|
|
1066
|
+
implied |= set(range(meta['start_year'], current_year))
|
|
1067
|
+
if curr:
|
|
1068
|
+
implied.add(current_year)
|
|
1069
|
+
return implied
|
|
1070
|
+
|
|
1071
|
+
# ── Private: search helpers ───────────────────────────────────────────────
|
|
1072
|
+
|
|
1073
|
+
def _normalize_criteria(self, criteria):
|
|
1074
|
+
"""
|
|
1075
|
+
Normalize criteria to list-of-lists (each inner list = AND group, outer = OR).
|
|
1076
|
+
|
|
1077
|
+
'term' → [['term']]
|
|
1078
|
+
['a', 'b'] → [['a'], ['b']]
|
|
1079
|
+
[['a', 'b'], 'c'] → [['a', 'b'], ['c']]
|
|
1080
|
+
"""
|
|
1081
|
+
if isinstance(criteria, str):
|
|
1082
|
+
return [[criteria]]
|
|
1083
|
+
if not isinstance(criteria, list):
|
|
1084
|
+
raise ValueError("criteria must be a string or list")
|
|
1085
|
+
result = []
|
|
1086
|
+
for item in criteria:
|
|
1087
|
+
if isinstance(item, str):
|
|
1088
|
+
result.append([item])
|
|
1089
|
+
elif isinstance(item, list):
|
|
1090
|
+
if not item:
|
|
1091
|
+
raise ValueError("Empty AND group in criteria")
|
|
1092
|
+
for t in item:
|
|
1093
|
+
if not isinstance(t, str):
|
|
1094
|
+
raise ValueError(f"All search terms must be strings, got {type(t).__name__}")
|
|
1095
|
+
result.append(item)
|
|
1096
|
+
else:
|
|
1097
|
+
raise ValueError(f"Criteria items must be str or list, got {type(item).__name__}")
|
|
1098
|
+
if not result:
|
|
1099
|
+
raise ValueError("criteria cannot be empty")
|
|
1100
|
+
return result
|
pymaude/metadata.py
ADDED
|
@@ -0,0 +1,54 @@
|
|
|
1
|
+
# metadata.py - MAUDE table configuration
|
|
2
|
+
# Copyright (C) 2026 Jacob Schwartz <jaschwa@umich.edu>
|
|
3
|
+
# MIT License
|
|
4
|
+
|
|
5
|
+
FDA_BASE_URL = "https://www.accessdata.fda.gov/MAUDE/ftparea"
|
|
6
|
+
FDA_PREMARKET_URL = "https://www.accessdata.fda.gov/premarket/ftparea"
|
|
7
|
+
|
|
8
|
+
# Table metadata: file patterns, availability, and structure
|
|
9
|
+
TABLE_METADATA = {
|
|
10
|
+
'master': {
|
|
11
|
+
'file_prefix': 'mdrfoi',
|
|
12
|
+
'pattern_type': 'cumulative', # mdrfoithru{year}.zip
|
|
13
|
+
'current_year_prefix': 'mdrfoi', # mdrfoi.zip for current year
|
|
14
|
+
'start_year': 2000,
|
|
15
|
+
'date_column': 'DATE_RECEIVED',
|
|
16
|
+
'description': 'Master records (adverse event reports)',
|
|
17
|
+
},
|
|
18
|
+
'device': {
|
|
19
|
+
'file_prefix': 'foidev',
|
|
20
|
+
'pattern_type': 'yearly', # device{year}.zip (special naming)
|
|
21
|
+
'current_year_prefix': 'device', # device.zip for current year
|
|
22
|
+
'start_year': 2000,
|
|
23
|
+
'date_column': 'DATE_RECEIVED',
|
|
24
|
+
'description': 'Device information',
|
|
25
|
+
},
|
|
26
|
+
'text': {
|
|
27
|
+
'file_prefix': 'foitext',
|
|
28
|
+
'pattern_type': 'yearly', # foitext{year}.zip
|
|
29
|
+
'current_year_prefix': 'foitext',
|
|
30
|
+
'start_year': 2000,
|
|
31
|
+
'description': 'Event narrative text (FOI_TEXT)',
|
|
32
|
+
},
|
|
33
|
+
'patient': {
|
|
34
|
+
'file_prefix': 'patient',
|
|
35
|
+
'pattern_type': 'cumulative', # patientthru{year}.zip
|
|
36
|
+
'current_year_prefix': 'patient',
|
|
37
|
+
'start_year': 2000,
|
|
38
|
+
# No date_column: patient table has no date field; joins to master via MDR_REPORT_KEY.
|
|
39
|
+
# The entire cumulative file is loaded (no year filtering possible).
|
|
40
|
+
'description': 'Patient demographics and outcomes',
|
|
41
|
+
'size_warning': (
|
|
42
|
+
'Patient data is distributed as a single large cumulative file. '
|
|
43
|
+
'All historical data will be downloaded and loaded.'
|
|
44
|
+
),
|
|
45
|
+
},
|
|
46
|
+
'problem': {
|
|
47
|
+
'file_prefix': 'foidevproblem',
|
|
48
|
+
'pattern_type': 'cumulative', # foidevproblem_thru{year}.zip + foidevproblem.zip
|
|
49
|
+
'current_year_prefix': 'foidevproblem',
|
|
50
|
+
'thru_separator': '_', # filename: foidevproblem_thru{year}.zip
|
|
51
|
+
'start_year': 2000,
|
|
52
|
+
'description': 'Device problem codes',
|
|
53
|
+
},
|
|
54
|
+
}
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: pymaude
|
|
3
|
+
Version: 0.2.0
|
|
4
|
+
Summary: PyMAUDE: FDA MAUDE adverse event database interface
|
|
5
|
+
Author-email: Jacob Schwartz <jaschwa@umich.edu>
|
|
6
|
+
License: MIT
|
|
7
|
+
Classifier: Development Status :: 4 - Beta
|
|
8
|
+
Classifier: Intended Audience :: Science/Research
|
|
9
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
10
|
+
Classifier: Programming Language :: Python :: 3
|
|
11
|
+
Requires-Python: >=3.9
|
|
12
|
+
Description-Content-Type: text/markdown
|
|
13
|
+
License-File: LICENSE
|
|
14
|
+
Requires-Dist: duckdb>=0.10.0
|
|
15
|
+
Requires-Dist: pandas>=1.3.0
|
|
16
|
+
Requires-Dist: requests>=2.25.0
|
|
17
|
+
Requires-Dist: pyyaml>=5.1
|
|
18
|
+
Provides-Extra: dev
|
|
19
|
+
Requires-Dist: pytest>=7.0; extra == "dev"
|
|
20
|
+
Requires-Dist: pytest-cov; extra == "dev"
|
|
21
|
+
Dynamic: license-file
|
|
@@ -0,0 +1,8 @@
|
|
|
1
|
+
pymaude/__init__.py,sha256=t9JOQuhJvE-BMfT8WPjTs5gFlkS2-7s5DofExY6CDsk,366
|
|
2
|
+
pymaude/database.py,sha256=kAkgTsxsRaR6IUOlgjCFM9U99H8pQChXQ02IM7vNqxs,46065
|
|
3
|
+
pymaude/metadata.py,sha256=_MTodSJAeCaIR-lcaf5_hUCx39vm5MVpjZx8hJAcaIc,2204
|
|
4
|
+
pymaude-0.2.0.dist-info/licenses/LICENSE,sha256=U2oyySnfshXoflE-ms876S4mtBbXswAK79Nh7aBWThY,1071
|
|
5
|
+
pymaude-0.2.0.dist-info/METADATA,sha256=uf0hl1adrrC0XUfj3bk861HyjnWcGRpVnAFIiwqlBaY,696
|
|
6
|
+
pymaude-0.2.0.dist-info/WHEEL,sha256=YVMoNqKzERt-wjUZwJ33xBGAwnFl-4cqbYkTtWa4itE,91
|
|
7
|
+
pymaude-0.2.0.dist-info/top_level.txt,sha256=hmUVXXLoByVQeCIALqwnycXMbv0--YfpcT1rI4a60Oc,8
|
|
8
|
+
pymaude-0.2.0.dist-info/RECORD,,
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Jacob Schwartz
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
pymaude
|