pymaude 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
pymaude/__init__.py ADDED
@@ -0,0 +1,14 @@
1
+ # PyMAUDE - FDA MAUDE Database Interface (DuckDB backend)
2
+ # Copyright (C) 2026 Jacob Schwartz <jaschwa@umich.edu>
3
+ # MIT License
4
+
5
+ from .database import MaudeDatabase
6
+ from .metadata import TABLE_METADATA, FDA_BASE_URL
7
+
8
+ __version__ = '0.2.0'
9
+ __author__ = 'Jacob Schwartz <jaschwa@umich.edu>'
10
+ __all__ = [
11
+ 'MaudeDatabase',
12
+ 'TABLE_METADATA',
13
+ 'FDA_BASE_URL',
14
+ ]
pymaude/database.py ADDED
@@ -0,0 +1,1100 @@
1
+ # database.py - FDA MAUDE Database Interface (DuckDB backend)
2
+ # Copyright (C) 2026 Jacob Schwartz <jaschwa@umich.edu>
3
+ # MIT License
4
+
5
+ """
6
+ MaudeDatabase: download, load, and query FDA MAUDE adverse event data.
7
+
8
+ Usage:
9
+ db = MaudeDatabase('maude.duckdb', data_dir='./maude_data')
10
+ db.add_years('2019-2024', tables=['master', 'device', 'text'], download=True)
11
+ results = db.search_by_device_names([['argon', 'cleaner'], 'angiojet'])
12
+ narratives = db.get_narratives(results['MDR_REPORT_KEY'])
13
+ db.close()
14
+ """
15
+
16
+ import os
17
+ import json
18
+ import shutil
19
+ import zipfile
20
+ import hashlib
21
+ from collections import defaultdict
22
+ from datetime import datetime
23
+
24
+ import duckdb
25
+ import pandas as pd
26
+ import requests
27
+
28
+ from .metadata import TABLE_METADATA, FDA_BASE_URL
29
+
30
+
31
+ class MaudeDatabase:
32
+ """
33
+ Interface to FDA MAUDE database via DuckDB.
34
+
35
+ Data is stored in a persistent DuckDB file. Raw downloaded CSVs are kept
36
+ in data_dir and can be deleted after loading if disk space is a concern.
37
+
38
+ Args:
39
+ db_path: Path to DuckDB database file (created if it doesn't exist).
40
+ data_dir: Directory for downloaded/extracted MAUDE files.
41
+ verbose: Print progress messages (default True).
42
+ memory_limit: DuckDB memory cap (e.g. '4GB', '512MB'). When set,
43
+ DuckDB spills intermediate data to disk instead of OOMing during
44
+ large CSV loads. Defaults to None (DuckDB manages its own limit).
45
+ """
46
+
47
+ def __init__(self, db_path, data_dir='./maude_data', verbose=True, memory_limit=None):
48
+ self.db_path = db_path
49
+ self.data_dir = data_dir
50
+ self.verbose = verbose
51
+ self._download_cache = set()
52
+
53
+ os.makedirs(data_dir, exist_ok=True)
54
+ self.conn = duckdb.connect(db_path)
55
+ if memory_limit is not None:
56
+ self.conn.execute(f"SET memory_limit='{memory_limit}'")
57
+ self._init_metadata_table()
58
+
59
+ def __enter__(self):
60
+ return self
61
+
62
+ def __exit__(self, *args):
63
+ self.close()
64
+
65
+ # ── Public: data management ───────────────────────────────────────────────
66
+
67
+ def add_years(self, years, tables=None, download=False,
68
+ force_download=False, force_reload=False, force_partial=False):
69
+ """
70
+ Load MAUDE data for the specified years into the database.
71
+
72
+ Uses checksum tracking to skip files that haven't changed since last load.
73
+
74
+ Args:
75
+ years: Years to load. One of:
76
+ - int: 2024
77
+ - list: [2020, 2021, 2022]
78
+ - str range: '2019-2024'
79
+ - 'all', 'latest', 'current'
80
+ tables: List of tables to load (default: ['master', 'device', 'text', 'patient']).
81
+ download: If True, download files from FDA before loading.
82
+ force_download: If True, re-download even if zip exists locally.
83
+ force_reload: If True, reload into DB even if checksum is unchanged.
84
+ force_partial: Cumulative tables (master, patient, problem) are fully
85
+ replaced on every reload, so a request that doesn't cover years
86
+ already loaded for one of them would silently drop those years.
87
+ That's rejected by default — pass True to proceed anyway. Doesn't
88
+ apply to yearly tables (device, text), which can't lose data this
89
+ way since each year is an independent file.
90
+ """
91
+ years_list = self._parse_year_range(years)
92
+ if tables is None:
93
+ tables = ['master', 'device', 'text', 'patient']
94
+
95
+ valid = self._validate(years_list, tables)
96
+ if not valid:
97
+ return
98
+
99
+ years_by_table = defaultdict(list)
100
+ for year, table in valid:
101
+ years_by_table[table].append(year)
102
+
103
+ if not force_partial:
104
+ for table in sorted(years_by_table):
105
+ meta = TABLE_METADATA[table]
106
+ if meta['pattern_type'] != 'cumulative':
107
+ continue
108
+ years_for_table = sorted(years_by_table[table])
109
+ implied = self._implied_years(table, years_for_table, meta)
110
+ missing = self._covered_years(table) - implied
111
+ if missing:
112
+ raise ValueError(
113
+ f"add_years({years_for_table}, tables=['{table}']) would not "
114
+ f"cover years {sorted(missing)} that '{table}' already has "
115
+ f"loaded. '{table}' is fully replaced on every reload, so this "
116
+ f"would silently drop that data. Call update() to refresh "
117
+ f"everything '{table}' has, or pass force_partial=True to "
118
+ f"proceed and accept the loss."
119
+ )
120
+
121
+ for table in sorted(years_by_table):
122
+ meta = TABLE_METADATA[table]
123
+ years_for_table = sorted(years_by_table[table])
124
+
125
+ if meta['pattern_type'] == 'yearly':
126
+ self._process_yearly_table(
127
+ table, years_for_table, download, force_download, force_reload
128
+ )
129
+ else:
130
+ self._process_cumulative_table(
131
+ table, years_for_table, meta, download, force_download, force_reload
132
+ )
133
+
134
+ self._create_indexes()
135
+
136
+ def update(self, download=True, force_download=True):
137
+ """
138
+ Refresh all currently-loaded years and add any new years since the last load.
139
+
140
+ Args:
141
+ download: If True, download updated files from FDA.
142
+ force_download: If True, re-download even if zip files exist locally.
143
+ """
144
+ years = self._get_years_in_db()
145
+ if not years:
146
+ if self.verbose:
147
+ print('Database is empty. Use add_years() to populate.')
148
+ return
149
+
150
+ all_years = list(range(min(years), datetime.now().year + 1))
151
+ loaded_tables = self._get_loaded_tables()
152
+ if self.verbose:
153
+ print(f'Refreshing {len(loaded_tables)} tables, years {min(years)}–{datetime.now().year}')
154
+ self.add_years(all_years, tables=loaded_tables, download=download,
155
+ force_download=force_download)
156
+
157
+ # ── Public: queries ───────────────────────────────────────────────────────
158
+
159
+ def query_device(self, brand_name=None, generic_name=None,
160
+ manufacturer_name=None, product_code=None,
161
+ start_date=None, end_date=None):
162
+ """
163
+ Query device events by exact field matching (case-insensitive).
164
+
165
+ All provided parameters are combined with AND logic. At least one
166
+ device field (brand_name, generic_name, manufacturer_name, product_code)
167
+ is required.
168
+
169
+ For partial/substring matching or boolean logic, use search_by_device_names().
170
+
171
+ Args:
172
+ brand_name: Exact BRAND_NAME match (e.g., 'Venovo').
173
+ generic_name: Exact GENERIC_NAME match (e.g., 'Venous Stent').
174
+ manufacturer_name: Exact MANUFACTURER_D_NAME match.
175
+ product_code: Exact DEVICE_REPORT_PRODUCT_CODE (e.g., 'NIQ').
176
+ start_date: Earliest DATE_RECEIVED to include (YYYY-MM-DD).
177
+ end_date: Latest DATE_RECEIVED to include (YYYY-MM-DD).
178
+
179
+ Returns:
180
+ DataFrame joining master and device tables.
181
+ """
182
+ conditions, params = [], []
183
+
184
+ if brand_name is not None:
185
+ conditions.append("d.BRAND_NAME ILIKE ?")
186
+ params.append(brand_name)
187
+ if generic_name is not None:
188
+ conditions.append("d.GENERIC_NAME ILIKE ?")
189
+ params.append(generic_name)
190
+ if manufacturer_name is not None:
191
+ conditions.append("d.MANUFACTURER_D_NAME ILIKE ?")
192
+ params.append(manufacturer_name)
193
+ if product_code is not None:
194
+ conditions.append("d.DEVICE_REPORT_PRODUCT_CODE = ?")
195
+ params.append(product_code)
196
+
197
+ if not conditions:
198
+ raise ValueError(
199
+ "At least one device field required: brand_name, generic_name, "
200
+ "manufacturer_name, or product_code."
201
+ )
202
+
203
+ if start_date:
204
+ conditions.append("m.DATE_RECEIVED >= ?::DATE")
205
+ params.append(start_date)
206
+ if end_date:
207
+ conditions.append("m.DATE_RECEIVED <= ?::DATE")
208
+ params.append(end_date)
209
+
210
+ where = " AND ".join(conditions)
211
+ sql = f"""
212
+ SELECT m.*, d.* EXCLUDE (MDR_REPORT_KEY)
213
+ FROM device d
214
+ JOIN master m USING (MDR_REPORT_KEY)
215
+ WHERE {where}
216
+ """
217
+ return self.conn.execute(sql, params).df()
218
+
219
+ def search_by_device_names(self, criteria, start_date=None, end_date=None,
220
+ group_column='search_group'):
221
+ """
222
+ Substring search across DEVICE_NAME_CONCAT (BRAND_NAME|GENERIC_NAME|MANUFACTURER_D_NAME).
223
+
224
+ Criteria formats:
225
+ 'term' — any record where any name field contains 'term'
226
+ ['a', 'b'] — contains 'a' OR 'b'
227
+ [['a', 'b'], 'c'] — (contains 'a' AND 'b') OR (contains 'c')
228
+ {'group1': [...], ...} — grouped search; result includes search_group column.
229
+ Use None as criteria to skip a group.
230
+
231
+ All matching is case-insensitive substring matching.
232
+
233
+ Args:
234
+ criteria: Search terms in one of the formats above.
235
+ start_date: Earliest DATE_RECEIVED to include (YYYY-MM-DD).
236
+ end_date: Latest DATE_RECEIVED to include (YYYY-MM-DD).
237
+ group_column: Column name for group label when using dict criteria.
238
+
239
+ Returns:
240
+ DataFrame joining master and device tables (plus group_column if dict input).
241
+ """
242
+ if isinstance(criteria, dict):
243
+ parts = []
244
+ for group_name, group_criteria in criteria.items():
245
+ if group_criteria is None:
246
+ continue
247
+ df = self.search_by_device_names(group_criteria, start_date, end_date)
248
+ df[group_column] = group_name
249
+ parts.append(df)
250
+ if not parts:
251
+ return pd.DataFrame()
252
+ combined = pd.concat(parts, ignore_index=True)
253
+ # Events matching multiple groups: keep first group assignment (dict order).
254
+ return combined.drop_duplicates(subset=['MDR_REPORT_KEY'])
255
+
256
+ normalized = self._normalize_criteria(criteria)
257
+ params = []
258
+ or_groups = []
259
+
260
+ for and_group in normalized:
261
+ and_parts = []
262
+ for term in and_group:
263
+ and_parts.append("d.DEVICE_NAME_CONCAT ILIKE ?")
264
+ params.append(f'%{term}%')
265
+ or_groups.append("(" + " AND ".join(and_parts) + ")")
266
+
267
+ where = "(" + " OR ".join(or_groups) + ")"
268
+
269
+ if start_date:
270
+ where += " AND m.DATE_RECEIVED >= ?::DATE"
271
+ params.append(start_date)
272
+ if end_date:
273
+ where += " AND m.DATE_RECEIVED <= ?::DATE"
274
+ params.append(end_date)
275
+
276
+ sql = f"""
277
+ SELECT m.*, d.* EXCLUDE (MDR_REPORT_KEY)
278
+ FROM device d
279
+ JOIN master m USING (MDR_REPORT_KEY)
280
+ WHERE {where}
281
+ """
282
+ return self.conn.execute(sql, params).df()
283
+
284
+ def get_narratives(self, mdr_report_keys):
285
+ """
286
+ Fetch FOI_TEXT narratives for the given MDR report keys.
287
+
288
+ Args:
289
+ mdr_report_keys: List or Series of MDR_REPORT_KEY values.
290
+
291
+ Returns:
292
+ DataFrame with MDR_REPORT_KEY and FOI_TEXT columns.
293
+ """
294
+ keys = list(mdr_report_keys)
295
+ if not keys:
296
+ return pd.DataFrame(columns=['MDR_REPORT_KEY', 'FOI_TEXT'])
297
+ placeholders = ', '.join(['?'] * len(keys))
298
+ sql = f"SELECT MDR_REPORT_KEY, FOI_TEXT FROM text WHERE MDR_REPORT_KEY IN ({placeholders})"
299
+ return self.conn.execute(sql, keys).df()
300
+
301
+ def get_trends_by_year(self, results_df):
302
+ """
303
+ Count events per year from a results DataFrame.
304
+
305
+ If results_df has a 'search_group' column (from grouped search), counts
306
+ are broken out per group.
307
+
308
+ Args:
309
+ results_df: DataFrame with at least DATE_RECEIVED and MDR_REPORT_KEY columns.
310
+
311
+ Returns:
312
+ DataFrame with year, event_count (and search_group if present).
313
+ """
314
+ if not isinstance(results_df, pd.DataFrame):
315
+ raise TypeError("results_df must be a pandas DataFrame")
316
+ if len(results_df) == 0:
317
+ cols = ['year', 'event_count']
318
+ if 'search_group' in results_df.columns:
319
+ cols.insert(0, 'search_group')
320
+ return pd.DataFrame(columns=cols)
321
+ if 'DATE_RECEIVED' not in results_df.columns:
322
+ raise ValueError("results_df must contain DATE_RECEIVED column")
323
+
324
+ df = results_df.copy()
325
+ df['year'] = pd.to_datetime(df['DATE_RECEIVED'], errors='coerce').dt.year
326
+
327
+ group_cols = ['year']
328
+ if 'search_group' in df.columns:
329
+ group_cols.insert(0, 'search_group')
330
+
331
+ trends = df.groupby(group_cols, as_index=False).size()
332
+ trends.rename(columns={'size': 'event_count'}, inplace=True)
333
+ return trends.sort_values(group_cols)
334
+
335
+ def enrich_with_patient_data(self, results_df):
336
+ """
337
+ Left-join patient outcome data onto a results DataFrame.
338
+
339
+ Args:
340
+ results_df: DataFrame with MDR_REPORT_KEY column.
341
+
342
+ Returns:
343
+ results_df with patient columns appended (suffixed '_patient' on collision).
344
+ """
345
+ keys = results_df['MDR_REPORT_KEY'].tolist()
346
+ if not keys:
347
+ return results_df
348
+ placeholders = ', '.join(['?'] * len(keys))
349
+ patient = self.conn.execute(
350
+ f"SELECT * FROM patient WHERE MDR_REPORT_KEY IN ({placeholders})", keys
351
+ ).df()
352
+ return results_df.merge(patient, on='MDR_REPORT_KEY', how='left',
353
+ suffixes=('', '_patient'))
354
+
355
+ def enrich_with_problems(self, results_df):
356
+ """
357
+ Left-join device problem codes onto a results DataFrame.
358
+
359
+ Args:
360
+ results_df: DataFrame with MDR_REPORT_KEY column.
361
+
362
+ Returns:
363
+ results_df with problem columns appended (suffixed '_problem' on collision).
364
+ """
365
+ keys = results_df['MDR_REPORT_KEY'].tolist()
366
+ if not keys:
367
+ return results_df
368
+ placeholders = ', '.join(['?'] * len(keys))
369
+ problems = self.conn.execute(
370
+ f"SELECT * FROM problem WHERE MDR_REPORT_KEY IN ({placeholders})", keys
371
+ ).df()
372
+ return results_df.merge(problems, on='MDR_REPORT_KEY', how='left',
373
+ suffixes=('', '_problem'))
374
+
375
+ def query(self, sql, params=None):
376
+ """Execute a raw DuckDB SQL query and return a DataFrame."""
377
+ return self.conn.execute(sql, params or []).df()
378
+
379
+ def info(self):
380
+ """Print a summary of loaded tables."""
381
+ print(f"Database : {self.db_path}")
382
+ print(f"Data dir : {self.data_dir}")
383
+ for t in ['master', 'device', 'text', 'patient', 'problem']:
384
+ if not self._table_exists(t):
385
+ print(f" {t:10s}: not loaded")
386
+ continue
387
+ count = self.conn.execute(f"SELECT COUNT(*) FROM {t}").fetchone()[0]
388
+ if t in ('master', 'device'):
389
+ try:
390
+ yr = self.conn.execute(
391
+ f"SELECT MIN(year(DATE_RECEIVED)), MAX(year(DATE_RECEIVED)) FROM {t}"
392
+ ).fetchone()
393
+ print(f" {t:10s}: {count:>10,} rows ({yr[0]}–{yr[1]})")
394
+ continue
395
+ except Exception:
396
+ pass
397
+ print(f" {t:10s}: {count:>10,} rows")
398
+
399
+ def close(self):
400
+ """Close the DuckDB connection."""
401
+ self.conn.close()
402
+
403
+ # ── Public: archiving ───────────────────────────────────────────────────────
404
+
405
+ def archive(self, output_dir, include_raw=False):
406
+ """
407
+ Prepare a citable snapshot of this database (e.g. for Zenodo upload).
408
+
409
+ Checkpoints and copies the DuckDB file, and writes a manifest.json
410
+ recording, per loaded table/year: source file, SHA-256 checksum, row
411
+ count, and load timestamp (from _load_metadata) — plus the DuckDB and
412
+ pymaude versions used to build it, so the snapshot can be reproduced
413
+ or verified later.
414
+
415
+ Args:
416
+ output_dir: Directory to write the archive into (created if missing).
417
+ include_raw: If True, also copy the raw MAUDE source files referenced
418
+ in _load_metadata (from data_dir) into an output_dir/raw/ subfolder.
419
+
420
+ Returns:
421
+ Path to the written manifest.json.
422
+ """
423
+ import pymaude
424
+
425
+ os.makedirs(output_dir, exist_ok=True)
426
+ self.conn.execute("CHECKPOINT")
427
+
428
+ db_filename = os.path.basename(self.db_path)
429
+ db_dest = os.path.join(output_dir, db_filename)
430
+ shutil.copy2(self.db_path, db_dest)
431
+
432
+ rows = self.conn.execute(
433
+ "SELECT table_name, year, source_file, checksum, row_count, loaded_at "
434
+ "FROM _load_metadata ORDER BY table_name, year"
435
+ ).fetchall()
436
+
437
+ tables = [
438
+ {
439
+ 'table': r[0],
440
+ 'year': r[1],
441
+ 'source_file': r[2],
442
+ 'sha256': r[3],
443
+ 'row_count': r[4],
444
+ 'loaded_at': r[5].isoformat(),
445
+ }
446
+ for r in rows
447
+ ]
448
+
449
+ manifest = {
450
+ 'generated_at': datetime.now().isoformat(),
451
+ 'pymaude_version': pymaude.__version__,
452
+ 'duckdb_version': duckdb.__version__,
453
+ 'checksum_algorithm': 'sha256',
454
+ 'database': {
455
+ 'filename': db_filename,
456
+ 'sha256': self._checksum(db_dest),
457
+ 'size_bytes': os.path.getsize(db_dest),
458
+ },
459
+ 'tables': tables,
460
+ }
461
+
462
+ if include_raw:
463
+ raw_dir = os.path.join(output_dir, 'raw')
464
+ os.makedirs(raw_dir, exist_ok=True)
465
+ raw_files = []
466
+ for source_file in sorted({r[2] for r in rows if r[2]}):
467
+ src = os.path.join(self.data_dir, source_file)
468
+ if not os.path.exists(src):
469
+ if self.verbose:
470
+ print(f' Skipping raw file (not found): {source_file}')
471
+ continue
472
+ dest = os.path.join(raw_dir, source_file)
473
+ shutil.copy2(src, dest)
474
+ raw_files.append({'filename': source_file, 'sha256': self._checksum(dest)})
475
+ manifest['raw_files'] = raw_files
476
+
477
+ manifest_path = os.path.join(output_dir, 'manifest.json')
478
+ with open(manifest_path, 'w') as f:
479
+ json.dump(manifest, f, indent=2)
480
+
481
+ if self.verbose:
482
+ print(f'Archive written to {output_dir}')
483
+ print(f' Database : {db_filename} ({manifest["database"]["size_bytes"]:,} bytes)')
484
+ print(f' Tables : {len(tables)} entries')
485
+ if include_raw:
486
+ print(f' Raw files: {len(manifest["raw_files"])}')
487
+
488
+ return manifest_path
489
+
490
+ # ── Private: loading orchestration ───────────────────────────────────────
491
+
492
+ def _process_yearly_table(self, table, years, download, force_download, force_reload):
493
+ for year in years:
494
+ if download:
495
+ self._download_file(table, year, force_download)
496
+ fp = self._make_file_path(table, year)
497
+ if not fp:
498
+ if self.verbose:
499
+ print(f' Skipping {table} {year}: file not found in {self.data_dir}')
500
+ continue
501
+ cksum = self._checksum(fp)
502
+ if not force_reload and self._get_stored_checksum(table, year) == cksum:
503
+ if self.verbose:
504
+ print(f' {table} {year}: up to date, skipping')
505
+ continue
506
+ rows = self._load_yearly(table, year, fp)
507
+ self._record_load(table, year, os.path.basename(fp), cksum, rows)
508
+
509
+ def _process_cumulative_table(self, table, years, meta, download, force_download, force_reload):
510
+ """
511
+ Load a cumulative table (master, patient, problem) from its two source
512
+ files: a historical "thru{N}" file and the small current-year file.
513
+
514
+ There's no way to know which specific rows inside a changed cumulative
515
+ file were actually modified — FDA ships one monolithic file per group,
516
+ not a diff. So rather than trying to scope a delete, any checksum
517
+ change in either file triggers a full delete-and-reload of the whole
518
+ table from both currently available source files.
519
+ """
520
+ current_year = datetime.now().year
521
+ prior_years = sorted(y for y in years if y < current_year)
522
+ curr_years = sorted(y for y in years if y == current_year)
523
+ date_column = meta.get('date_column')
524
+
525
+ groups = []
526
+ if prior_years:
527
+ groups.append((max(prior_years), False))
528
+ if curr_years:
529
+ groups.append((current_year, True))
530
+ if not groups:
531
+ return
532
+
533
+ fetched = []
534
+ for anchor, is_current in groups:
535
+ if download:
536
+ self._download_file(table, anchor, force_download)
537
+ fp = self._make_file_path(table, anchor)
538
+ if not fp:
539
+ if self.verbose:
540
+ label = 'current year' if is_current else f'thru {anchor}'
541
+ print(f' Skipping {table} ({label}): file not found in {self.data_dir}')
542
+ continue
543
+ fetched.append((anchor, is_current, fp, self._checksum(fp)))
544
+
545
+ if len(fetched) != len(groups):
546
+ # Can't safely rebuild the whole table without every expected
547
+ # source file — a partial reload could permanently drop the group
548
+ # we couldn't fetch, since there's no scoped delete to fall back on.
549
+ if self.verbose:
550
+ print(f' Skipping {table}: not all source files available, leaving table untouched')
551
+ return
552
+
553
+ any_changed = force_reload or any(
554
+ self._get_stored_checksum(table, current_year if is_current else anchor) != cksum
555
+ for anchor, is_current, fp, cksum in fetched
556
+ )
557
+ if not any_changed:
558
+ if self.verbose:
559
+ print(f' {table}: up to date, skipping')
560
+ return
561
+
562
+ if self._table_exists(table):
563
+ self.conn.execute(f"DELETE FROM {table}")
564
+ self.conn.execute("DELETE FROM _load_metadata WHERE table_name = ?", [table])
565
+
566
+ for i, (anchor, is_current, fp, cksum) in enumerate(fetched):
567
+ rows = self._load_all(table, fp, date_column=date_column, dedup=(i > 0))
568
+ source_file = os.path.basename(fp)
569
+ if is_current:
570
+ self._record_load(table, current_year, source_file, cksum, rows)
571
+ else:
572
+ if date_column:
573
+ # Record the years this file *actually* contains, rather
574
+ # than assuming it only covers up to the requested anchor —
575
+ # thru-file selection chases whatever's latest-available
576
+ # regardless of the specific year requested, so the real
577
+ # content can cover more than that (see _implied_years).
578
+ covered_rows = self.conn.execute(
579
+ f'SELECT year("{date_column}") AS y, COUNT(*) FROM {table} '
580
+ f"WHERE source_file = '{source_file}' GROUP BY y"
581
+ ).fetchall()
582
+ for y, c in covered_rows:
583
+ if y is not None:
584
+ self._record_load(table, y, source_file, cksum, c)
585
+ else:
586
+ # No date column to verify against (patient, problem) — the
587
+ # same latest-available fetch behavior applies, so assume
588
+ # the same full range _implied_years does.
589
+ for y in range(meta['start_year'], current_year):
590
+ self._record_load(table, y, source_file, cksum, rows)
591
+
592
+ if self.verbose:
593
+ total = self.conn.execute(f"SELECT COUNT(*) FROM {table}").fetchone()[0]
594
+ print(f' {table}: {total:,} total rows')
595
+
596
+ # ── Private: DuckDB loading ───────────────────────────────────────────────
597
+
598
+ # DuckDB CSV read options shared across all load methods.
599
+ # all_varchar=true prevents DuckDB from auto-inferring types, which would
600
+ # cause TRY_STRPTIME to fail (it expects VARCHAR, not auto-detected DATE).
601
+ # Types for key columns are handled explicitly in each SELECT.
602
+ _CSV_OPTS = "sep='|', encoding='latin-1', quote='', ignore_errors=true, all_varchar=true, strict_mode=false"
603
+
604
+ def _parse_date_expr(self, col):
605
+ """Return a SQL expression that parses a VARCHAR date column to DATE."""
606
+ return (
607
+ f"coalesce("
608
+ f"TRY_STRPTIME({col}, '%m/%d/%Y'), "
609
+ f"TRY_STRPTIME({col}, '%Y-%m-%d'), "
610
+ f"TRY_STRPTIME({col}, '%Y/%m/%d'))::DATE"
611
+ )
612
+
613
+ def _load_yearly(self, table, year, filepath):
614
+ """Load one year's file into its table. Returns row count."""
615
+ if self.verbose:
616
+ print(f' Loading {table} {year}...')
617
+
618
+ fp = filepath.replace("'", "''")
619
+ source_file = os.path.basename(filepath).replace("'", "''")
620
+
621
+ if table == 'device':
622
+ # Parse DATE_RECEIVED and add DEVICE_NAME_CONCAT in one CTE pass.
623
+ date_expr = self._parse_date_expr('DATE_RECEIVED')
624
+ select_sql = f"""
625
+ WITH raw AS (
626
+ SELECT * FROM read_csv('{fp}', {self._CSV_OPTS})
627
+ )
628
+ SELECT * REPLACE ({date_expr} AS DATE_RECEIVED),
629
+ upper(coalesce(BRAND_NAME, '') || '|' ||
630
+ coalesce(GENERIC_NAME, '') || '|' ||
631
+ coalesce(MANUFACTURER_D_NAME, '')) AS DEVICE_NAME_CONCAT,
632
+ '{source_file}' AS source_file
633
+ FROM raw
634
+ """
635
+ else:
636
+ # text/problem: no date column that needs parsing.
637
+ select_sql = f"""
638
+ SELECT *, '{source_file}' AS source_file
639
+ FROM read_csv('{fp}', {self._CSV_OPTS})
640
+ """
641
+
642
+ if self._table_exists(table):
643
+ # Scope the delete to exactly this file's prior contribution.
644
+ # Filename <-> year is fixed for life for yearly tables
645
+ # (device2020.zip always means 2020, never anything else), so this
646
+ # is precise and doesn't depend on a per-row date — unlike text,
647
+ # which has no DATE_RECEIVED column at all, or device, where a
648
+ # row's date can fail to parse and never match a year() filter.
649
+ self.conn.execute(
650
+ f"DELETE FROM {table} WHERE source_file = '{source_file}'"
651
+ )
652
+ self._ensure_new_columns(table, filepath)
653
+ self.conn.execute(f"INSERT INTO {table} BY NAME {select_sql}")
654
+ else:
655
+ self.conn.execute(f"CREATE TABLE {table} AS {select_sql}")
656
+
657
+ rows = self.conn.execute(
658
+ f"SELECT COUNT(*) FROM {table} WHERE source_file = '{source_file}'"
659
+ ).fetchone()[0]
660
+
661
+ if self.verbose:
662
+ print(f' {rows:,} rows')
663
+ return rows
664
+
665
+ def _load_all(self, table, filepath, date_column=None, dedup=False):
666
+ """
667
+ Load a cumulative file's full content into table, creating it if it
668
+ doesn't exist yet. Used for master, patient, problem — the caller
669
+ (_process_cumulative_table) is responsible for wiping the table first
670
+ when a fresh load is needed; this just inserts/creates.
671
+
672
+ date_column: if given, that column is parsed from VARCHAR to DATE
673
+ (master has one; patient/problem don't).
674
+ dedup: the current-year file's rows can overlap with what the thru-file
675
+ already loaded — pass True (for every group after the first in a given
676
+ table-processing pass) to only insert rows not already present.
677
+ Compared on the data columns only (EXCLUDE source_file), since a row
678
+ that's identical except for which file it came from should still count
679
+ as a duplicate; source_file is attached only after dedup so it doesn't
680
+ interfere with the comparison.
681
+ """
682
+ if self.verbose:
683
+ print(f' Loading {table} ({os.path.basename(filepath)})...')
684
+
685
+ fp = filepath.replace("'", "''")
686
+ source_file = os.path.basename(filepath).replace("'", "''")
687
+
688
+ if table == 'problem':
689
+ # foidevproblem has no header row — name columns explicitly so DuckDB
690
+ # doesn't fall back to column0/column1/column2 naming.
691
+ # The file has shipped with 2 or 3 columns depending on the release year.
692
+ _opts = (f"sep='|', encoding='latin-1', quote='', ignore_errors=true, "
693
+ f"all_varchar=true, strict_mode=false, header=false")
694
+ with open(filepath, 'r', encoding='latin1') as _f:
695
+ _ncols = len(_f.readline().split('|'))
696
+ if _ncols >= 3:
697
+ _col_select = ("column0 AS MDR_REPORT_KEY, "
698
+ "column1 AS DEVICE_PROBLEM_CODE, "
699
+ "column2 AS DATE_ADDED_FLAG")
700
+ else:
701
+ _col_select = ("column0 AS MDR_REPORT_KEY, "
702
+ "column1 AS DEVICE_PROBLEM_CODE")
703
+ raw_select_sql = f"""
704
+ SELECT {_col_select}
705
+ FROM read_csv('{fp}', {_opts})
706
+ """
707
+ if self._table_exists(table):
708
+ # Migrate column names if db was created before explicit-naming fix.
709
+ existing = {r[0] for r in self.conn.execute("DESCRIBE problem").fetchall()}
710
+ if 'column0' in existing and 'MDR_REPORT_KEY' not in existing:
711
+ for i, name in enumerate(['MDR_REPORT_KEY', 'DEVICE_PROBLEM_CODE', 'DATE_ADDED_FLAG']):
712
+ if f'column{i}' in existing:
713
+ self.conn.execute(f'ALTER TABLE problem RENAME COLUMN "column{i}" TO "{name}"')
714
+ else:
715
+ if date_column:
716
+ date_expr = self._parse_date_expr(f'"{date_column}"')
717
+ raw_select_sql = f"""
718
+ SELECT * REPLACE ({date_expr} AS "{date_column}")
719
+ FROM read_csv('{fp}', {self._CSV_OPTS})
720
+ """
721
+ else:
722
+ raw_select_sql = f"""
723
+ SELECT * FROM read_csv('{fp}', {self._CSV_OPTS})
724
+ """
725
+ if self._table_exists(table):
726
+ self._ensure_new_columns(table, filepath)
727
+
728
+ if self._table_exists(table):
729
+ if not dedup:
730
+ self.conn.execute(f"""
731
+ INSERT INTO {table} BY NAME
732
+ SELECT *, '{source_file}' AS source_file FROM ({raw_select_sql})
733
+ """)
734
+ else:
735
+ self.conn.execute(f"""
736
+ WITH truly_new AS (
737
+ SELECT * FROM ({raw_select_sql})
738
+ EXCEPT
739
+ SELECT * EXCLUDE (source_file) FROM {table}
740
+ )
741
+ INSERT INTO {table} BY NAME
742
+ SELECT *, '{source_file}' AS source_file FROM truly_new
743
+ """)
744
+ else:
745
+ self.conn.execute(f"""
746
+ CREATE TABLE {table} AS
747
+ SELECT *, '{source_file}' AS source_file FROM ({raw_select_sql})
748
+ """)
749
+
750
+ rows = self.conn.execute(
751
+ f"SELECT COUNT(*) FROM {table} WHERE source_file = '{source_file}'"
752
+ ).fetchone()[0]
753
+ if self.verbose:
754
+ print(f' {rows:,} rows')
755
+ return rows
756
+
757
+ def _ensure_new_columns(self, table, filepath):
758
+ """
759
+ Add any columns present in the CSV that are missing from the DuckDB table.
760
+ Handles MAUDE's schema variations across years (e.g., new fields added in later files).
761
+ """
762
+ with open(filepath, 'r', encoding='latin1') as f:
763
+ csv_cols = set(f.readline().strip().split('|'))
764
+
765
+ existing = {row[0] for row in self.conn.execute(f"DESCRIBE {table}").fetchall()}
766
+
767
+ for col in csv_cols:
768
+ col = col.strip()
769
+ if col and col not in existing:
770
+ try:
771
+ self.conn.execute(f'ALTER TABLE "{table}" ADD COLUMN "{col}" VARCHAR')
772
+ except Exception:
773
+ pass
774
+
775
+ def _create_indexes(self):
776
+ """Create indexes on join keys. DuckDB ART indexes help point lookups and joins."""
777
+ for table, col in [
778
+ ('master', 'MDR_REPORT_KEY'),
779
+ ('master', 'DATE_RECEIVED'),
780
+ ('device', 'MDR_REPORT_KEY'),
781
+ ('device', 'DEVICE_REPORT_PRODUCT_CODE'),
782
+ ('text', 'MDR_REPORT_KEY'),
783
+ ('patient', 'MDR_REPORT_KEY'),
784
+ ('problem', 'MDR_REPORT_KEY'),
785
+ ]:
786
+ if self._table_exists(table):
787
+ idx = f"idx_{table}_{col.lower()}"
788
+ try:
789
+ self.conn.execute(
790
+ f'CREATE INDEX IF NOT EXISTS {idx} ON "{table}"("{col}")'
791
+ )
792
+ except Exception:
793
+ pass
794
+
795
+ # ── Private: downloads ────────────────────────────────────────────────────
796
+
797
+ def _download_file(self, table, year, force_download=False):
798
+ """Download and extract a MAUDE zip from the FDA FTP area."""
799
+ url, filename = self._construct_url(table, year)
800
+ if not url:
801
+ return False
802
+
803
+ cache_key = (table, filename)
804
+ if not force_download and cache_key in self._download_cache:
805
+ return True
806
+
807
+ zip_path = os.path.join(self.data_dir, filename)
808
+
809
+ if not force_download and os.path.exists(zip_path):
810
+ if self.verbose:
811
+ print(f' Using cached {filename}')
812
+ try:
813
+ with zipfile.ZipFile(zip_path, 'r') as z:
814
+ z.extractall(self.data_dir)
815
+ self._download_cache.add(cache_key)
816
+ return True
817
+ except Exception:
818
+ os.remove(zip_path)
819
+
820
+ try:
821
+ if self.verbose:
822
+ print(f' Downloading {filename}...')
823
+ headers = {'User-Agent': 'Mozilla/5.0'}
824
+ r = requests.get(url, headers=headers, timeout=60)
825
+ r.raise_for_status()
826
+ with open(zip_path, 'wb') as f:
827
+ f.write(r.content)
828
+ with zipfile.ZipFile(zip_path, 'r') as z:
829
+ z.extractall(self.data_dir)
830
+ self._download_cache.add(cache_key)
831
+ return True
832
+ except Exception as e:
833
+ if self.verbose:
834
+ print(f' Error downloading {filename}: {e}')
835
+ return False
836
+
837
+ def _construct_url(self, table, year):
838
+ """
839
+ Return (url, filename) for a given table and year.
840
+
841
+ Yearly tables: device{year}.zip / {prefix}{year}.zip
842
+ Cumulative: {prefix}thru{N}.zip (N = most recent available year)
843
+ Current year: {current_year_prefix}.zip
844
+ """
845
+ if table not in TABLE_METADATA:
846
+ return None, None
847
+
848
+ meta = TABLE_METADATA[table]
849
+ prefix = meta['file_prefix']
850
+ current_year = datetime.now().year
851
+
852
+ if year == current_year:
853
+ filename = f"{meta['current_year_prefix']}.zip"
854
+ return f"{FDA_BASE_URL}/{filename}", filename
855
+
856
+ if meta['pattern_type'] == 'yearly':
857
+ # Device table uses 'device{year}.zip', others use '{prefix}{year}.zip'.
858
+ filename = f"device{year}.zip" if table == 'device' else f"{prefix}{year}.zip"
859
+ return f"{FDA_BASE_URL}/{filename}", filename
860
+
861
+ # Cumulative: FDA releases thru{prev_year} files; probe for the latest available.
862
+ sep = meta.get('thru_separator', '')
863
+ for offset in [1, 2, 3]:
864
+ thru_year = current_year - offset
865
+ filename = f"{prefix}{sep}thru{thru_year}.zip"
866
+ url = f"{FDA_BASE_URL}/{filename}"
867
+ if self._url_exists(url):
868
+ if offset > 1 and self.verbose:
869
+ print(f' Note: using {filename} (expected {prefix}{sep}thru{current_year - 1} not available)')
870
+ return url, filename
871
+
872
+ filename = f"{prefix}{sep}thru{current_year - 1}.zip"
873
+ return f"{FDA_BASE_URL}/{filename}", filename
874
+
875
+ def _url_exists(self, url):
876
+ try:
877
+ r = requests.head(url, headers={'User-Agent': 'Mozilla/5.0'},
878
+ timeout=5, allow_redirects=True)
879
+ return 200 <= r.status_code < 300
880
+ except Exception:
881
+ return False
882
+
883
+ def _make_file_path(self, table, year):
884
+ """
885
+ Find the extracted .txt file for a table/year in data_dir.
886
+ Returns full path or None if not found.
887
+ """
888
+ if table not in TABLE_METADATA:
889
+ return None
890
+
891
+ meta = TABLE_METADATA[table]
892
+ prefix = meta['file_prefix']
893
+ current_year = datetime.now().year
894
+
895
+ try:
896
+ files = set(os.listdir(self.data_dir))
897
+ except FileNotFoundError:
898
+ return None
899
+
900
+ candidates = []
901
+
902
+ if year == current_year:
903
+ cp = meta['current_year_prefix']
904
+ candidates += [f"{cp}.txt", f"{cp.upper()}.txt"]
905
+
906
+ if meta['pattern_type'] == 'yearly':
907
+ if table == 'device':
908
+ candidates += [f"device{year}.txt", f"DEVICE{year}.txt"]
909
+ else:
910
+ candidates += [f"{prefix}{year}.txt", f"{prefix.upper()}{year}.txt"]
911
+ elif meta['pattern_type'] == 'cumulative':
912
+ # Cumulative: check for thru files from most recent backwards.
913
+ sep = meta.get('thru_separator', '')
914
+ for offset in [1, 2, 3]:
915
+ thru_year = current_year - offset
916
+ candidates += [
917
+ f"{prefix}{sep}thru{thru_year}.txt",
918
+ f"{prefix.upper()}{sep}thru{thru_year}.txt",
919
+ ]
920
+ # Fallback: any file matching the cumulative pattern.
921
+ for fn in sorted(files):
922
+ if (fn.lower().startswith(prefix.lower())
923
+ and 'thru' in fn.lower()
924
+ and fn.endswith('.txt')):
925
+ candidates.append(fn)
926
+
927
+ for c in candidates:
928
+ if c in files:
929
+ return os.path.join(self.data_dir, c)
930
+ return None
931
+
932
+ # ── Private: checksum / metadata ─────────────────────────────────────────
933
+
934
+ def _init_metadata_table(self):
935
+ self.conn.execute("""
936
+ CREATE TABLE IF NOT EXISTS _load_metadata (
937
+ table_name VARCHAR,
938
+ year INTEGER,
939
+ source_file VARCHAR,
940
+ checksum VARCHAR,
941
+ row_count INTEGER,
942
+ loaded_at TIMESTAMP
943
+ )
944
+ """)
945
+
946
+ def _get_stored_checksum(self, table, year):
947
+ row = self.conn.execute(
948
+ "SELECT checksum FROM _load_metadata WHERE table_name = ? AND year = ?",
949
+ [table, year]
950
+ ).fetchone()
951
+ return row[0] if row else None
952
+
953
+ def _record_load(self, table, year, source_file, checksum, rows):
954
+ self.conn.execute(
955
+ "DELETE FROM _load_metadata WHERE table_name = ? AND year = ?",
956
+ [table, year]
957
+ )
958
+ self.conn.execute(
959
+ "INSERT INTO _load_metadata VALUES (?, ?, ?, ?, ?, current_timestamp)",
960
+ [table, year, source_file, checksum, rows]
961
+ )
962
+
963
+ def _checksum(self, filepath):
964
+ h = hashlib.sha256()
965
+ with open(filepath, 'rb') as f:
966
+ for chunk in iter(lambda: f.read(65536), b''):
967
+ h.update(chunk)
968
+ return h.hexdigest()
969
+
970
+ def _table_exists(self, table):
971
+ return self.conn.execute(
972
+ "SELECT COUNT(*) FROM information_schema.tables "
973
+ "WHERE table_name = ? AND table_schema = 'main'",
974
+ [table]
975
+ ).fetchone()[0] > 0
976
+
977
+ def _get_years_in_db(self):
978
+ try:
979
+ rows = self.conn.execute(
980
+ "SELECT DISTINCT year FROM _load_metadata"
981
+ ).fetchall()
982
+ return sorted(r[0] for r in rows)
983
+ except Exception:
984
+ return []
985
+
986
+ def _get_loaded_tables(self):
987
+ try:
988
+ rows = self.conn.execute(
989
+ "SELECT DISTINCT table_name FROM _load_metadata"
990
+ ).fetchall()
991
+ return [r[0] for r in rows]
992
+ except Exception:
993
+ return []
994
+
995
+ # ── Private: year parsing / validation ───────────────────────────────────
996
+
997
+ def _parse_year_range(self, years):
998
+ if isinstance(years, int):
999
+ return [years]
1000
+ if isinstance(years, list):
1001
+ return years
1002
+ s = str(years)
1003
+ if s == 'all':
1004
+ return list(range(1991, datetime.now().year + 1))
1005
+ if s == 'latest':
1006
+ return [datetime.now().year - 1]
1007
+ if s == 'current':
1008
+ return [datetime.now().year]
1009
+ if '-' in s:
1010
+ a, b = s.split('-', 1)
1011
+ return list(range(int(a), int(b) + 1))
1012
+ return [int(s)]
1013
+
1014
+ def _validate(self, years, tables):
1015
+ """Return list of (year, table) pairs that are valid to load."""
1016
+ valid = []
1017
+ current_year = datetime.now().year
1018
+ for table in tables:
1019
+ if table not in TABLE_METADATA:
1020
+ if self.verbose:
1021
+ print(f" Unknown table '{table}', skipping")
1022
+ continue
1023
+ start_year = TABLE_METADATA[table]['start_year']
1024
+ for year in years:
1025
+ if year < start_year:
1026
+ if self.verbose:
1027
+ print(f" Skipping {table} {year}: available from {start_year} onwards")
1028
+ continue
1029
+ if year > current_year:
1030
+ if self.verbose:
1031
+ print(f" Skipping {table} {year}: future year")
1032
+ continue
1033
+ valid.append((year, table))
1034
+ return valid
1035
+
1036
+ def _covered_years(self, table):
1037
+ """Years currently recorded as loaded for `table`, from _load_metadata."""
1038
+ rows = self.conn.execute(
1039
+ "SELECT DISTINCT year FROM _load_metadata WHERE table_name = ?", [table]
1040
+ ).fetchall()
1041
+ return {r[0] for r in rows}
1042
+
1043
+ def _implied_years(self, table, years_for_table, meta):
1044
+ """
1045
+ Years `table` would end up covering after loading `years_for_table`.
1046
+
1047
+ Yearly tables (device, text): each year is an independent file, so this
1048
+ is just the requested years themselves.
1049
+
1050
+ Cumulative tables (master, patient, problem): the "thru{N}" file fetch
1051
+ (_construct_url/_make_file_path) always resolves to whichever historical
1052
+ dump is actually latest-available, essentially ignoring the specific
1053
+ prior year requested — so requesting any prior year is treated as
1054
+ implying coverage through last year, not just up to the requested year.
1055
+ The current year, if requested, comes from a separate, current-year-only
1056
+ file and only implies itself.
1057
+ """
1058
+ if meta['pattern_type'] == 'yearly':
1059
+ return set(years_for_table)
1060
+
1061
+ current_year = datetime.now().year
1062
+ prior = [y for y in years_for_table if y < current_year]
1063
+ curr = [y for y in years_for_table if y == current_year]
1064
+ implied = set()
1065
+ if prior:
1066
+ implied |= set(range(meta['start_year'], current_year))
1067
+ if curr:
1068
+ implied.add(current_year)
1069
+ return implied
1070
+
1071
+ # ── Private: search helpers ───────────────────────────────────────────────
1072
+
1073
+ def _normalize_criteria(self, criteria):
1074
+ """
1075
+ Normalize criteria to list-of-lists (each inner list = AND group, outer = OR).
1076
+
1077
+ 'term' → [['term']]
1078
+ ['a', 'b'] → [['a'], ['b']]
1079
+ [['a', 'b'], 'c'] → [['a', 'b'], ['c']]
1080
+ """
1081
+ if isinstance(criteria, str):
1082
+ return [[criteria]]
1083
+ if not isinstance(criteria, list):
1084
+ raise ValueError("criteria must be a string or list")
1085
+ result = []
1086
+ for item in criteria:
1087
+ if isinstance(item, str):
1088
+ result.append([item])
1089
+ elif isinstance(item, list):
1090
+ if not item:
1091
+ raise ValueError("Empty AND group in criteria")
1092
+ for t in item:
1093
+ if not isinstance(t, str):
1094
+ raise ValueError(f"All search terms must be strings, got {type(t).__name__}")
1095
+ result.append(item)
1096
+ else:
1097
+ raise ValueError(f"Criteria items must be str or list, got {type(item).__name__}")
1098
+ if not result:
1099
+ raise ValueError("criteria cannot be empty")
1100
+ return result
pymaude/metadata.py ADDED
@@ -0,0 +1,54 @@
1
+ # metadata.py - MAUDE table configuration
2
+ # Copyright (C) 2026 Jacob Schwartz <jaschwa@umich.edu>
3
+ # MIT License
4
+
5
+ FDA_BASE_URL = "https://www.accessdata.fda.gov/MAUDE/ftparea"
6
+ FDA_PREMARKET_URL = "https://www.accessdata.fda.gov/premarket/ftparea"
7
+
8
+ # Table metadata: file patterns, availability, and structure
9
+ TABLE_METADATA = {
10
+ 'master': {
11
+ 'file_prefix': 'mdrfoi',
12
+ 'pattern_type': 'cumulative', # mdrfoithru{year}.zip
13
+ 'current_year_prefix': 'mdrfoi', # mdrfoi.zip for current year
14
+ 'start_year': 2000,
15
+ 'date_column': 'DATE_RECEIVED',
16
+ 'description': 'Master records (adverse event reports)',
17
+ },
18
+ 'device': {
19
+ 'file_prefix': 'foidev',
20
+ 'pattern_type': 'yearly', # device{year}.zip (special naming)
21
+ 'current_year_prefix': 'device', # device.zip for current year
22
+ 'start_year': 2000,
23
+ 'date_column': 'DATE_RECEIVED',
24
+ 'description': 'Device information',
25
+ },
26
+ 'text': {
27
+ 'file_prefix': 'foitext',
28
+ 'pattern_type': 'yearly', # foitext{year}.zip
29
+ 'current_year_prefix': 'foitext',
30
+ 'start_year': 2000,
31
+ 'description': 'Event narrative text (FOI_TEXT)',
32
+ },
33
+ 'patient': {
34
+ 'file_prefix': 'patient',
35
+ 'pattern_type': 'cumulative', # patientthru{year}.zip
36
+ 'current_year_prefix': 'patient',
37
+ 'start_year': 2000,
38
+ # No date_column: patient table has no date field; joins to master via MDR_REPORT_KEY.
39
+ # The entire cumulative file is loaded (no year filtering possible).
40
+ 'description': 'Patient demographics and outcomes',
41
+ 'size_warning': (
42
+ 'Patient data is distributed as a single large cumulative file. '
43
+ 'All historical data will be downloaded and loaded.'
44
+ ),
45
+ },
46
+ 'problem': {
47
+ 'file_prefix': 'foidevproblem',
48
+ 'pattern_type': 'cumulative', # foidevproblem_thru{year}.zip + foidevproblem.zip
49
+ 'current_year_prefix': 'foidevproblem',
50
+ 'thru_separator': '_', # filename: foidevproblem_thru{year}.zip
51
+ 'start_year': 2000,
52
+ 'description': 'Device problem codes',
53
+ },
54
+ }
@@ -0,0 +1,21 @@
1
+ Metadata-Version: 2.4
2
+ Name: pymaude
3
+ Version: 0.2.0
4
+ Summary: PyMAUDE: FDA MAUDE adverse event database interface
5
+ Author-email: Jacob Schwartz <jaschwa@umich.edu>
6
+ License: MIT
7
+ Classifier: Development Status :: 4 - Beta
8
+ Classifier: Intended Audience :: Science/Research
9
+ Classifier: License :: OSI Approved :: MIT License
10
+ Classifier: Programming Language :: Python :: 3
11
+ Requires-Python: >=3.9
12
+ Description-Content-Type: text/markdown
13
+ License-File: LICENSE
14
+ Requires-Dist: duckdb>=0.10.0
15
+ Requires-Dist: pandas>=1.3.0
16
+ Requires-Dist: requests>=2.25.0
17
+ Requires-Dist: pyyaml>=5.1
18
+ Provides-Extra: dev
19
+ Requires-Dist: pytest>=7.0; extra == "dev"
20
+ Requires-Dist: pytest-cov; extra == "dev"
21
+ Dynamic: license-file
@@ -0,0 +1,8 @@
1
+ pymaude/__init__.py,sha256=t9JOQuhJvE-BMfT8WPjTs5gFlkS2-7s5DofExY6CDsk,366
2
+ pymaude/database.py,sha256=kAkgTsxsRaR6IUOlgjCFM9U99H8pQChXQ02IM7vNqxs,46065
3
+ pymaude/metadata.py,sha256=_MTodSJAeCaIR-lcaf5_hUCx39vm5MVpjZx8hJAcaIc,2204
4
+ pymaude-0.2.0.dist-info/licenses/LICENSE,sha256=U2oyySnfshXoflE-ms876S4mtBbXswAK79Nh7aBWThY,1071
5
+ pymaude-0.2.0.dist-info/METADATA,sha256=uf0hl1adrrC0XUfj3bk861HyjnWcGRpVnAFIiwqlBaY,696
6
+ pymaude-0.2.0.dist-info/WHEEL,sha256=YVMoNqKzERt-wjUZwJ33xBGAwnFl-4cqbYkTtWa4itE,91
7
+ pymaude-0.2.0.dist-info/top_level.txt,sha256=hmUVXXLoByVQeCIALqwnycXMbv0--YfpcT1rI4a60Oc,8
8
+ pymaude-0.2.0.dist-info/RECORD,,
@@ -0,0 +1,5 @@
1
+ Wheel-Version: 1.0
2
+ Generator: setuptools (84.0.0)
3
+ Root-Is-Purelib: true
4
+ Tag: py3-none-any
5
+
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Jacob Schwartz
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1 @@
1
+ pymaude