healthdata 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- healthdata-0.1.0/LICENSE +21 -0
- healthdata-0.1.0/MANIFEST.in +1 -0
- healthdata-0.1.0/PKG-INFO +89 -0
- healthdata-0.1.0/README.md +66 -0
- healthdata-0.1.0/pyproject.toml +41 -0
- healthdata-0.1.0/setup.cfg +4 -0
- healthdata-0.1.0/src/healthdata/__init__.py +10 -0
- healthdata-0.1.0/src/healthdata/nhanes/__init__.py +17 -0
- healthdata-0.1.0/src/healthdata/nhanes/downloader.py +282 -0
- healthdata-0.1.0/src/healthdata/nhanes/reader.py +444 -0
- healthdata-0.1.0/src/healthdata/nhanes/settings.py +10 -0
- healthdata-0.1.0/src/healthdata.egg-info/PKG-INFO +89 -0
- healthdata-0.1.0/src/healthdata.egg-info/SOURCES.txt +17 -0
- healthdata-0.1.0/src/healthdata.egg-info/dependency_links.txt +1 -0
- healthdata-0.1.0/src/healthdata.egg-info/requires.txt +4 -0
- healthdata-0.1.0/src/healthdata.egg-info/top_level.txt +1 -0
- healthdata-0.1.0/tests/test_downloader.py +72 -0
- healthdata-0.1.0/tests/test_public_api.py +10 -0
- healthdata-0.1.0/tests/test_reader.py +87 -0
healthdata-0.1.0/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Ismail Elouafiq
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
exclude tests/test_geo*.py
|
|
@@ -0,0 +1,89 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: healthdata
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Download and read NHANES health survey data.
|
|
5
|
+
Author-email: Ismail Elouafiq <contact@ismail.bio>
|
|
6
|
+
License-Expression: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/nidhog/healthdata
|
|
8
|
+
Project-URL: Repository, https://github.com/nidhog/healthdata
|
|
9
|
+
Project-URL: Documentation, https://github.com/nidhog/healthdata/tree/main/docs/nhanes
|
|
10
|
+
Project-URL: Issues, https://github.com/nidhog/healthdata/issues
|
|
11
|
+
Classifier: Development Status :: 3 - Alpha
|
|
12
|
+
Classifier: Intended Audience :: Science/Research
|
|
13
|
+
Classifier: Programming Language :: Python :: 3.14
|
|
14
|
+
Classifier: Topic :: Scientific/Engineering :: Medical Science Apps.
|
|
15
|
+
Requires-Python: >=3.14
|
|
16
|
+
Description-Content-Type: text/markdown
|
|
17
|
+
License-File: LICENSE
|
|
18
|
+
Requires-Dist: pandas
|
|
19
|
+
Requires-Dist: requests
|
|
20
|
+
Requires-Dist: beautifulsoup4
|
|
21
|
+
Requires-Dist: tqdm
|
|
22
|
+
Dynamic: license-file
|
|
23
|
+
|
|
24
|
+
# Health Data Downloader Package
|
|
25
|
+
The Health Data Downloader is a Python package that makes it easy to download and read real world health data.
|
|
26
|
+
|
|
27
|
+
Current data sources supported:
|
|
28
|
+
* **NHANES** National Health and Nutrition Examination Survey (CDC)
|
|
29
|
+
## The NHANES Submodule
|
|
30
|
+
If you want to download health data from NHANES this module can help you not only download the data but also add documentation from the official NHANES website, and read this data into a pandas dataframe.
|
|
31
|
+
|
|
32
|
+
[NHANES](https://www.cdc.gov/nchs/nhanes/about/) is the National Health and Nutrition Examination Survey which collects data about the health of adults and children in the United States.
|
|
33
|
+
|
|
34
|
+
_Note: This is not an official repository from the NHANES project nor the CDC_
|
|
35
|
+
|
|
36
|
+
This project provides workflows that enable you to:
|
|
37
|
+
- discover and download datasets from NHANES by component (meaning what type of data like Laboratory data, examination etc.) and by cycle (which years)
|
|
38
|
+
- read the XPT files into a pandas DataFrames
|
|
39
|
+
- extract NHANES variable documentation into CSV files and rename variable codes into human-readable names (such as `heart_rate` instead of `LF321`)
|
|
40
|
+
- search and summarize local files if they are already downloaded
|
|
41
|
+
|
|
42
|
+
## Quick start
|
|
43
|
+
In thie quick start we will:
|
|
44
|
+
1. Install required packages.
|
|
45
|
+
2. Run a download.
|
|
46
|
+
3. Read and inspect data.
|
|
47
|
+
|
|
48
|
+
```bash
|
|
49
|
+
pip install healthdata
|
|
50
|
+
```
|
|
51
|
+
You can choose which `components` you want to download (such as Demographics, Laboratory etc.) and which years (can be either "2017-2020" or "2016"). This will also download documentation about these fields from the NHANES website to add them when the data is read.
|
|
52
|
+
If the data is already downloaded in the output directory the module will log this and skip the download.
|
|
53
|
+
```python
|
|
54
|
+
from healthdata.nhanes.downloader import download_nhanes_data
|
|
55
|
+
|
|
56
|
+
download_nhanes_data(
|
|
57
|
+
components=["Demographics", "Examination"],
|
|
58
|
+
output_dir="nhanes_data",
|
|
59
|
+
years=["2017-2018"],
|
|
60
|
+
)
|
|
61
|
+
```
|
|
62
|
+
|
|
63
|
+
You can read the data you already downloaded (you will be prompted to download it if it doesn't exist)
|
|
64
|
+
```python
|
|
65
|
+
from healthdata.nhanes.reader import read_nhanes_data, search_nhanes_local_data
|
|
66
|
+
|
|
67
|
+
df = read_nhanes_data("nhanes_data/2017-2018/2017/Demographics/DEMO_J.XPT")
|
|
68
|
+
print(df.head())
|
|
69
|
+
|
|
70
|
+
inventory = search_nhanes_local_data(search_dir="nhanes_data")
|
|
71
|
+
print(inventory.head())
|
|
72
|
+
```
|
|
73
|
+
That's all folks! Now you will see that your dataframe already includes documentation.
|
|
74
|
+
## Current caveats
|
|
75
|
+
- HTML codebook parsing depends on current NHANES page text patterns. CDC page structure changes may require parser updates.
|
|
76
|
+
- Network errors and partial downloads are not retried automatically.
|
|
77
|
+
|
|
78
|
+
## Upcoming Additions
|
|
79
|
+
Planned additions include NCBI GEO and other open health and biomedical datasets. The current release supports NHANES only.
|
|
80
|
+
|
|
81
|
+
## License
|
|
82
|
+
This project is licensed under the [MIT License](LICENSE). Commercial use is permitted under its terms. Downloaded datasets may be subject to separate terms set by their respective providers.
|
|
83
|
+
|
|
84
|
+
For paid support or consulting, contact the author at contact@ismail.bio.
|
|
85
|
+
|
|
86
|
+
## Author
|
|
87
|
+
[Ismail Elouafiq](https://ismail.bio)
|
|
88
|
+
|
|
89
|
+
Contact: contact@ismail.bio
|
|
@@ -0,0 +1,66 @@
|
|
|
1
|
+
# Health Data Downloader Package
|
|
2
|
+
The Health Data Downloader is a Python package that makes it easy to download and read real world health data.
|
|
3
|
+
|
|
4
|
+
Current data sources supported:
|
|
5
|
+
* **NHANES** National Health and Nutrition Examination Survey (CDC)
|
|
6
|
+
## The NHANES Submodule
|
|
7
|
+
If you want to download health data from NHANES this module can help you not only download the data but also add documentation from the official NHANES website, and read this data into a pandas dataframe.
|
|
8
|
+
|
|
9
|
+
[NHANES](https://www.cdc.gov/nchs/nhanes/about/) is the National Health and Nutrition Examination Survey which collects data about the health of adults and children in the United States.
|
|
10
|
+
|
|
11
|
+
_Note: This is not an official repository from the NHANES project nor the CDC_
|
|
12
|
+
|
|
13
|
+
This project provides workflows that enable you to:
|
|
14
|
+
- discover and download datasets from NHANES by component (meaning what type of data like Laboratory data, examination etc.) and by cycle (which years)
|
|
15
|
+
- read the XPT files into a pandas DataFrames
|
|
16
|
+
- extract NHANES variable documentation into CSV files and rename variable codes into human-readable names (such as `heart_rate` instead of `LF321`)
|
|
17
|
+
- search and summarize local files if they are already downloaded
|
|
18
|
+
|
|
19
|
+
## Quick start
|
|
20
|
+
In thie quick start we will:
|
|
21
|
+
1. Install required packages.
|
|
22
|
+
2. Run a download.
|
|
23
|
+
3. Read and inspect data.
|
|
24
|
+
|
|
25
|
+
```bash
|
|
26
|
+
pip install healthdata
|
|
27
|
+
```
|
|
28
|
+
You can choose which `components` you want to download (such as Demographics, Laboratory etc.) and which years (can be either "2017-2020" or "2016"). This will also download documentation about these fields from the NHANES website to add them when the data is read.
|
|
29
|
+
If the data is already downloaded in the output directory the module will log this and skip the download.
|
|
30
|
+
```python
|
|
31
|
+
from healthdata.nhanes.downloader import download_nhanes_data
|
|
32
|
+
|
|
33
|
+
download_nhanes_data(
|
|
34
|
+
components=["Demographics", "Examination"],
|
|
35
|
+
output_dir="nhanes_data",
|
|
36
|
+
years=["2017-2018"],
|
|
37
|
+
)
|
|
38
|
+
```
|
|
39
|
+
|
|
40
|
+
You can read the data you already downloaded (you will be prompted to download it if it doesn't exist)
|
|
41
|
+
```python
|
|
42
|
+
from healthdata.nhanes.reader import read_nhanes_data, search_nhanes_local_data
|
|
43
|
+
|
|
44
|
+
df = read_nhanes_data("nhanes_data/2017-2018/2017/Demographics/DEMO_J.XPT")
|
|
45
|
+
print(df.head())
|
|
46
|
+
|
|
47
|
+
inventory = search_nhanes_local_data(search_dir="nhanes_data")
|
|
48
|
+
print(inventory.head())
|
|
49
|
+
```
|
|
50
|
+
That's all folks! Now you will see that your dataframe already includes documentation.
|
|
51
|
+
## Current caveats
|
|
52
|
+
- HTML codebook parsing depends on current NHANES page text patterns. CDC page structure changes may require parser updates.
|
|
53
|
+
- Network errors and partial downloads are not retried automatically.
|
|
54
|
+
|
|
55
|
+
## Upcoming Additions
|
|
56
|
+
Planned additions include NCBI GEO and other open health and biomedical datasets. The current release supports NHANES only.
|
|
57
|
+
|
|
58
|
+
## License
|
|
59
|
+
This project is licensed under the [MIT License](LICENSE). Commercial use is permitted under its terms. Downloaded datasets may be subject to separate terms set by their respective providers.
|
|
60
|
+
|
|
61
|
+
For paid support or consulting, contact the author at contact@ismail.bio.
|
|
62
|
+
|
|
63
|
+
## Author
|
|
64
|
+
[Ismail Elouafiq](https://ismail.bio)
|
|
65
|
+
|
|
66
|
+
Contact: contact@ismail.bio
|
|
@@ -0,0 +1,41 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools>=77", "wheel"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "healthdata"
|
|
7
|
+
version = "0.1.0"
|
|
8
|
+
description = "Download and read NHANES health survey data."
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
license = "MIT"
|
|
11
|
+
license-files = ["LICENSE"]
|
|
12
|
+
authors = [
|
|
13
|
+
{ name = "Ismail Elouafiq", email = "contact@ismail.bio" },
|
|
14
|
+
]
|
|
15
|
+
requires-python = ">=3.14"
|
|
16
|
+
dependencies = [
|
|
17
|
+
"pandas",
|
|
18
|
+
"requests",
|
|
19
|
+
"beautifulsoup4",
|
|
20
|
+
"tqdm",
|
|
21
|
+
]
|
|
22
|
+
classifiers = [
|
|
23
|
+
"Development Status :: 3 - Alpha",
|
|
24
|
+
"Intended Audience :: Science/Research",
|
|
25
|
+
"Programming Language :: Python :: 3.14",
|
|
26
|
+
"Topic :: Scientific/Engineering :: Medical Science Apps.",
|
|
27
|
+
]
|
|
28
|
+
|
|
29
|
+
[project.urls]
|
|
30
|
+
Homepage = "https://github.com/nidhog/healthdata"
|
|
31
|
+
Repository = "https://github.com/nidhog/healthdata"
|
|
32
|
+
Documentation = "https://github.com/nidhog/healthdata/tree/main/docs/nhanes"
|
|
33
|
+
Issues = "https://github.com/nidhog/healthdata/issues"
|
|
34
|
+
|
|
35
|
+
[tool.setuptools]
|
|
36
|
+
package-dir = { "" = "src" }
|
|
37
|
+
packages = ["healthdata", "healthdata.nhanes"]
|
|
38
|
+
|
|
39
|
+
[tool.setuptools.package-data]
|
|
40
|
+
healthdata = []
|
|
41
|
+
"healthdata.nhanes" = []
|
|
@@ -0,0 +1,10 @@
|
|
|
1
|
+
"""healthdata package
|
|
2
|
+
|
|
3
|
+
Provides tools to: download, read and preprocess health datasets.
|
|
4
|
+
It is specific to open health datasets.
|
|
5
|
+
Dataset implementations are under separate modules, such as ``healthdata.nhanes``
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from . import nhanes
|
|
9
|
+
|
|
10
|
+
__all__ = ["nhanes"]
|
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
"""NHANES submodule for the healthdata package."""
|
|
2
|
+
|
|
3
|
+
from .downloader import (
|
|
4
|
+
download_and_extract_docs,
|
|
5
|
+
download_nhanes_data,
|
|
6
|
+
get_file_links_for_component,
|
|
7
|
+
)
|
|
8
|
+
from .reader import get_column_doc, read_nhanes_data, search_nhanes_local_data
|
|
9
|
+
|
|
10
|
+
__all__ = [
|
|
11
|
+
"download_nhanes_data",
|
|
12
|
+
"download_and_extract_docs",
|
|
13
|
+
"get_file_links_for_component",
|
|
14
|
+
"read_nhanes_data",
|
|
15
|
+
"get_column_doc",
|
|
16
|
+
"search_nhanes_local_data",
|
|
17
|
+
]
|
|
@@ -0,0 +1,282 @@
|
|
|
1
|
+
import csv
|
|
2
|
+
import logging
|
|
3
|
+
import os
|
|
4
|
+
import re
|
|
5
|
+
import json
|
|
6
|
+
from typing import List, Optional, Union
|
|
7
|
+
from urllib.parse import urlparse
|
|
8
|
+
|
|
9
|
+
import requests
|
|
10
|
+
from bs4 import BeautifulSoup
|
|
11
|
+
from tqdm import tqdm
|
|
12
|
+
|
|
13
|
+
from .settings import DEFAULT_BASE_DOMAIN, DEFAULT_BASE_SEARCH_URL, DEFAULT_COMPONENTS
|
|
14
|
+
|
|
15
|
+
logger = logging.getLogger(__name__)
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def ensure_dir(path: str):
|
|
19
|
+
if not os.path.exists(path):
|
|
20
|
+
os.makedirs(path)
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def extract_date_from_url(url: str) -> str:
|
|
24
|
+
"""
|
|
25
|
+
Extract NHANES cycle or year from url patterns.
|
|
26
|
+
The patterns we look for are either "YYYY-YYYY" or "/YYYY/".
|
|
27
|
+
Returns 'nodate' if none found.
|
|
28
|
+
"""
|
|
29
|
+
match = re.search(r"\d{4}-\d{4}", url)
|
|
30
|
+
if match:
|
|
31
|
+
return match.group(0)
|
|
32
|
+
match = re.search(r"/(19|20)\d{2}/", url)
|
|
33
|
+
if match:
|
|
34
|
+
return match.group(0).strip("/")
|
|
35
|
+
return "nodate"
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def download_file(url: str, out_path: str):
|
|
39
|
+
"""Download an XPT file"""
|
|
40
|
+
r = requests.get(url, stream=True)
|
|
41
|
+
r.raise_for_status()
|
|
42
|
+
total = int(r.headers.get("content-length", 0))
|
|
43
|
+
|
|
44
|
+
with open(out_path, "wb") as f, tqdm(
|
|
45
|
+
desc=os.path.basename(out_path),
|
|
46
|
+
total=total if total > 0 else None,
|
|
47
|
+
unit="B",
|
|
48
|
+
unit_scale=True,
|
|
49
|
+
) as bar:
|
|
50
|
+
for chunk in r.iter_content(chunk_size=8192):
|
|
51
|
+
if not chunk:
|
|
52
|
+
continue
|
|
53
|
+
f.write(chunk)
|
|
54
|
+
if total > 0:
|
|
55
|
+
bar.update(len(chunk))
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def parse_nhanes_doc_variables(doc_html: str, doc_url: str):
|
|
59
|
+
"""
|
|
60
|
+
Parse NHANES doc HTML into: variable_name, label, description, value_meanings, link.
|
|
61
|
+
|
|
62
|
+
This extracts metadata that is not available in the XPT files.
|
|
63
|
+
|
|
64
|
+
TODO: add patterns as config instead?
|
|
65
|
+
"""
|
|
66
|
+
soup = BeautifulSoup(doc_html, "html.parser")
|
|
67
|
+
text = soup.get_text("\n")
|
|
68
|
+
text = re.sub(r"\r", "", text)
|
|
69
|
+
text = re.sub(r"[ \t]+", " ", text)
|
|
70
|
+
|
|
71
|
+
rows = []
|
|
72
|
+
chunks = re.split(r"(?=\bVariable Name:\s*[A-Za-z0-9_]+)", text)
|
|
73
|
+
for chunk in chunks:
|
|
74
|
+
chunk = chunk.strip()
|
|
75
|
+
if not chunk.startswith("Variable Name:"):
|
|
76
|
+
continue
|
|
77
|
+
|
|
78
|
+
var_match = re.search(r"Variable Name:\s*(?P<var>[A-Za-z0-9_]+)", chunk)
|
|
79
|
+
if not var_match:
|
|
80
|
+
continue
|
|
81
|
+
var = var_match.group("var").strip()
|
|
82
|
+
|
|
83
|
+
label_match = re.search(
|
|
84
|
+
r"SAS Label:\s*(?P<label>.*?)(?=\n\s*English Text:|\n\s*English Instructions:|\n\s*Target:|\n\s*Code or Value|\Z)",
|
|
85
|
+
chunk,
|
|
86
|
+
re.DOTALL,
|
|
87
|
+
)
|
|
88
|
+
desc_match = re.search(
|
|
89
|
+
r"English Text:\s*(?P<desc>.*?)(?=\n\s*English Instructions:|\n\s*Target:|\n\s*Code or Value|\Z)",
|
|
90
|
+
chunk,
|
|
91
|
+
re.DOTALL,
|
|
92
|
+
)
|
|
93
|
+
|
|
94
|
+
label = re.sub(r"\s+", " ", label_match.group("label")).strip() if label_match else ""
|
|
95
|
+
desc = re.sub(r"\s+", " ", desc_match.group("desc")).strip() if desc_match else ""
|
|
96
|
+
|
|
97
|
+
value_meanings = []
|
|
98
|
+
value_section = re.search(
|
|
99
|
+
r"Code or Value\s*(?P<section>.*?)(?=\n\s*Variable Name:|\Z)",
|
|
100
|
+
chunk,
|
|
101
|
+
re.DOTALL,
|
|
102
|
+
)
|
|
103
|
+
if value_section:
|
|
104
|
+
section_text = value_section.group("section")
|
|
105
|
+
# Common NHANES rows are represented as: code whitespace value-description.
|
|
106
|
+
for line in [ln.strip() for ln in section_text.splitlines() if ln.strip()]:
|
|
107
|
+
if line.lower() in {
|
|
108
|
+
"value description",
|
|
109
|
+
"unweighted count",
|
|
110
|
+
"weighted percent",
|
|
111
|
+
"weighted frequency",
|
|
112
|
+
"cumulative",
|
|
113
|
+
"skip to item",
|
|
114
|
+
}:
|
|
115
|
+
continue
|
|
116
|
+
m = re.match(r"^(?P<code>[-+A-Za-z0-9\.]+)\s{1,}(?P<meaning>.+)$", line)
|
|
117
|
+
if not m:
|
|
118
|
+
continue
|
|
119
|
+
value_meanings.append(
|
|
120
|
+
{
|
|
121
|
+
"code": m.group("code").strip(),
|
|
122
|
+
"meaning": re.sub(r"\s+", " ", m.group("meaning")).strip(),
|
|
123
|
+
}
|
|
124
|
+
)
|
|
125
|
+
|
|
126
|
+
rows.append(
|
|
127
|
+
{
|
|
128
|
+
"variable_name": var,
|
|
129
|
+
"label": label,
|
|
130
|
+
"description": desc,
|
|
131
|
+
"value_meanings": json.dumps(value_meanings, ensure_ascii=False) if value_meanings else "",
|
|
132
|
+
"link": doc_url,
|
|
133
|
+
}
|
|
134
|
+
)
|
|
135
|
+
return rows
|
|
136
|
+
|
|
137
|
+
|
|
138
|
+
def download_and_extract_docs(doc_url: str, out_dir: str):
|
|
139
|
+
r = requests.get(doc_url)
|
|
140
|
+
r.raise_for_status()
|
|
141
|
+
rows = parse_nhanes_doc_variables(r.text, doc_url)
|
|
142
|
+
|
|
143
|
+
doc_base = os.path.basename(urlparse(doc_url).path)
|
|
144
|
+
csv_name = os.path.splitext(doc_base)[0] + "_variables.csv"
|
|
145
|
+
out_csv = os.path.join(out_dir, csv_name)
|
|
146
|
+
|
|
147
|
+
|
|
148
|
+
ensure_dir(os.path.dirname(out_csv))
|
|
149
|
+
with open(out_csv, "w", newline="", encoding="utf-8") as f:
|
|
150
|
+
w = csv.DictWriter(
|
|
151
|
+
f,
|
|
152
|
+
fieldnames=["variable_name", "label", "description", "value_meanings", "link"],
|
|
153
|
+
)
|
|
154
|
+
w.writeheader()
|
|
155
|
+
w.writerows(rows)
|
|
156
|
+
print(f"> [SAVED] Docs CSV saved: {out_csv}")
|
|
157
|
+
|
|
158
|
+
|
|
159
|
+
def get_file_links_for_component(
|
|
160
|
+
component: str,
|
|
161
|
+
years: Optional[List[Union[int, str]]] = None,
|
|
162
|
+
base_search_url: str = DEFAULT_BASE_SEARCH_URL,
|
|
163
|
+
base_domain: str = DEFAULT_BASE_DOMAIN,
|
|
164
|
+
):
|
|
165
|
+
"""Return dataset links for an NHANES component, filtered by years if provided.
|
|
166
|
+
This is where most of the logic for parsing the website is defined.
|
|
167
|
+
TODO: perhaps this can be provided as a function to the downloader
|
|
168
|
+
in case the website changes its structure in the future?
|
|
169
|
+
"""
|
|
170
|
+
params = {"Component": component}
|
|
171
|
+
print(f"> [FETCH] Fetching data page for Component = {component} ...")
|
|
172
|
+
resp = requests.get(base_search_url, params=params)
|
|
173
|
+
resp.raise_for_status()
|
|
174
|
+
|
|
175
|
+
soup = BeautifulSoup(resp.text, "html.parser")
|
|
176
|
+
rows = soup.select("table tr")
|
|
177
|
+
out = []
|
|
178
|
+
|
|
179
|
+
years_set = set()
|
|
180
|
+
if years:
|
|
181
|
+
for y in years:
|
|
182
|
+
years_set.add(str(y))
|
|
183
|
+
|
|
184
|
+
for tr_field in rows:
|
|
185
|
+
td_fields = tr_field.find_all("td")
|
|
186
|
+
if not td_fields:
|
|
187
|
+
continue
|
|
188
|
+
|
|
189
|
+
cycle_str = td_fields[0].get_text(strip=True) if len(td_fields) >= 1 else "unknown"
|
|
190
|
+
|
|
191
|
+
if years:
|
|
192
|
+
cycle_parts = re.findall(r"\d{4}", cycle_str)
|
|
193
|
+
is_match = False
|
|
194
|
+
for part in cycle_parts:
|
|
195
|
+
if part in years_set:
|
|
196
|
+
is_match = True
|
|
197
|
+
break
|
|
198
|
+
if cycle_str in years_set:
|
|
199
|
+
is_match = True
|
|
200
|
+
|
|
201
|
+
if not is_match:
|
|
202
|
+
continue
|
|
203
|
+
|
|
204
|
+
doc_url = None
|
|
205
|
+
xpt_url = None
|
|
206
|
+
data_file_name = None
|
|
207
|
+
|
|
208
|
+
for a in tr_field.select("a[href]"):
|
|
209
|
+
href = a.get("href", "").strip()
|
|
210
|
+
if not href:
|
|
211
|
+
continue
|
|
212
|
+
|
|
213
|
+
if href.startswith("http"):
|
|
214
|
+
full_url = href
|
|
215
|
+
elif href.startswith("/"):
|
|
216
|
+
full_url = base_domain + href
|
|
217
|
+
else:
|
|
218
|
+
full_url = base_domain + "/" + href.lstrip("/")
|
|
219
|
+
|
|
220
|
+
if href.lower().endswith(".htm"):
|
|
221
|
+
doc_url = full_url
|
|
222
|
+
elif href.lower().endswith(".xpt"):
|
|
223
|
+
xpt_url = full_url
|
|
224
|
+
data_file_name = a.get_text(strip=True) or None
|
|
225
|
+
|
|
226
|
+
if xpt_url and doc_url:
|
|
227
|
+
out.append(
|
|
228
|
+
{
|
|
229
|
+
"years": cycle_str,
|
|
230
|
+
"xpt_url": xpt_url,
|
|
231
|
+
"doc_url": doc_url,
|
|
232
|
+
"data_file_name": data_file_name,
|
|
233
|
+
}
|
|
234
|
+
)
|
|
235
|
+
|
|
236
|
+
return out
|
|
237
|
+
|
|
238
|
+
|
|
239
|
+
def download_nhanes_data(
|
|
240
|
+
output_dir: str,
|
|
241
|
+
components: Optional[List[str]] = DEFAULT_COMPONENTS,
|
|
242
|
+
years: Optional[List[Union[int, str]]] = None,
|
|
243
|
+
with_docs: bool = True,
|
|
244
|
+
base_search_url: str = DEFAULT_BASE_SEARCH_URL,
|
|
245
|
+
base_domain: str = DEFAULT_BASE_DOMAIN,
|
|
246
|
+
):
|
|
247
|
+
"""Download NHANES XPT files and optional docs for components and years."""
|
|
248
|
+
ensure_dir(output_dir)
|
|
249
|
+
|
|
250
|
+
for comp in components:
|
|
251
|
+
entries = get_file_links_for_component(
|
|
252
|
+
comp,
|
|
253
|
+
years=years,
|
|
254
|
+
base_search_url=base_search_url,
|
|
255
|
+
base_domain=base_domain,
|
|
256
|
+
)
|
|
257
|
+
print(f"> [FOUND] Found {len(entries)} datasets for component '{comp}' (matching years: {years})")
|
|
258
|
+
|
|
259
|
+
for e in entries:
|
|
260
|
+
xpt_url = e["xpt_url"]
|
|
261
|
+
doc_url = e["doc_url"]
|
|
262
|
+
cycle = e["years"] or "unknown"
|
|
263
|
+
|
|
264
|
+
date = extract_date_from_url(doc_url)
|
|
265
|
+
|
|
266
|
+
out_subdir = os.path.join(output_dir, cycle, date, comp)
|
|
267
|
+
ensure_dir(out_subdir)
|
|
268
|
+
|
|
269
|
+
filename = xpt_url.split("/")[-1]
|
|
270
|
+
out_path = os.path.join(out_subdir, filename)
|
|
271
|
+
|
|
272
|
+
if os.path.exists(out_path):
|
|
273
|
+
print(f"> [CHECK] Already downloaded: {cycle}/{filename}")
|
|
274
|
+
else:
|
|
275
|
+
print(f"> [.....] Downloading {cycle}/{filename}")
|
|
276
|
+
download_file(xpt_url, out_path)
|
|
277
|
+
|
|
278
|
+
if with_docs:
|
|
279
|
+
try:
|
|
280
|
+
download_and_extract_docs(doc_url, out_subdir)
|
|
281
|
+
except Exception as err:
|
|
282
|
+
logger.warning("Failed to extract docs for %s (%s)", doc_url, err)
|
|
@@ -0,0 +1,444 @@
|
|
|
1
|
+
import logging
|
|
2
|
+
import os
|
|
3
|
+
import re
|
|
4
|
+
import json
|
|
5
|
+
from datetime import datetime
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
from typing import Any, Dict, List, Optional
|
|
8
|
+
|
|
9
|
+
import pandas as pd
|
|
10
|
+
|
|
11
|
+
from .downloader import download_nhanes_data
|
|
12
|
+
|
|
13
|
+
logger = logging.getLogger(__name__)
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
def snake_case(s: str) -> str:
|
|
17
|
+
s = s.strip().lower()
|
|
18
|
+
s = re.sub(r"[^\w\s]", " ", s)
|
|
19
|
+
s = re.sub(r"\s+", "_", s)
|
|
20
|
+
s = re.sub(r"_+", "_", s).strip("_")
|
|
21
|
+
return s or "unnamed"
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def read_nhanes_data(
|
|
25
|
+
file_path: str,
|
|
26
|
+
use_csv_for_labels: bool = True,
|
|
27
|
+
local_doc_csv_dir: Optional[str] = None,
|
|
28
|
+
allow_download_prompt: bool = True,
|
|
29
|
+
attach_documentation: bool = True,
|
|
30
|
+
add_documentation_column: bool = False,
|
|
31
|
+
) -> pd.DataFrame:
|
|
32
|
+
"""
|
|
33
|
+
Read an NHANES XPT file and optionally rename columns using <dataset>_variables.csv.
|
|
34
|
+
|
|
35
|
+
use_csv_for_labels: If True, attempts to find a corresponding <dataset>_variables.csv file
|
|
36
|
+
and uses it to rename columns based on the 'label' field.
|
|
37
|
+
These can be downloaded using the `download_nhanes_data` function with `with_docs=True`
|
|
38
|
+
attach_documentation: If True, attach per-column documentation metadata under
|
|
39
|
+
df.attrs['documentation'] and a docs table under df.attrs['documentation_df'].
|
|
40
|
+
add_documentation_column: If True, add a 'documentation' column where each row
|
|
41
|
+
contains the same per-column documentation dictionary.
|
|
42
|
+
"""
|
|
43
|
+
if not os.path.exists(file_path):
|
|
44
|
+
if allow_download_prompt:
|
|
45
|
+
parts = file_path.split(os.sep)
|
|
46
|
+
if len(parts) >= 4:
|
|
47
|
+
component_candidate = parts[-2]
|
|
48
|
+
cycle_candidate = parts[-4]
|
|
49
|
+
|
|
50
|
+
if re.match(r"\d{4}-\d{4}", cycle_candidate):
|
|
51
|
+
print(f"File not found: {file_path}")
|
|
52
|
+
print(
|
|
53
|
+
"It looks like you're missing data for "
|
|
54
|
+
f"Component='{component_candidate}', Cycle='{cycle_candidate}'"
|
|
55
|
+
)
|
|
56
|
+
response = input("Would you like to try downloading it now? [y/N] ").lower().strip()
|
|
57
|
+
if response == "y":
|
|
58
|
+
output_dir = os.sep.join(parts[:-4])
|
|
59
|
+
if not output_dir:
|
|
60
|
+
output_dir = "."
|
|
61
|
+
|
|
62
|
+
print(f"Downloading to {output_dir}...")
|
|
63
|
+
download_nhanes_data(
|
|
64
|
+
components=[component_candidate],
|
|
65
|
+
output_dir=output_dir,
|
|
66
|
+
years=[cycle_candidate],
|
|
67
|
+
)
|
|
68
|
+
if not os.path.exists(file_path):
|
|
69
|
+
raise FileNotFoundError(
|
|
70
|
+
f"Download completed but file is still missing: {file_path}"
|
|
71
|
+
)
|
|
72
|
+
else:
|
|
73
|
+
raise FileNotFoundError(f"File not found: {file_path}")
|
|
74
|
+
else:
|
|
75
|
+
raise FileNotFoundError(f"File not found: {file_path}")
|
|
76
|
+
else:
|
|
77
|
+
raise FileNotFoundError(f"File not found: {file_path}")
|
|
78
|
+
else:
|
|
79
|
+
raise FileNotFoundError(f"File not found: {file_path}")
|
|
80
|
+
|
|
81
|
+
try:
|
|
82
|
+
df = pd.read_sas(file_path)
|
|
83
|
+
except Exception as e:
|
|
84
|
+
raise IOError(f"Failed to read XPT file {file_path}: {e}")
|
|
85
|
+
|
|
86
|
+
if use_csv_for_labels or attach_documentation:
|
|
87
|
+
base_name = os.path.splitext(os.path.basename(file_path))[0]
|
|
88
|
+
doc_csv_name = f"{base_name}_variables.csv"
|
|
89
|
+
|
|
90
|
+
search_dirs = [os.path.dirname(file_path)]
|
|
91
|
+
if local_doc_csv_dir:
|
|
92
|
+
search_dirs.insert(0, local_doc_csv_dir)
|
|
93
|
+
|
|
94
|
+
doc_csv_path = None
|
|
95
|
+
for d in search_dirs:
|
|
96
|
+
candidate = os.path.join(d, doc_csv_name)
|
|
97
|
+
if os.path.exists(candidate):
|
|
98
|
+
doc_csv_path = candidate
|
|
99
|
+
break
|
|
100
|
+
|
|
101
|
+
if not doc_csv_path and allow_download_prompt:
|
|
102
|
+
parts = file_path.split(os.sep)
|
|
103
|
+
if len(parts) >= 4:
|
|
104
|
+
component_candidate = parts[-2]
|
|
105
|
+
cycle_candidate = parts[-4]
|
|
106
|
+
if re.match(r"\d{4}-\d{4}", cycle_candidate):
|
|
107
|
+
print(
|
|
108
|
+
"Documentation CSV not found. "
|
|
109
|
+
f"Try downloading docs for Component='{component_candidate}', Cycle='{cycle_candidate}'?"
|
|
110
|
+
)
|
|
111
|
+
response = input("Download docs now? [y/N] ").lower().strip()
|
|
112
|
+
if response == "y":
|
|
113
|
+
output_dir = os.sep.join(parts[:-4])
|
|
114
|
+
if not output_dir:
|
|
115
|
+
output_dir = "."
|
|
116
|
+
try:
|
|
117
|
+
download_nhanes_data(
|
|
118
|
+
components=[component_candidate],
|
|
119
|
+
output_dir=output_dir,
|
|
120
|
+
years=[cycle_candidate],
|
|
121
|
+
with_docs=True,
|
|
122
|
+
)
|
|
123
|
+
except Exception as e:
|
|
124
|
+
logger.warning("Failed to download docs for %s: %s", file_path, e)
|
|
125
|
+
|
|
126
|
+
for d in search_dirs:
|
|
127
|
+
candidate = os.path.join(d, doc_csv_name)
|
|
128
|
+
if os.path.exists(candidate):
|
|
129
|
+
doc_csv_path = candidate
|
|
130
|
+
break
|
|
131
|
+
|
|
132
|
+
if not doc_csv_path:
|
|
133
|
+
logger.warning(
|
|
134
|
+
"Variables CSV not found for %s. Returning DataFrame with original column names.",
|
|
135
|
+
base_name,
|
|
136
|
+
)
|
|
137
|
+
if attach_documentation:
|
|
138
|
+
df.attrs["documentation"] = {}
|
|
139
|
+
df.attrs["documentation_df"] = pd.DataFrame(
|
|
140
|
+
columns=["variable_name", "label", "description", "value_meanings", "link"]
|
|
141
|
+
)
|
|
142
|
+
if add_documentation_column:
|
|
143
|
+
df["documentation"] = [df.attrs["documentation"] for _ in range(len(df))]
|
|
144
|
+
return df
|
|
145
|
+
try:
|
|
146
|
+
doc_df = pd.read_csv(doc_csv_path)
|
|
147
|
+
except Exception as e:
|
|
148
|
+
logger.warning(
|
|
149
|
+
"Failed to read variables CSV for documentation %s: %s",
|
|
150
|
+
doc_csv_path,
|
|
151
|
+
e,
|
|
152
|
+
)
|
|
153
|
+
if attach_documentation:
|
|
154
|
+
df.attrs["documentation"] = {}
|
|
155
|
+
df.attrs["documentation_df"] = pd.DataFrame(
|
|
156
|
+
columns=["variable_name", "label", "description", "value_meanings", "link"]
|
|
157
|
+
)
|
|
158
|
+
if add_documentation_column:
|
|
159
|
+
df["documentation"] = [df.attrs["documentation"] for _ in range(len(df))]
|
|
160
|
+
return df
|
|
161
|
+
|
|
162
|
+
if "variable_name" not in doc_df.columns or "label" not in doc_df.columns:
|
|
163
|
+
logger.warning(
|
|
164
|
+
"Variables CSV for documentation %s missing 'variable_name' or 'label' columns.",
|
|
165
|
+
doc_csv_path,
|
|
166
|
+
)
|
|
167
|
+
if attach_documentation:
|
|
168
|
+
df.attrs["documentation"] = {}
|
|
169
|
+
df.attrs["documentation_df"] = pd.DataFrame(
|
|
170
|
+
columns=["variable_name", "label", "description", "value_meanings", "link"]
|
|
171
|
+
)
|
|
172
|
+
if add_documentation_column:
|
|
173
|
+
df["documentation"] = [df.attrs["documentation"] for _ in range(len(df))]
|
|
174
|
+
return df
|
|
175
|
+
|
|
176
|
+
label_map = doc_df.dropna(subset=["variable_name", "label"]).set_index("variable_name")[
|
|
177
|
+
"label"
|
|
178
|
+
].to_dict()
|
|
179
|
+
|
|
180
|
+
doc_records = {}
|
|
181
|
+
for _, row in doc_df.iterrows():
|
|
182
|
+
var = str(row.get("variable_name", "") or "").strip()
|
|
183
|
+
if not var:
|
|
184
|
+
continue
|
|
185
|
+
|
|
186
|
+
raw_value_meanings = row.get("value_meanings", "")
|
|
187
|
+
value_meanings: Any = None
|
|
188
|
+
if isinstance(raw_value_meanings, str) and raw_value_meanings.strip():
|
|
189
|
+
try:
|
|
190
|
+
value_meanings = json.loads(raw_value_meanings)
|
|
191
|
+
except Exception:
|
|
192
|
+
value_meanings = raw_value_meanings.strip()
|
|
193
|
+
|
|
194
|
+
doc_records[var] = {
|
|
195
|
+
"variable_name": var,
|
|
196
|
+
"label": None if pd.isna(row.get("label")) else str(row.get("label")),
|
|
197
|
+
"description": None if pd.isna(row.get("description")) else str(row.get("description")),
|
|
198
|
+
"value_meanings": value_meanings,
|
|
199
|
+
"link": None if pd.isna(row.get("link")) else str(row.get("link")),
|
|
200
|
+
}
|
|
201
|
+
|
|
202
|
+
rename_map = {}
|
|
203
|
+
used_names = set()
|
|
204
|
+
|
|
205
|
+
def get_unique_name(base, existing):
|
|
206
|
+
if base not in existing:
|
|
207
|
+
return base
|
|
208
|
+
i = 2
|
|
209
|
+
while f"{base}_{i}" in existing:
|
|
210
|
+
i += 1
|
|
211
|
+
return f"{base}_{i}"
|
|
212
|
+
|
|
213
|
+
for col in df.columns:
|
|
214
|
+
original_col = col
|
|
215
|
+
if original_col in label_map and str(label_map[original_col]).strip():
|
|
216
|
+
new_name = snake_case(str(label_map[original_col]))
|
|
217
|
+
else:
|
|
218
|
+
new_name = snake_case(original_col)
|
|
219
|
+
|
|
220
|
+
final_name = get_unique_name(new_name, used_names)
|
|
221
|
+
used_names.add(final_name)
|
|
222
|
+
rename_map[original_col] = final_name
|
|
223
|
+
|
|
224
|
+
if use_csv_for_labels:
|
|
225
|
+
df = df.rename(columns=rename_map)
|
|
226
|
+
|
|
227
|
+
if attach_documentation:
|
|
228
|
+
documentation: Dict[str, Dict[str, Any]] = {}
|
|
229
|
+
for original_col in rename_map:
|
|
230
|
+
final_col = rename_map[original_col] if use_csv_for_labels else original_col
|
|
231
|
+
rec = doc_records.get(original_col, {})
|
|
232
|
+
documentation[final_col] = {
|
|
233
|
+
"variable_name": original_col,
|
|
234
|
+
"label": rec.get("label"),
|
|
235
|
+
"description": rec.get("description"),
|
|
236
|
+
"value_meanings": rec.get("value_meanings"),
|
|
237
|
+
"link": rec.get("link"),
|
|
238
|
+
}
|
|
239
|
+
|
|
240
|
+
df.attrs["documentation"] = documentation
|
|
241
|
+
df.attrs["documentation_df"] = pd.DataFrame(list(documentation.values()))
|
|
242
|
+
|
|
243
|
+
if add_documentation_column:
|
|
244
|
+
df["documentation"] = [documentation for _ in range(len(df))]
|
|
245
|
+
return df
|
|
246
|
+
|
|
247
|
+
|
|
248
|
+
def get_column_doc(df: pd.DataFrame, column_name: str) -> Optional[Dict[str, Any]]:
|
|
249
|
+
"""Return documentation metadata for a column from df.attrs['documentation']."""
|
|
250
|
+
docs = df.attrs.get("documentation")
|
|
251
|
+
if isinstance(docs, dict):
|
|
252
|
+
entry = docs.get(column_name)
|
|
253
|
+
if isinstance(entry, dict):
|
|
254
|
+
return entry
|
|
255
|
+
|
|
256
|
+
# Fallback to variable_name lookup in case docs are keyed differently.
|
|
257
|
+
if isinstance(docs, dict):
|
|
258
|
+
for _, entry in docs.items():
|
|
259
|
+
if isinstance(entry, dict) and entry.get("variable_name") == column_name:
|
|
260
|
+
return entry
|
|
261
|
+
|
|
262
|
+
return None
|
|
263
|
+
|
|
264
|
+
|
|
265
|
+
def search_nhanes_local_data(
|
|
266
|
+
search_dir: str,
|
|
267
|
+
*,
|
|
268
|
+
recursive: bool = True,
|
|
269
|
+
include_extensions: Optional[List[str]] = None,
|
|
270
|
+
only_nhanes_like: bool = True,
|
|
271
|
+
print_results: bool = True,
|
|
272
|
+
) -> pd.DataFrame:
|
|
273
|
+
"""Search a local directory for NHANES-relevant files and return an inventory DataFrame."""
|
|
274
|
+
root = Path(search_dir).expanduser().resolve()
|
|
275
|
+
if not root.exists() or not root.is_dir():
|
|
276
|
+
raise NotADirectoryError(f"search_dir is not a valid directory: {search_dir}")
|
|
277
|
+
|
|
278
|
+
exts = include_extensions or ["xpt", "csv"]
|
|
279
|
+
exts = {("." + e.lower().lstrip(".")) for e in exts}
|
|
280
|
+
|
|
281
|
+
cycle_re = re.compile(r"^\d{4}-\d{4}$")
|
|
282
|
+
year_re = re.compile(r"^\d{4}$")
|
|
283
|
+
|
|
284
|
+
def infer_parts(p: Path) -> Dict[str, Optional[str]]:
|
|
285
|
+
parts = list(p.parts)
|
|
286
|
+
cycle = None
|
|
287
|
+
year = None
|
|
288
|
+
component = None
|
|
289
|
+
|
|
290
|
+
cycle_idx = None
|
|
291
|
+
for i, name in enumerate(parts):
|
|
292
|
+
if cycle_re.match(name):
|
|
293
|
+
cycle = name
|
|
294
|
+
cycle_idx = i
|
|
295
|
+
break
|
|
296
|
+
|
|
297
|
+
if cycle_idx is not None:
|
|
298
|
+
if cycle_idx + 1 < len(parts) and year_re.match(parts[cycle_idx + 1]):
|
|
299
|
+
year = parts[cycle_idx + 1]
|
|
300
|
+
if cycle_idx + 2 < len(parts):
|
|
301
|
+
component = parts[cycle_idx + 2]
|
|
302
|
+
else:
|
|
303
|
+
if cycle_idx + 1 < len(parts):
|
|
304
|
+
component = parts[cycle_idx + 1]
|
|
305
|
+
|
|
306
|
+
return {"cycle": cycle, "year": year, "component": component}
|
|
307
|
+
|
|
308
|
+
def classify(p: Path) -> str:
|
|
309
|
+
name = p.name.lower()
|
|
310
|
+
if p.suffix.lower() == ".xpt":
|
|
311
|
+
return "data-xpt"
|
|
312
|
+
if name.endswith("_variables.csv"):
|
|
313
|
+
return "doc-variables-csv"
|
|
314
|
+
if p.suffix.lower() == ".csv":
|
|
315
|
+
return "csv"
|
|
316
|
+
return "other"
|
|
317
|
+
|
|
318
|
+
def nhanes_like_filter(p: Path) -> bool:
|
|
319
|
+
if p.suffix.lower() == ".xpt" or p.name.lower().endswith("_variables.csv"):
|
|
320
|
+
return True
|
|
321
|
+
|
|
322
|
+
if p.suffix.lower() != ".csv":
|
|
323
|
+
return False
|
|
324
|
+
|
|
325
|
+
meta = infer_parts(p)
|
|
326
|
+
if meta.get("cycle"):
|
|
327
|
+
return True
|
|
328
|
+
|
|
329
|
+
stem = p.stem
|
|
330
|
+
if stem.endswith("_variables"):
|
|
331
|
+
xpt_candidate = p.with_name(stem.replace("_variables", "") + ".xpt")
|
|
332
|
+
if xpt_candidate.exists():
|
|
333
|
+
return True
|
|
334
|
+
|
|
335
|
+
return False
|
|
336
|
+
|
|
337
|
+
it = root.rglob("*") if recursive else root.glob("*")
|
|
338
|
+
|
|
339
|
+
rows: List[Dict[str, Any]] = []
|
|
340
|
+
for p in it:
|
|
341
|
+
if not p.is_file():
|
|
342
|
+
continue
|
|
343
|
+
if p.suffix.lower() not in exts:
|
|
344
|
+
continue
|
|
345
|
+
if only_nhanes_like and not nhanes_like_filter(p):
|
|
346
|
+
continue
|
|
347
|
+
|
|
348
|
+
meta = infer_parts(p)
|
|
349
|
+
kind = classify(p)
|
|
350
|
+
|
|
351
|
+
try:
|
|
352
|
+
st = p.stat()
|
|
353
|
+
size_kb = st.st_size / 1024
|
|
354
|
+
mtime = datetime.fromtimestamp(st.st_mtime)
|
|
355
|
+
except OSError:
|
|
356
|
+
size_kb = None
|
|
357
|
+
mtime = None
|
|
358
|
+
|
|
359
|
+
rows.append(
|
|
360
|
+
{
|
|
361
|
+
"type": kind,
|
|
362
|
+
"cycle": meta.get("cycle"),
|
|
363
|
+
"year": meta.get("year"),
|
|
364
|
+
"component": meta.get("component"),
|
|
365
|
+
"file": p.name,
|
|
366
|
+
"stem": p.stem,
|
|
367
|
+
"ext": p.suffix.lower().lstrip("."),
|
|
368
|
+
"size_kb": None if size_kb is None else round(size_kb, 1),
|
|
369
|
+
"modified": None if mtime is None else mtime.strftime("%Y-%m-%d %H:%M"),
|
|
370
|
+
"path": str(p),
|
|
371
|
+
}
|
|
372
|
+
)
|
|
373
|
+
|
|
374
|
+
df = pd.DataFrame(rows)
|
|
375
|
+
if df.empty:
|
|
376
|
+
if print_results:
|
|
377
|
+
print(f"No NHANES-like files found in: {root}")
|
|
378
|
+
return df
|
|
379
|
+
|
|
380
|
+
sort_cols = ["cycle", "component", "year", "type", "file"]
|
|
381
|
+
sort_cols = [c for c in sort_cols if c in df.columns]
|
|
382
|
+
df = df.sort_values(sort_cols, na_position="last").reset_index(drop=True)
|
|
383
|
+
|
|
384
|
+
if print_results:
|
|
385
|
+
_print_nhanes_search_results(df, root=str(root))
|
|
386
|
+
|
|
387
|
+
return df
|
|
388
|
+
|
|
389
|
+
|
|
390
|
+
def _print_nhanes_search_results(df: pd.DataFrame, root: str) -> None:
|
|
391
|
+
"""Pretty-printer using rich when available, otherwise pandas text output."""
|
|
392
|
+
title = f"NHANES local files found in: {root} (n={len(df)})"
|
|
393
|
+
|
|
394
|
+
try:
|
|
395
|
+
from rich.console import Console
|
|
396
|
+
from rich.table import Table
|
|
397
|
+
from rich.text import Text
|
|
398
|
+
|
|
399
|
+
console = Console()
|
|
400
|
+
|
|
401
|
+
table = Table(title=title, show_lines=False, header_style="bold")
|
|
402
|
+
table.add_column("Type", style="cyan", no_wrap=True)
|
|
403
|
+
table.add_column("Cycle", style="magenta", no_wrap=True)
|
|
404
|
+
table.add_column("Year", style="magenta", no_wrap=True)
|
|
405
|
+
table.add_column("Component", style="green")
|
|
406
|
+
table.add_column("File", style="white")
|
|
407
|
+
table.add_column("Size KB", justify="right")
|
|
408
|
+
table.add_column("Modified", style="dim", no_wrap=True)
|
|
409
|
+
table.add_column("Path", style="dim")
|
|
410
|
+
|
|
411
|
+
for _, r in df.iterrows():
|
|
412
|
+
table.add_row(
|
|
413
|
+
str(r.get("type") or ""),
|
|
414
|
+
str(r.get("cycle") or ""),
|
|
415
|
+
str(r.get("year") or ""),
|
|
416
|
+
str(r.get("component") or ""),
|
|
417
|
+
str(r.get("file") or ""),
|
|
418
|
+
"" if pd.isna(r.get("size_kb")) else str(r.get("size_kb")),
|
|
419
|
+
str(r.get("modified") or ""),
|
|
420
|
+
str(r.get("path") or ""),
|
|
421
|
+
)
|
|
422
|
+
|
|
423
|
+
counts = df["type"].value_counts(dropna=False).to_dict()
|
|
424
|
+
summary = ", ".join([f"{k}={v}" for k, v in counts.items()])
|
|
425
|
+
console.print(table)
|
|
426
|
+
console.print(Text(f"Summary: {summary}", style="bold"))
|
|
427
|
+
|
|
428
|
+
return
|
|
429
|
+
|
|
430
|
+
except Exception:
|
|
431
|
+
pass
|
|
432
|
+
|
|
433
|
+
# TODO: log instead of print
|
|
434
|
+
print(title)
|
|
435
|
+
show_cols = ["type", "cycle", "year", "component", "file", "size_kb", "modified", "path"]
|
|
436
|
+
show_cols = [c for c in show_cols if c in df.columns]
|
|
437
|
+
df_out = df[show_cols].copy()
|
|
438
|
+
|
|
439
|
+
with pd.option_context("display.max_rows", 200, "display.max_colwidth", 120, "display.width", 200):
|
|
440
|
+
print(df_out.to_string(index=False))
|
|
441
|
+
|
|
442
|
+
counts = df["type"].value_counts(dropna=False).to_dict()
|
|
443
|
+
summary = ", ".join([f"{k}={v}" for k, v in counts.items()])
|
|
444
|
+
print(f"Summary: {summary}")
|
|
@@ -0,0 +1,89 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: healthdata
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Download and read NHANES health survey data.
|
|
5
|
+
Author-email: Ismail Elouafiq <contact@ismail.bio>
|
|
6
|
+
License-Expression: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/nidhog/healthdata
|
|
8
|
+
Project-URL: Repository, https://github.com/nidhog/healthdata
|
|
9
|
+
Project-URL: Documentation, https://github.com/nidhog/healthdata/tree/main/docs/nhanes
|
|
10
|
+
Project-URL: Issues, https://github.com/nidhog/healthdata/issues
|
|
11
|
+
Classifier: Development Status :: 3 - Alpha
|
|
12
|
+
Classifier: Intended Audience :: Science/Research
|
|
13
|
+
Classifier: Programming Language :: Python :: 3.14
|
|
14
|
+
Classifier: Topic :: Scientific/Engineering :: Medical Science Apps.
|
|
15
|
+
Requires-Python: >=3.14
|
|
16
|
+
Description-Content-Type: text/markdown
|
|
17
|
+
License-File: LICENSE
|
|
18
|
+
Requires-Dist: pandas
|
|
19
|
+
Requires-Dist: requests
|
|
20
|
+
Requires-Dist: beautifulsoup4
|
|
21
|
+
Requires-Dist: tqdm
|
|
22
|
+
Dynamic: license-file
|
|
23
|
+
|
|
24
|
+
# Health Data Downloader Package
|
|
25
|
+
The Health Data Downloader is a Python package that makes it easy to download and read real world health data.
|
|
26
|
+
|
|
27
|
+
Current data sources supported:
|
|
28
|
+
* **NHANES** National Health and Nutrition Examination Survey (CDC)
|
|
29
|
+
## The NHANES Submodule
|
|
30
|
+
If you want to download health data from NHANES this module can help you not only download the data but also add documentation from the official NHANES website, and read this data into a pandas dataframe.
|
|
31
|
+
|
|
32
|
+
[NHANES](https://www.cdc.gov/nchs/nhanes/about/) is the National Health and Nutrition Examination Survey which collects data about the health of adults and children in the United States.
|
|
33
|
+
|
|
34
|
+
_Note: This is not an official repository from the NHANES project nor the CDC_
|
|
35
|
+
|
|
36
|
+
This project provides workflows that enable you to:
|
|
37
|
+
- discover and download datasets from NHANES by component (meaning what type of data like Laboratory data, examination etc.) and by cycle (which years)
|
|
38
|
+
- read the XPT files into a pandas DataFrames
|
|
39
|
+
- extract NHANES variable documentation into CSV files and rename variable codes into human-readable names (such as `heart_rate` instead of `LF321`)
|
|
40
|
+
- search and summarize local files if they are already downloaded
|
|
41
|
+
|
|
42
|
+
## Quick start
|
|
43
|
+
In thie quick start we will:
|
|
44
|
+
1. Install required packages.
|
|
45
|
+
2. Run a download.
|
|
46
|
+
3. Read and inspect data.
|
|
47
|
+
|
|
48
|
+
```bash
|
|
49
|
+
pip install healthdata
|
|
50
|
+
```
|
|
51
|
+
You can choose which `components` you want to download (such as Demographics, Laboratory etc.) and which years (can be either "2017-2020" or "2016"). This will also download documentation about these fields from the NHANES website to add them when the data is read.
|
|
52
|
+
If the data is already downloaded in the output directory the module will log this and skip the download.
|
|
53
|
+
```python
|
|
54
|
+
from healthdata.nhanes.downloader import download_nhanes_data
|
|
55
|
+
|
|
56
|
+
download_nhanes_data(
|
|
57
|
+
components=["Demographics", "Examination"],
|
|
58
|
+
output_dir="nhanes_data",
|
|
59
|
+
years=["2017-2018"],
|
|
60
|
+
)
|
|
61
|
+
```
|
|
62
|
+
|
|
63
|
+
You can read the data you already downloaded (you will be prompted to download it if it doesn't exist)
|
|
64
|
+
```python
|
|
65
|
+
from healthdata.nhanes.reader import read_nhanes_data, search_nhanes_local_data
|
|
66
|
+
|
|
67
|
+
df = read_nhanes_data("nhanes_data/2017-2018/2017/Demographics/DEMO_J.XPT")
|
|
68
|
+
print(df.head())
|
|
69
|
+
|
|
70
|
+
inventory = search_nhanes_local_data(search_dir="nhanes_data")
|
|
71
|
+
print(inventory.head())
|
|
72
|
+
```
|
|
73
|
+
That's all folks! Now you will see that your dataframe already includes documentation.
|
|
74
|
+
## Current caveats
|
|
75
|
+
- HTML codebook parsing depends on current NHANES page text patterns. CDC page structure changes may require parser updates.
|
|
76
|
+
- Network errors and partial downloads are not retried automatically.
|
|
77
|
+
|
|
78
|
+
## Upcoming Additions
|
|
79
|
+
Planned additions include NCBI GEO and other open health and biomedical datasets. The current release supports NHANES only.
|
|
80
|
+
|
|
81
|
+
## License
|
|
82
|
+
This project is licensed under the [MIT License](LICENSE). Commercial use is permitted under its terms. Downloaded datasets may be subject to separate terms set by their respective providers.
|
|
83
|
+
|
|
84
|
+
For paid support or consulting, contact the author at contact@ismail.bio.
|
|
85
|
+
|
|
86
|
+
## Author
|
|
87
|
+
[Ismail Elouafiq](https://ismail.bio)
|
|
88
|
+
|
|
89
|
+
Contact: contact@ismail.bio
|
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
LICENSE
|
|
2
|
+
MANIFEST.in
|
|
3
|
+
README.md
|
|
4
|
+
pyproject.toml
|
|
5
|
+
src/healthdata/__init__.py
|
|
6
|
+
src/healthdata.egg-info/PKG-INFO
|
|
7
|
+
src/healthdata.egg-info/SOURCES.txt
|
|
8
|
+
src/healthdata.egg-info/dependency_links.txt
|
|
9
|
+
src/healthdata.egg-info/requires.txt
|
|
10
|
+
src/healthdata.egg-info/top_level.txt
|
|
11
|
+
src/healthdata/nhanes/__init__.py
|
|
12
|
+
src/healthdata/nhanes/downloader.py
|
|
13
|
+
src/healthdata/nhanes/reader.py
|
|
14
|
+
src/healthdata/nhanes/settings.py
|
|
15
|
+
tests/test_downloader.py
|
|
16
|
+
tests/test_public_api.py
|
|
17
|
+
tests/test_reader.py
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
healthdata
|
|
@@ -0,0 +1,72 @@
|
|
|
1
|
+
from healthdata.nhanes.downloader import (
|
|
2
|
+
extract_date_from_url,
|
|
3
|
+
get_file_links_for_component,
|
|
4
|
+
parse_nhanes_doc_variables,
|
|
5
|
+
)
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
class _DummyResponse:
|
|
9
|
+
def __init__(self, text):
|
|
10
|
+
self.text = text
|
|
11
|
+
|
|
12
|
+
def raise_for_status(self):
|
|
13
|
+
return None
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
def test_extract_date_from_url_cycle_and_year_and_fallback():
|
|
17
|
+
# We expect the function to extract the cycle or year based on the URL similar to below
|
|
18
|
+
assert extract_date_from_url("https://example.org/2017-2018/doc.htm") == "2017-2018"
|
|
19
|
+
assert extract_date_from_url("https://example.org/data/2019/file.xpt") == "2019"
|
|
20
|
+
assert extract_date_from_url("https://example.org/no-date") == "nodate"
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def test_parse_nhanes_doc_variables_extracts_basic_fields():
|
|
24
|
+
# this is currently what we expect to find in the HTML
|
|
25
|
+
html = """
|
|
26
|
+
<html><body>
|
|
27
|
+
Variable Name: SEQN
|
|
28
|
+
SAS Label: Respondent sequence number
|
|
29
|
+
English Text: Unique participant identifier
|
|
30
|
+
Target: Both males and females
|
|
31
|
+
</body></html>
|
|
32
|
+
"""
|
|
33
|
+
|
|
34
|
+
rows = parse_nhanes_doc_variables(html, "https://example.org/doc.htm")
|
|
35
|
+
|
|
36
|
+
assert len(rows) == 1
|
|
37
|
+
assert rows[0]["variable_name"] == "SEQN"
|
|
38
|
+
assert rows[0]["label"] == "Respondent sequence number"
|
|
39
|
+
assert rows[0]["description"] == "Unique participant identifier"
|
|
40
|
+
assert "value_meanings" in rows[0]
|
|
41
|
+
assert rows[0]["link"] == "https://example.org/doc.htm"
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def test_get_file_links_for_component_filters_years(monkeypatch):
|
|
45
|
+
# On NHANES website, the files are linked in this format:
|
|
46
|
+
expected_html = """
|
|
47
|
+
<table>
|
|
48
|
+
<tr>
|
|
49
|
+
<td>2017-2018</td>
|
|
50
|
+
<td><a href="/nchs/nhanes/2017-2018/P_DEMO.htm">P_DEMO</a></td>
|
|
51
|
+
<td><a href="/nchs/nhanes/2017-2018/P_DEMO.xpt">P_DEMO</a></td>
|
|
52
|
+
</tr>
|
|
53
|
+
<tr>
|
|
54
|
+
<td>2015-2016</td>
|
|
55
|
+
<td><a href="/nchs/nhanes/2015-2016/OLD_DEMO.htm">OLD_DEMO</a></td>
|
|
56
|
+
<td><a href="/nchs/nhanes/2015-2016/OLD_DEMO.xpt">OLD_DEMO</a></td>
|
|
57
|
+
</tr>
|
|
58
|
+
</table>
|
|
59
|
+
"""
|
|
60
|
+
|
|
61
|
+
def _fake_get(url, params=None):
|
|
62
|
+
assert params == {"Component": "Demographics"}
|
|
63
|
+
return _DummyResponse(expected_html)
|
|
64
|
+
|
|
65
|
+
monkeypatch.setattr("healthdata.nhanes.downloader.requests.get", _fake_get)
|
|
66
|
+
|
|
67
|
+
links = get_file_links_for_component("Demographics", years=["2017-2018"])
|
|
68
|
+
|
|
69
|
+
assert len(links) == 1
|
|
70
|
+
assert links[0]["years"] == "2017-2018"
|
|
71
|
+
assert links[0]["doc_url"].endswith("/P_DEMO.htm")
|
|
72
|
+
assert links[0]["xpt_url"].endswith("/P_DEMO.xpt")
|
|
@@ -0,0 +1,10 @@
|
|
|
1
|
+
import healthdata.nhanes as nhanes
|
|
2
|
+
|
|
3
|
+
|
|
4
|
+
def test_nhanes_public_api_exports_main_functions():
|
|
5
|
+
assert callable(nhanes.download_nhanes_data)
|
|
6
|
+
assert callable(nhanes.download_and_extract_docs)
|
|
7
|
+
assert callable(nhanes.get_file_links_for_component)
|
|
8
|
+
assert callable(nhanes.read_nhanes_data)
|
|
9
|
+
assert callable(nhanes.get_column_doc)
|
|
10
|
+
assert callable(nhanes.search_nhanes_local_data)
|
|
@@ -0,0 +1,87 @@
|
|
|
1
|
+
from pathlib import Path
|
|
2
|
+
|
|
3
|
+
import pandas as pd
|
|
4
|
+
|
|
5
|
+
from healthdata.nhanes.reader import get_column_doc, read_nhanes_data, search_nhanes_local_data, snake_case
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
def test_snake_case_handles_spacing_and_symbols():
|
|
9
|
+
assert snake_case(". Heart Rate (BPM)") == "heart_rate_bpm"
|
|
10
|
+
assert snake_case("___") == "unnamed"
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
def test_read_nhanes_data_renames_columns_from_variables_csv(tmp_path, monkeypatch):
|
|
14
|
+
xpt_path = tmp_path / "P_DEMO.xpt"
|
|
15
|
+
xpt_path.write_bytes(b"placeholder")
|
|
16
|
+
# we create a csv file similar to what we have
|
|
17
|
+
csv_path = tmp_path / "P_DEMO_variables.csv"
|
|
18
|
+
csv_path.write_text(
|
|
19
|
+
"variable_name,label\n"
|
|
20
|
+
"SEQN,Participant ID\n"
|
|
21
|
+
"RIDAGEYR,Age\n"
|
|
22
|
+
"DUPCOL,Age\n",
|
|
23
|
+
encoding="utf-8",
|
|
24
|
+
)
|
|
25
|
+
|
|
26
|
+
source_df = pd.DataFrame({"SEQN": [1], "RIDAGEYR": [30], "DUPCOL": [31]})
|
|
27
|
+
monkeypatch.setattr("healthdata.nhanes.reader.pd.read_sas", lambda _: source_df)
|
|
28
|
+
|
|
29
|
+
result = read_nhanes_data(str(xpt_path), use_csv_for_labels=True)
|
|
30
|
+
|
|
31
|
+
assert list(result.columns) == ["participant_id", "age", "age_2"]
|
|
32
|
+
assert "documentation" in result.attrs
|
|
33
|
+
assert result.attrs["documentation"]["participant_id"]["variable_name"] == "SEQN"
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def test_read_nhanes_data_attaches_documentation_without_renaming(tmp_path, monkeypatch):
|
|
37
|
+
xpt_path = tmp_path / "P_DEMO.xpt"
|
|
38
|
+
xpt_path.write_bytes(b"placeholder")
|
|
39
|
+
csv_path = tmp_path / "P_DEMO_variables.csv"
|
|
40
|
+
csv_path.write_text(
|
|
41
|
+
"variable_name,label,description,value_meanings,link\n"
|
|
42
|
+
"SEQN,Participant ID,Unique id,\"[{\"\"code\"\": \"\"1\"\", \"\"meaning\"\": \"\"Yes\"\"}]\",https://example.org/doc.htm\n",
|
|
43
|
+
encoding="utf-8",
|
|
44
|
+
)
|
|
45
|
+
|
|
46
|
+
source_df = pd.DataFrame({"SEQN": [1]})
|
|
47
|
+
monkeypatch.setattr("healthdata.nhanes.reader.pd.read_sas", lambda _: source_df)
|
|
48
|
+
|
|
49
|
+
result = read_nhanes_data(
|
|
50
|
+
str(xpt_path),
|
|
51
|
+
use_csv_for_labels=False,
|
|
52
|
+
attach_documentation=True,
|
|
53
|
+
)
|
|
54
|
+
|
|
55
|
+
assert list(result.columns) == ["SEQN"]
|
|
56
|
+
assert "documentation" in result.attrs
|
|
57
|
+
assert result.attrs["documentation"]["SEQN"]["description"] == "Unique id"
|
|
58
|
+
assert result.attrs["documentation"]["SEQN"]["value_meanings"] == [{"code": "1", "meaning": "Yes"}]
|
|
59
|
+
assert get_column_doc(result, "SEQN")["label"] == "Participant ID"
|
|
60
|
+
assert get_column_doc(result, "MISSING") is None
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
def test_search_nhanes_local_data_returns_expected_inventory(tmp_path):
|
|
64
|
+
base = tmp_path / "2017-2020" / "2017" / "Demographics"
|
|
65
|
+
base.mkdir(parents=True)
|
|
66
|
+
|
|
67
|
+
(base / "P_DEMO.xpt").write_bytes(b"xpt")
|
|
68
|
+
(base / "P_DEMO_variables.csv").write_text("variable_name,label\nSEQN,Participant ID\n", encoding="utf-8")
|
|
69
|
+
(tmp_path / "random.csv").write_text("a,b\n1,2\n", encoding="utf-8")
|
|
70
|
+
|
|
71
|
+
result = search_nhanes_local_data(str(tmp_path), print_results=False)
|
|
72
|
+
|
|
73
|
+
assert len(result) == 2
|
|
74
|
+
assert set(result["type"]) == {"data-xpt", "doc-variables-csv"}
|
|
75
|
+
assert set(result["cycle"]) == {"2017-2020"}
|
|
76
|
+
assert set(result["component"]) == {"Demographics"}
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
def test_search_nhanes_local_data_invalid_dir_raises(tmp_path):
|
|
80
|
+
missing = tmp_path / "does-not-exist"
|
|
81
|
+
|
|
82
|
+
try:
|
|
83
|
+
search_nhanes_local_data(str(missing), print_results=False)
|
|
84
|
+
except NotADirectoryError as exc:
|
|
85
|
+
assert "not a valid directory" in str(exc)
|
|
86
|
+
else:
|
|
87
|
+
raise AssertionError("Expected NotADirectoryError")
|