ctkit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- ctkit/__init__.py +144 -0
- ctkit/api.py +358 -0
- ctkit/cli.py +754 -0
- ctkit/config.py +453 -0
- ctkit/constants.py +297 -0
- ctkit/dataset.py +1449 -0
- ctkit/datasets.py +216 -0
- ctkit/features.py +177 -0
- ctkit/image.py +1141 -0
- ctkit/io.py +384 -0
- ctkit/metadata.py +233 -0
- ctkit/py.typed +0 -0
- ctkit/qc.py +482 -0
- ctkit/segmentation.py +367 -0
- ctkit/tcia.py +427 -0
- ctkit/validation.py +187 -0
- ctkit-0.1.0.dist-info/METADATA +168 -0
- ctkit-0.1.0.dist-info/RECORD +22 -0
- ctkit-0.1.0.dist-info/WHEEL +5 -0
- ctkit-0.1.0.dist-info/entry_points.txt +2 -0
- ctkit-0.1.0.dist-info/licenses/LICENSE +24 -0
- ctkit-0.1.0.dist-info/top_level.txt +1 -0
ctkit/datasets.py
ADDED
|
@@ -0,0 +1,216 @@
|
|
|
1
|
+
"""The catalog of datasets this package knows how to fetch and process.
|
|
2
|
+
|
|
3
|
+
Two groups:
|
|
4
|
+
|
|
5
|
+
* the TCGA and CPTAC collections in
|
|
6
|
+
:data:`~ctkit.constants.tcia_dataset_to_info`, which
|
|
7
|
+
carry curated processing settings (organs to segment, clipping window,
|
|
8
|
+
output dimensions);
|
|
9
|
+
* widely used public CT collections in :data:`EXTRA_CT_COLLECTIONS`, which are
|
|
10
|
+
listed for convenience and download the same way.
|
|
11
|
+
|
|
12
|
+
Any TCIA collection can be downloaded whether or not it appears here — see
|
|
13
|
+
:func:`~ctkit.tcia.list_collections`, which queries
|
|
14
|
+
the archive directly.
|
|
15
|
+
"""
|
|
16
|
+
|
|
17
|
+
from __future__ import annotations
|
|
18
|
+
|
|
19
|
+
from typing import Optional
|
|
20
|
+
|
|
21
|
+
from .constants import tcia_dataset_to_info
|
|
22
|
+
|
|
23
|
+
#: Public CT collections that are commonly used as benchmarks or pretraining
|
|
24
|
+
#: corpora. ``clip_min,clip_max`` follows the same convention as the curated
|
|
25
|
+
#: TCGA/CPTAC entries: the window to clip to before feature extraction.
|
|
26
|
+
EXTRA_CT_COLLECTIONS = {
|
|
27
|
+
"lidc-idri": {
|
|
28
|
+
"collection": "LIDC-IDRI",
|
|
29
|
+
"project": "tcia",
|
|
30
|
+
"cancer_organ": "lung",
|
|
31
|
+
"cancer_type": "lung nodules (screening)",
|
|
32
|
+
"description": "1,010 thoracic CT scans with annotated lung nodules.",
|
|
33
|
+
"totalsegmentator_organs": [
|
|
34
|
+
"lung_upper_lobe_left", "lung_lower_lobe_left", "lung_upper_lobe_right",
|
|
35
|
+
"lung_middle_lobe_right", "lung_lower_lobe_right",
|
|
36
|
+
],
|
|
37
|
+
"clip_min,clip_max": (-1000, 400),
|
|
38
|
+
},
|
|
39
|
+
"nsclc-radiomics": {
|
|
40
|
+
"collection": "NSCLC-Radiomics",
|
|
41
|
+
"project": "tcia",
|
|
42
|
+
"cancer_organ": "lung",
|
|
43
|
+
"cancer_type": "non-small cell lung cancer",
|
|
44
|
+
"description": "422 NSCLC CT scans with manual tumor delineations (Aerts Lung1).",
|
|
45
|
+
"totalsegmentator_organs": [
|
|
46
|
+
"lung_upper_lobe_left", "lung_lower_lobe_left", "lung_upper_lobe_right",
|
|
47
|
+
"lung_middle_lobe_right", "lung_lower_lobe_right",
|
|
48
|
+
],
|
|
49
|
+
"clip_min,clip_max": (-1000, 400),
|
|
50
|
+
},
|
|
51
|
+
"nsclc-radiogenomics": {
|
|
52
|
+
"collection": "NSCLC Radiogenomics",
|
|
53
|
+
"project": "tcia",
|
|
54
|
+
"cancer_organ": "lung",
|
|
55
|
+
"cancer_type": "non-small cell lung cancer",
|
|
56
|
+
"description": "CT/PET with matched RNA-seq and mutation data.",
|
|
57
|
+
"totalsegmentator_organs": [
|
|
58
|
+
"lung_upper_lobe_left", "lung_lower_lobe_left", "lung_upper_lobe_right",
|
|
59
|
+
"lung_middle_lobe_right", "lung_lower_lobe_right",
|
|
60
|
+
],
|
|
61
|
+
"clip_min,clip_max": (-1000, 400),
|
|
62
|
+
},
|
|
63
|
+
"pancreas-ct": {
|
|
64
|
+
"collection": "Pancreas-CT",
|
|
65
|
+
"project": "tcia",
|
|
66
|
+
"cancer_organ": "pancreas",
|
|
67
|
+
"cancer_type": "healthy pancreas (segmentation benchmark)",
|
|
68
|
+
"description": "82 contrast-enhanced abdominal CT scans with pancreas masks.",
|
|
69
|
+
"totalsegmentator_organs": ["pancreas"],
|
|
70
|
+
"clip_min,clip_max": (-200, 300),
|
|
71
|
+
},
|
|
72
|
+
"hcc-tace-seg": {
|
|
73
|
+
"collection": "HCC-TACE-Seg",
|
|
74
|
+
"project": "tcia",
|
|
75
|
+
"cancer_organ": "liver",
|
|
76
|
+
"cancer_type": "hepatocellular carcinoma",
|
|
77
|
+
"description": "Liver CT before/after transarterial chemoembolization, with masks.",
|
|
78
|
+
"totalsegmentator_organs": ["liver"],
|
|
79
|
+
"clip_min,clip_max": (-200, 400),
|
|
80
|
+
},
|
|
81
|
+
"colorectal-liver-metastases": {
|
|
82
|
+
"collection": "Colorectal-Liver-Metastases",
|
|
83
|
+
"project": "tcia",
|
|
84
|
+
"cancer_organ": "liver",
|
|
85
|
+
"cancer_type": "colorectal liver metastases",
|
|
86
|
+
"description": "Preoperative liver CT with tumor and liver segmentations.",
|
|
87
|
+
"totalsegmentator_organs": ["liver"],
|
|
88
|
+
"clip_min,clip_max": (-200, 400),
|
|
89
|
+
},
|
|
90
|
+
"stageii-colorectal-ct": {
|
|
91
|
+
"collection": "StageII-Colorectal-CT",
|
|
92
|
+
"project": "tcia",
|
|
93
|
+
"cancer_organ": "colon",
|
|
94
|
+
"cancer_type": "stage II colorectal cancer",
|
|
95
|
+
"description": "Preoperative CT for stage II colorectal cancer.",
|
|
96
|
+
"totalsegmentator_organs": ["colon"],
|
|
97
|
+
"clip_min,clip_max": (-200, 300),
|
|
98
|
+
},
|
|
99
|
+
"lungct-diagnosis": {
|
|
100
|
+
"collection": "LungCT-Diagnosis",
|
|
101
|
+
"project": "tcia",
|
|
102
|
+
"cancer_organ": "lung",
|
|
103
|
+
"cancer_type": "lung adenocarcinoma",
|
|
104
|
+
"description": "Diagnostic chest CT with survival outcomes.",
|
|
105
|
+
"totalsegmentator_organs": [
|
|
106
|
+
"lung_upper_lobe_left", "lung_lower_lobe_left", "lung_upper_lobe_right",
|
|
107
|
+
"lung_middle_lobe_right", "lung_lower_lobe_right",
|
|
108
|
+
],
|
|
109
|
+
"clip_min,clip_max": (-1000, 400),
|
|
110
|
+
},
|
|
111
|
+
"rider-lung-ct": {
|
|
112
|
+
"collection": "RIDER Lung CT",
|
|
113
|
+
"project": "tcia",
|
|
114
|
+
"cancer_organ": "lung",
|
|
115
|
+
"cancer_type": "non-small cell lung cancer (test-retest)",
|
|
116
|
+
"description": "Same-day repeat CT scans — the standard set for testing "
|
|
117
|
+
"whether a feature is reproducible.",
|
|
118
|
+
"totalsegmentator_organs": [
|
|
119
|
+
"lung_upper_lobe_left", "lung_lower_lobe_left", "lung_upper_lobe_right",
|
|
120
|
+
"lung_middle_lobe_right", "lung_lower_lobe_right",
|
|
121
|
+
],
|
|
122
|
+
"clip_min,clip_max": (-1000, 400),
|
|
123
|
+
},
|
|
124
|
+
"ct-colonography": {
|
|
125
|
+
"collection": "CT COLONOGRAPHY",
|
|
126
|
+
"project": "tcia",
|
|
127
|
+
"cancer_organ": "colon",
|
|
128
|
+
"cancer_type": "colorectal polyps (screening)",
|
|
129
|
+
"description": "825 CT colonography cases; large and mostly healthy anatomy.",
|
|
130
|
+
"totalsegmentator_organs": ["colon"],
|
|
131
|
+
"clip_min,clip_max": (-1000, 400),
|
|
132
|
+
},
|
|
133
|
+
}
|
|
134
|
+
|
|
135
|
+
#: Extra files that go with a collection but live outside TCIA.
|
|
136
|
+
SUPPLEMENTARY_DOWNLOADS = {
|
|
137
|
+
"tcga-kirc": {
|
|
138
|
+
"segmentations": {
|
|
139
|
+
"url": "https://zenodo.org/records/13244892/files/kidney-ct.zip?download=1",
|
|
140
|
+
"description": "AI and radiologist-reviewed kidney/tumor segmentations "
|
|
141
|
+
"(Scientific Reports, 2024).",
|
|
142
|
+
"filename": "kidney-ct.zip",
|
|
143
|
+
},
|
|
144
|
+
},
|
|
145
|
+
}
|
|
146
|
+
|
|
147
|
+
|
|
148
|
+
def all_datasets() -> dict:
|
|
149
|
+
"""Every cataloged dataset, keyed by its short name."""
|
|
150
|
+
catalog = {}
|
|
151
|
+
for key, info in tcia_dataset_to_info.items():
|
|
152
|
+
entry = dict(info)
|
|
153
|
+
entry.setdefault("collection", key.upper())
|
|
154
|
+
entry.setdefault("curated", True)
|
|
155
|
+
catalog[key] = entry
|
|
156
|
+
for key, info in EXTRA_CT_COLLECTIONS.items():
|
|
157
|
+
entry = dict(info)
|
|
158
|
+
entry.setdefault("curated", False)
|
|
159
|
+
catalog[key] = entry
|
|
160
|
+
return catalog
|
|
161
|
+
|
|
162
|
+
|
|
163
|
+
def normalize_name(name: str) -> str:
|
|
164
|
+
"""Accept ``TCGA-KIRC``, ``tcga_kirc`` or ``TCGA KIRC`` for the same dataset."""
|
|
165
|
+
return str(name).strip().lower().replace("_", "-").replace(" ", "-")
|
|
166
|
+
|
|
167
|
+
|
|
168
|
+
def get_dataset_info(name: str) -> dict:
|
|
169
|
+
"""Look up one dataset. Raises :class:`KeyError` with suggestions."""
|
|
170
|
+
catalog = all_datasets()
|
|
171
|
+
key = normalize_name(name)
|
|
172
|
+
if key in catalog:
|
|
173
|
+
return {"name": key, **catalog[key]}
|
|
174
|
+
|
|
175
|
+
# Allow the TCIA collection name itself, e.g. "NSCLC-Radiomics".
|
|
176
|
+
for candidate, info in catalog.items():
|
|
177
|
+
if normalize_name(info.get("collection", candidate)) == key:
|
|
178
|
+
return {"name": candidate, **info}
|
|
179
|
+
|
|
180
|
+
close = [candidate for candidate in catalog if key in candidate or candidate in key]
|
|
181
|
+
hint = f" Did you mean: {', '.join(sorted(close))}?" if close else ""
|
|
182
|
+
raise KeyError(
|
|
183
|
+
f"Unknown dataset {name!r}.{hint} Use list_datasets() to see the catalog, "
|
|
184
|
+
"or pass any TCIA collection name directly to download()."
|
|
185
|
+
)
|
|
186
|
+
|
|
187
|
+
|
|
188
|
+
def collection_for(name: str) -> str:
|
|
189
|
+
"""The TCIA collection name for a dataset (falls back to the name itself)."""
|
|
190
|
+
try:
|
|
191
|
+
return get_dataset_info(name)["collection"]
|
|
192
|
+
except KeyError:
|
|
193
|
+
return str(name)
|
|
194
|
+
|
|
195
|
+
|
|
196
|
+
def list_datasets(project: Optional[str] = None):
|
|
197
|
+
"""The catalog as a DataFrame: name, collection, organ, cancer type."""
|
|
198
|
+
import pandas as pd
|
|
199
|
+
|
|
200
|
+
rows = []
|
|
201
|
+
for key, info in sorted(all_datasets().items()):
|
|
202
|
+
rows.append({
|
|
203
|
+
"name": key,
|
|
204
|
+
"collection": info.get("collection", key.upper()),
|
|
205
|
+
"project": info.get("project", ""),
|
|
206
|
+
"organ": info.get("cancer_organ", ""),
|
|
207
|
+
"cancer_type": info.get("cancer_type", ""),
|
|
208
|
+
"curated_settings": bool(info.get("curated")),
|
|
209
|
+
"organs_to_segment": ", ".join(info.get("totalsegmentator_organs") or []),
|
|
210
|
+
"clip_window": str(info.get("clip_min,clip_max", "")),
|
|
211
|
+
"description": info.get("description", ""),
|
|
212
|
+
})
|
|
213
|
+
frame = pd.DataFrame(rows)
|
|
214
|
+
if project:
|
|
215
|
+
frame = frame[frame["project"] == project].reset_index(drop=True)
|
|
216
|
+
return frame
|
ctkit/features.py
ADDED
|
@@ -0,0 +1,177 @@
|
|
|
1
|
+
"""Radiomic feature extraction with PyRadiomics.
|
|
2
|
+
|
|
3
|
+
PyRadiomics accepts SimpleITK images directly, so features are computed from
|
|
4
|
+
the in-memory volume without writing the processed image to disk first.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
import logging
|
|
10
|
+
import os
|
|
11
|
+
import tempfile
|
|
12
|
+
from typing import TYPE_CHECKING, Any, Optional, Union
|
|
13
|
+
|
|
14
|
+
import numpy as np
|
|
15
|
+
|
|
16
|
+
from .validation import Labels
|
|
17
|
+
|
|
18
|
+
if TYPE_CHECKING: # pragma: no cover - import only for type checkers
|
|
19
|
+
from .image import RadiologyImage
|
|
20
|
+
|
|
21
|
+
logger = logging.getLogger(__name__)
|
|
22
|
+
|
|
23
|
+
#: PyRadiomics settings used in the protocol. ``resampledPixelSpacing`` is set
|
|
24
|
+
#: per call from the image dimensionality.
|
|
25
|
+
DEFAULT_RADIOMICS_PARAMS = {
|
|
26
|
+
"imageType": {"Original": {}},
|
|
27
|
+
"setting": {
|
|
28
|
+
"binWidth": 25,
|
|
29
|
+
"resampledPixelSpacing": [1, 1, 1],
|
|
30
|
+
"interpolator": "sitkBSpline",
|
|
31
|
+
"normalize": False,
|
|
32
|
+
"padDistance": 5,
|
|
33
|
+
},
|
|
34
|
+
}
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def default_params(dimensionality: str = "3D") -> dict:
|
|
38
|
+
"""Protocol defaults, with the resampling grid matched to `dimensionality`."""
|
|
39
|
+
params = {
|
|
40
|
+
"imageType": {key: dict(value) for key, value in DEFAULT_RADIOMICS_PARAMS["imageType"].items()},
|
|
41
|
+
"setting": dict(DEFAULT_RADIOMICS_PARAMS["setting"]),
|
|
42
|
+
}
|
|
43
|
+
if dimensionality == "2D":
|
|
44
|
+
params["setting"]["resampledPixelSpacing"] = [1, 1]
|
|
45
|
+
params["setting"]["force2D"] = True
|
|
46
|
+
return params
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def extract_features(
|
|
50
|
+
image: Union["RadiologyImage", Any],
|
|
51
|
+
labels: Labels = (1, 2),
|
|
52
|
+
params: Optional[Union[str, dict]] = None,
|
|
53
|
+
mask: Optional[Any] = None,
|
|
54
|
+
drop_diagnostics: bool = False,
|
|
55
|
+
) -> dict:
|
|
56
|
+
"""Extract features from one image/mask pair.
|
|
57
|
+
|
|
58
|
+
Parameters
|
|
59
|
+
----------
|
|
60
|
+
image:
|
|
61
|
+
A :class:`~ctkit.image.RadiologyImage`, or anything
|
|
62
|
+
:func:`~ctkit.io.load_image` accepts.
|
|
63
|
+
labels:
|
|
64
|
+
Mask values that make up the region of interest. Several values are
|
|
65
|
+
merged into a single region before extraction, so ``(1, 2)`` measures
|
|
66
|
+
organ and tumor together; pass ``2`` for tumor only.
|
|
67
|
+
params:
|
|
68
|
+
Path to a PyRadiomics parameter YAML, or a settings dict. Defaults to
|
|
69
|
+
:func:`default_params`.
|
|
70
|
+
drop_diagnostics:
|
|
71
|
+
Remove the ``diagnostics_*`` provenance entries from the result.
|
|
72
|
+
|
|
73
|
+
Returns
|
|
74
|
+
-------
|
|
75
|
+
dict
|
|
76
|
+
Feature name to value, plus ``series_id`` when the input carries one.
|
|
77
|
+
"""
|
|
78
|
+
from radiomics import featureextractor
|
|
79
|
+
|
|
80
|
+
from .image import RadiologyImage
|
|
81
|
+
from .io import nifti_to_sitk
|
|
82
|
+
|
|
83
|
+
if isinstance(image, RadiologyImage):
|
|
84
|
+
radiology_image = image
|
|
85
|
+
else:
|
|
86
|
+
radiology_image = RadiologyImage(image, mask=mask)
|
|
87
|
+
|
|
88
|
+
if radiology_image.mask is None:
|
|
89
|
+
raise ValueError(
|
|
90
|
+
f"{radiology_image.series_id or 'image'}: radiomics needs a mask defining "
|
|
91
|
+
"the region of interest."
|
|
92
|
+
)
|
|
93
|
+
|
|
94
|
+
dimensionality = "2D" if radiology_image.ndim < 3 else "3D"
|
|
95
|
+
sitk_image = nifti_to_sitk(radiology_image.image)
|
|
96
|
+
sitk_mask, label = _merge_labels(radiology_image, labels)
|
|
97
|
+
|
|
98
|
+
extractor = _build_extractor(
|
|
99
|
+
featureextractor, params if params is not None else default_params(dimensionality)
|
|
100
|
+
)
|
|
101
|
+
|
|
102
|
+
with _quiet_radiomics():
|
|
103
|
+
features = extractor.execute(sitk_image, sitk_mask, label=label)
|
|
104
|
+
|
|
105
|
+
result = {
|
|
106
|
+
key: (float(value) if isinstance(value, (np.floating, np.integer)) else value)
|
|
107
|
+
for key, value in features.items()
|
|
108
|
+
if not (drop_diagnostics and key.startswith("diagnostics_"))
|
|
109
|
+
}
|
|
110
|
+
if radiology_image.series_id:
|
|
111
|
+
result = {"series_id": radiology_image.series_id, **result}
|
|
112
|
+
return result
|
|
113
|
+
|
|
114
|
+
|
|
115
|
+
def _merge_labels(radiology_image: "RadiologyImage", labels: Labels):
|
|
116
|
+
"""Collapse the requested labels into a single region valued 1."""
|
|
117
|
+
from .io import nifti_to_sitk
|
|
118
|
+
|
|
119
|
+
if isinstance(labels, (int, np.integer)):
|
|
120
|
+
wanted = [int(labels)]
|
|
121
|
+
else:
|
|
122
|
+
wanted = [int(value) for value in labels]
|
|
123
|
+
|
|
124
|
+
mask_data = np.rint(radiology_image.mask_array).astype(np.int32)
|
|
125
|
+
present = set(int(value) for value in np.unique(mask_data)) - {0}
|
|
126
|
+
missing = [value for value in wanted if value not in present]
|
|
127
|
+
if missing and len(missing) == len(wanted):
|
|
128
|
+
raise ValueError(
|
|
129
|
+
f"{radiology_image.series_id or 'image'}: none of the labels {wanted} are "
|
|
130
|
+
f"in the mask (it contains {sorted(present) or 'nothing'})."
|
|
131
|
+
)
|
|
132
|
+
if missing:
|
|
133
|
+
logger.debug(
|
|
134
|
+
"%s: labels %s absent from the mask; extracting from %s.",
|
|
135
|
+
radiology_image.series_id or "image", missing,
|
|
136
|
+
[value for value in wanted if value in present],
|
|
137
|
+
)
|
|
138
|
+
|
|
139
|
+
merged = np.isin(mask_data, wanted).astype(np.uint8)
|
|
140
|
+
|
|
141
|
+
import nibabel as nib
|
|
142
|
+
|
|
143
|
+
mask_image = nib.Nifti1Image(merged, radiology_image.mask.affine, radiology_image.mask.header)
|
|
144
|
+
return nifti_to_sitk(mask_image), 1
|
|
145
|
+
|
|
146
|
+
|
|
147
|
+
def _build_extractor(featureextractor, params: Union[str, dict]):
|
|
148
|
+
if isinstance(params, str):
|
|
149
|
+
if not os.path.exists(params):
|
|
150
|
+
raise FileNotFoundError(f"PyRadiomics parameter file not found: {params}")
|
|
151
|
+
return featureextractor.RadiomicsFeatureExtractor(params)
|
|
152
|
+
|
|
153
|
+
import yaml
|
|
154
|
+
|
|
155
|
+
# PyRadiomics only reads a nested imageType/featureClass/setting structure
|
|
156
|
+
# from a file, so a dict is written to a short-lived temporary file.
|
|
157
|
+
with tempfile.TemporaryDirectory(prefix="ctkit_radiomics_") as tmp:
|
|
158
|
+
path = os.path.join(tmp, "params.yaml")
|
|
159
|
+
with open(path, "w") as handle:
|
|
160
|
+
yaml.safe_dump(params, handle, sort_keys=False, default_flow_style=False)
|
|
161
|
+
return featureextractor.RadiomicsFeatureExtractor(path)
|
|
162
|
+
|
|
163
|
+
|
|
164
|
+
class _quiet_radiomics:
|
|
165
|
+
"""Silence PyRadiomics' very chatty per-image logging."""
|
|
166
|
+
|
|
167
|
+
def __enter__(self):
|
|
168
|
+
import radiomics
|
|
169
|
+
|
|
170
|
+
self._level = radiomics.logger.level
|
|
171
|
+
radiomics.logger.setLevel(logging.ERROR)
|
|
172
|
+
return self
|
|
173
|
+
|
|
174
|
+
def __exit__(self, *exc_info) -> None:
|
|
175
|
+
import radiomics
|
|
176
|
+
|
|
177
|
+
radiomics.logger.setLevel(self._level)
|