ctkit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
ctkit/datasets.py ADDED
@@ -0,0 +1,216 @@
1
+ """The catalog of datasets this package knows how to fetch and process.
2
+
3
+ Two groups:
4
+
5
+ * the TCGA and CPTAC collections in
6
+ :data:`~ctkit.constants.tcia_dataset_to_info`, which
7
+ carry curated processing settings (organs to segment, clipping window,
8
+ output dimensions);
9
+ * widely used public CT collections in :data:`EXTRA_CT_COLLECTIONS`, which are
10
+ listed for convenience and download the same way.
11
+
12
+ Any TCIA collection can be downloaded whether or not it appears here — see
13
+ :func:`~ctkit.tcia.list_collections`, which queries
14
+ the archive directly.
15
+ """
16
+
17
+ from __future__ import annotations
18
+
19
+ from typing import Optional
20
+
21
+ from .constants import tcia_dataset_to_info
22
+
23
+ #: Public CT collections that are commonly used as benchmarks or pretraining
24
+ #: corpora. ``clip_min,clip_max`` follows the same convention as the curated
25
+ #: TCGA/CPTAC entries: the window to clip to before feature extraction.
26
+ EXTRA_CT_COLLECTIONS = {
27
+ "lidc-idri": {
28
+ "collection": "LIDC-IDRI",
29
+ "project": "tcia",
30
+ "cancer_organ": "lung",
31
+ "cancer_type": "lung nodules (screening)",
32
+ "description": "1,010 thoracic CT scans with annotated lung nodules.",
33
+ "totalsegmentator_organs": [
34
+ "lung_upper_lobe_left", "lung_lower_lobe_left", "lung_upper_lobe_right",
35
+ "lung_middle_lobe_right", "lung_lower_lobe_right",
36
+ ],
37
+ "clip_min,clip_max": (-1000, 400),
38
+ },
39
+ "nsclc-radiomics": {
40
+ "collection": "NSCLC-Radiomics",
41
+ "project": "tcia",
42
+ "cancer_organ": "lung",
43
+ "cancer_type": "non-small cell lung cancer",
44
+ "description": "422 NSCLC CT scans with manual tumor delineations (Aerts Lung1).",
45
+ "totalsegmentator_organs": [
46
+ "lung_upper_lobe_left", "lung_lower_lobe_left", "lung_upper_lobe_right",
47
+ "lung_middle_lobe_right", "lung_lower_lobe_right",
48
+ ],
49
+ "clip_min,clip_max": (-1000, 400),
50
+ },
51
+ "nsclc-radiogenomics": {
52
+ "collection": "NSCLC Radiogenomics",
53
+ "project": "tcia",
54
+ "cancer_organ": "lung",
55
+ "cancer_type": "non-small cell lung cancer",
56
+ "description": "CT/PET with matched RNA-seq and mutation data.",
57
+ "totalsegmentator_organs": [
58
+ "lung_upper_lobe_left", "lung_lower_lobe_left", "lung_upper_lobe_right",
59
+ "lung_middle_lobe_right", "lung_lower_lobe_right",
60
+ ],
61
+ "clip_min,clip_max": (-1000, 400),
62
+ },
63
+ "pancreas-ct": {
64
+ "collection": "Pancreas-CT",
65
+ "project": "tcia",
66
+ "cancer_organ": "pancreas",
67
+ "cancer_type": "healthy pancreas (segmentation benchmark)",
68
+ "description": "82 contrast-enhanced abdominal CT scans with pancreas masks.",
69
+ "totalsegmentator_organs": ["pancreas"],
70
+ "clip_min,clip_max": (-200, 300),
71
+ },
72
+ "hcc-tace-seg": {
73
+ "collection": "HCC-TACE-Seg",
74
+ "project": "tcia",
75
+ "cancer_organ": "liver",
76
+ "cancer_type": "hepatocellular carcinoma",
77
+ "description": "Liver CT before/after transarterial chemoembolization, with masks.",
78
+ "totalsegmentator_organs": ["liver"],
79
+ "clip_min,clip_max": (-200, 400),
80
+ },
81
+ "colorectal-liver-metastases": {
82
+ "collection": "Colorectal-Liver-Metastases",
83
+ "project": "tcia",
84
+ "cancer_organ": "liver",
85
+ "cancer_type": "colorectal liver metastases",
86
+ "description": "Preoperative liver CT with tumor and liver segmentations.",
87
+ "totalsegmentator_organs": ["liver"],
88
+ "clip_min,clip_max": (-200, 400),
89
+ },
90
+ "stageii-colorectal-ct": {
91
+ "collection": "StageII-Colorectal-CT",
92
+ "project": "tcia",
93
+ "cancer_organ": "colon",
94
+ "cancer_type": "stage II colorectal cancer",
95
+ "description": "Preoperative CT for stage II colorectal cancer.",
96
+ "totalsegmentator_organs": ["colon"],
97
+ "clip_min,clip_max": (-200, 300),
98
+ },
99
+ "lungct-diagnosis": {
100
+ "collection": "LungCT-Diagnosis",
101
+ "project": "tcia",
102
+ "cancer_organ": "lung",
103
+ "cancer_type": "lung adenocarcinoma",
104
+ "description": "Diagnostic chest CT with survival outcomes.",
105
+ "totalsegmentator_organs": [
106
+ "lung_upper_lobe_left", "lung_lower_lobe_left", "lung_upper_lobe_right",
107
+ "lung_middle_lobe_right", "lung_lower_lobe_right",
108
+ ],
109
+ "clip_min,clip_max": (-1000, 400),
110
+ },
111
+ "rider-lung-ct": {
112
+ "collection": "RIDER Lung CT",
113
+ "project": "tcia",
114
+ "cancer_organ": "lung",
115
+ "cancer_type": "non-small cell lung cancer (test-retest)",
116
+ "description": "Same-day repeat CT scans — the standard set for testing "
117
+ "whether a feature is reproducible.",
118
+ "totalsegmentator_organs": [
119
+ "lung_upper_lobe_left", "lung_lower_lobe_left", "lung_upper_lobe_right",
120
+ "lung_middle_lobe_right", "lung_lower_lobe_right",
121
+ ],
122
+ "clip_min,clip_max": (-1000, 400),
123
+ },
124
+ "ct-colonography": {
125
+ "collection": "CT COLONOGRAPHY",
126
+ "project": "tcia",
127
+ "cancer_organ": "colon",
128
+ "cancer_type": "colorectal polyps (screening)",
129
+ "description": "825 CT colonography cases; large and mostly healthy anatomy.",
130
+ "totalsegmentator_organs": ["colon"],
131
+ "clip_min,clip_max": (-1000, 400),
132
+ },
133
+ }
134
+
135
+ #: Extra files that go with a collection but live outside TCIA.
136
+ SUPPLEMENTARY_DOWNLOADS = {
137
+ "tcga-kirc": {
138
+ "segmentations": {
139
+ "url": "https://zenodo.org/records/13244892/files/kidney-ct.zip?download=1",
140
+ "description": "AI and radiologist-reviewed kidney/tumor segmentations "
141
+ "(Scientific Reports, 2024).",
142
+ "filename": "kidney-ct.zip",
143
+ },
144
+ },
145
+ }
146
+
147
+
148
+ def all_datasets() -> dict:
149
+ """Every cataloged dataset, keyed by its short name."""
150
+ catalog = {}
151
+ for key, info in tcia_dataset_to_info.items():
152
+ entry = dict(info)
153
+ entry.setdefault("collection", key.upper())
154
+ entry.setdefault("curated", True)
155
+ catalog[key] = entry
156
+ for key, info in EXTRA_CT_COLLECTIONS.items():
157
+ entry = dict(info)
158
+ entry.setdefault("curated", False)
159
+ catalog[key] = entry
160
+ return catalog
161
+
162
+
163
+ def normalize_name(name: str) -> str:
164
+ """Accept ``TCGA-KIRC``, ``tcga_kirc`` or ``TCGA KIRC`` for the same dataset."""
165
+ return str(name).strip().lower().replace("_", "-").replace(" ", "-")
166
+
167
+
168
+ def get_dataset_info(name: str) -> dict:
169
+ """Look up one dataset. Raises :class:`KeyError` with suggestions."""
170
+ catalog = all_datasets()
171
+ key = normalize_name(name)
172
+ if key in catalog:
173
+ return {"name": key, **catalog[key]}
174
+
175
+ # Allow the TCIA collection name itself, e.g. "NSCLC-Radiomics".
176
+ for candidate, info in catalog.items():
177
+ if normalize_name(info.get("collection", candidate)) == key:
178
+ return {"name": candidate, **info}
179
+
180
+ close = [candidate for candidate in catalog if key in candidate or candidate in key]
181
+ hint = f" Did you mean: {', '.join(sorted(close))}?" if close else ""
182
+ raise KeyError(
183
+ f"Unknown dataset {name!r}.{hint} Use list_datasets() to see the catalog, "
184
+ "or pass any TCIA collection name directly to download()."
185
+ )
186
+
187
+
188
+ def collection_for(name: str) -> str:
189
+ """The TCIA collection name for a dataset (falls back to the name itself)."""
190
+ try:
191
+ return get_dataset_info(name)["collection"]
192
+ except KeyError:
193
+ return str(name)
194
+
195
+
196
+ def list_datasets(project: Optional[str] = None):
197
+ """The catalog as a DataFrame: name, collection, organ, cancer type."""
198
+ import pandas as pd
199
+
200
+ rows = []
201
+ for key, info in sorted(all_datasets().items()):
202
+ rows.append({
203
+ "name": key,
204
+ "collection": info.get("collection", key.upper()),
205
+ "project": info.get("project", ""),
206
+ "organ": info.get("cancer_organ", ""),
207
+ "cancer_type": info.get("cancer_type", ""),
208
+ "curated_settings": bool(info.get("curated")),
209
+ "organs_to_segment": ", ".join(info.get("totalsegmentator_organs") or []),
210
+ "clip_window": str(info.get("clip_min,clip_max", "")),
211
+ "description": info.get("description", ""),
212
+ })
213
+ frame = pd.DataFrame(rows)
214
+ if project:
215
+ frame = frame[frame["project"] == project].reset_index(drop=True)
216
+ return frame
ctkit/features.py ADDED
@@ -0,0 +1,177 @@
1
+ """Radiomic feature extraction with PyRadiomics.
2
+
3
+ PyRadiomics accepts SimpleITK images directly, so features are computed from
4
+ the in-memory volume without writing the processed image to disk first.
5
+ """
6
+
7
+ from __future__ import annotations
8
+
9
+ import logging
10
+ import os
11
+ import tempfile
12
+ from typing import TYPE_CHECKING, Any, Optional, Union
13
+
14
+ import numpy as np
15
+
16
+ from .validation import Labels
17
+
18
+ if TYPE_CHECKING: # pragma: no cover - import only for type checkers
19
+ from .image import RadiologyImage
20
+
21
+ logger = logging.getLogger(__name__)
22
+
23
+ #: PyRadiomics settings used in the protocol. ``resampledPixelSpacing`` is set
24
+ #: per call from the image dimensionality.
25
+ DEFAULT_RADIOMICS_PARAMS = {
26
+ "imageType": {"Original": {}},
27
+ "setting": {
28
+ "binWidth": 25,
29
+ "resampledPixelSpacing": [1, 1, 1],
30
+ "interpolator": "sitkBSpline",
31
+ "normalize": False,
32
+ "padDistance": 5,
33
+ },
34
+ }
35
+
36
+
37
+ def default_params(dimensionality: str = "3D") -> dict:
38
+ """Protocol defaults, with the resampling grid matched to `dimensionality`."""
39
+ params = {
40
+ "imageType": {key: dict(value) for key, value in DEFAULT_RADIOMICS_PARAMS["imageType"].items()},
41
+ "setting": dict(DEFAULT_RADIOMICS_PARAMS["setting"]),
42
+ }
43
+ if dimensionality == "2D":
44
+ params["setting"]["resampledPixelSpacing"] = [1, 1]
45
+ params["setting"]["force2D"] = True
46
+ return params
47
+
48
+
49
+ def extract_features(
50
+ image: Union["RadiologyImage", Any],
51
+ labels: Labels = (1, 2),
52
+ params: Optional[Union[str, dict]] = None,
53
+ mask: Optional[Any] = None,
54
+ drop_diagnostics: bool = False,
55
+ ) -> dict:
56
+ """Extract features from one image/mask pair.
57
+
58
+ Parameters
59
+ ----------
60
+ image:
61
+ A :class:`~ctkit.image.RadiologyImage`, or anything
62
+ :func:`~ctkit.io.load_image` accepts.
63
+ labels:
64
+ Mask values that make up the region of interest. Several values are
65
+ merged into a single region before extraction, so ``(1, 2)`` measures
66
+ organ and tumor together; pass ``2`` for tumor only.
67
+ params:
68
+ Path to a PyRadiomics parameter YAML, or a settings dict. Defaults to
69
+ :func:`default_params`.
70
+ drop_diagnostics:
71
+ Remove the ``diagnostics_*`` provenance entries from the result.
72
+
73
+ Returns
74
+ -------
75
+ dict
76
+ Feature name to value, plus ``series_id`` when the input carries one.
77
+ """
78
+ from radiomics import featureextractor
79
+
80
+ from .image import RadiologyImage
81
+ from .io import nifti_to_sitk
82
+
83
+ if isinstance(image, RadiologyImage):
84
+ radiology_image = image
85
+ else:
86
+ radiology_image = RadiologyImage(image, mask=mask)
87
+
88
+ if radiology_image.mask is None:
89
+ raise ValueError(
90
+ f"{radiology_image.series_id or 'image'}: radiomics needs a mask defining "
91
+ "the region of interest."
92
+ )
93
+
94
+ dimensionality = "2D" if radiology_image.ndim < 3 else "3D"
95
+ sitk_image = nifti_to_sitk(radiology_image.image)
96
+ sitk_mask, label = _merge_labels(radiology_image, labels)
97
+
98
+ extractor = _build_extractor(
99
+ featureextractor, params if params is not None else default_params(dimensionality)
100
+ )
101
+
102
+ with _quiet_radiomics():
103
+ features = extractor.execute(sitk_image, sitk_mask, label=label)
104
+
105
+ result = {
106
+ key: (float(value) if isinstance(value, (np.floating, np.integer)) else value)
107
+ for key, value in features.items()
108
+ if not (drop_diagnostics and key.startswith("diagnostics_"))
109
+ }
110
+ if radiology_image.series_id:
111
+ result = {"series_id": radiology_image.series_id, **result}
112
+ return result
113
+
114
+
115
+ def _merge_labels(radiology_image: "RadiologyImage", labels: Labels):
116
+ """Collapse the requested labels into a single region valued 1."""
117
+ from .io import nifti_to_sitk
118
+
119
+ if isinstance(labels, (int, np.integer)):
120
+ wanted = [int(labels)]
121
+ else:
122
+ wanted = [int(value) for value in labels]
123
+
124
+ mask_data = np.rint(radiology_image.mask_array).astype(np.int32)
125
+ present = set(int(value) for value in np.unique(mask_data)) - {0}
126
+ missing = [value for value in wanted if value not in present]
127
+ if missing and len(missing) == len(wanted):
128
+ raise ValueError(
129
+ f"{radiology_image.series_id or 'image'}: none of the labels {wanted} are "
130
+ f"in the mask (it contains {sorted(present) or 'nothing'})."
131
+ )
132
+ if missing:
133
+ logger.debug(
134
+ "%s: labels %s absent from the mask; extracting from %s.",
135
+ radiology_image.series_id or "image", missing,
136
+ [value for value in wanted if value in present],
137
+ )
138
+
139
+ merged = np.isin(mask_data, wanted).astype(np.uint8)
140
+
141
+ import nibabel as nib
142
+
143
+ mask_image = nib.Nifti1Image(merged, radiology_image.mask.affine, radiology_image.mask.header)
144
+ return nifti_to_sitk(mask_image), 1
145
+
146
+
147
+ def _build_extractor(featureextractor, params: Union[str, dict]):
148
+ if isinstance(params, str):
149
+ if not os.path.exists(params):
150
+ raise FileNotFoundError(f"PyRadiomics parameter file not found: {params}")
151
+ return featureextractor.RadiomicsFeatureExtractor(params)
152
+
153
+ import yaml
154
+
155
+ # PyRadiomics only reads a nested imageType/featureClass/setting structure
156
+ # from a file, so a dict is written to a short-lived temporary file.
157
+ with tempfile.TemporaryDirectory(prefix="ctkit_radiomics_") as tmp:
158
+ path = os.path.join(tmp, "params.yaml")
159
+ with open(path, "w") as handle:
160
+ yaml.safe_dump(params, handle, sort_keys=False, default_flow_style=False)
161
+ return featureextractor.RadiomicsFeatureExtractor(path)
162
+
163
+
164
+ class _quiet_radiomics:
165
+ """Silence PyRadiomics' very chatty per-image logging."""
166
+
167
+ def __enter__(self):
168
+ import radiomics
169
+
170
+ self._level = radiomics.logger.level
171
+ radiomics.logger.setLevel(logging.ERROR)
172
+ return self
173
+
174
+ def __exit__(self, *exc_info) -> None:
175
+ import radiomics
176
+
177
+ radiomics.logger.setLevel(self._level)