ctkit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- ctkit/__init__.py +144 -0
- ctkit/api.py +358 -0
- ctkit/cli.py +754 -0
- ctkit/config.py +453 -0
- ctkit/constants.py +297 -0
- ctkit/dataset.py +1449 -0
- ctkit/datasets.py +216 -0
- ctkit/features.py +177 -0
- ctkit/image.py +1141 -0
- ctkit/io.py +384 -0
- ctkit/metadata.py +233 -0
- ctkit/py.typed +0 -0
- ctkit/qc.py +482 -0
- ctkit/segmentation.py +367 -0
- ctkit/tcia.py +427 -0
- ctkit/validation.py +187 -0
- ctkit-0.1.0.dist-info/METADATA +168 -0
- ctkit-0.1.0.dist-info/RECORD +22 -0
- ctkit-0.1.0.dist-info/WHEEL +5 -0
- ctkit-0.1.0.dist-info/entry_points.txt +2 -0
- ctkit-0.1.0.dist-info/licenses/LICENSE +24 -0
- ctkit-0.1.0.dist-info/top_level.txt +1 -0
ctkit/cli.py
ADDED
|
@@ -0,0 +1,754 @@
|
|
|
1
|
+
"""Command line interface: ``ctkit <command>``.
|
|
2
|
+
|
|
3
|
+
Every command mirrors a piece of the Python API, so a protocol can be run from
|
|
4
|
+
a shell script or a workflow manager without writing Python.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
import argparse
|
|
10
|
+
import json
|
|
11
|
+
import logging
|
|
12
|
+
import os
|
|
13
|
+
import sys
|
|
14
|
+
from typing import Optional, Sequence
|
|
15
|
+
|
|
16
|
+
from . import __version__
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def main(argv: Optional[Sequence[str]] = None) -> int:
|
|
20
|
+
parser = _build_parser()
|
|
21
|
+
args = parser.parse_args(argv)
|
|
22
|
+
|
|
23
|
+
logging.basicConfig(
|
|
24
|
+
level=getattr(logging, args.log_level.upper(), logging.INFO),
|
|
25
|
+
format="%(levelname)s %(name)s: %(message)s",
|
|
26
|
+
)
|
|
27
|
+
|
|
28
|
+
if not getattr(args, "command", None):
|
|
29
|
+
parser.print_help()
|
|
30
|
+
return 1
|
|
31
|
+
|
|
32
|
+
try:
|
|
33
|
+
return args.handler(args)
|
|
34
|
+
except KeyboardInterrupt: # pragma: no cover - interactive
|
|
35
|
+
print("\nInterrupted.", file=sys.stderr)
|
|
36
|
+
return 130
|
|
37
|
+
except Exception as error: # noqa: BLE001 - the CLI reports, it does not trace
|
|
38
|
+
if args.log_level.upper() == "DEBUG":
|
|
39
|
+
raise
|
|
40
|
+
print(f"error: {error}", file=sys.stderr)
|
|
41
|
+
return 1
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
# ----------------------------------------------------------------------
|
|
45
|
+
# parser
|
|
46
|
+
# ----------------------------------------------------------------------
|
|
47
|
+
def _build_parser() -> argparse.ArgumentParser:
|
|
48
|
+
parser = argparse.ArgumentParser(
|
|
49
|
+
prog="ctkit",
|
|
50
|
+
description="Reproducible radiology image processing for AI and radiomics.",
|
|
51
|
+
formatter_class=argparse.RawDescriptionHelpFormatter,
|
|
52
|
+
epilog=(
|
|
53
|
+
"examples:\n"
|
|
54
|
+
" ctkit datasets\n"
|
|
55
|
+
" ctkit download tcga-kirc --out data/raw --limit 20\n"
|
|
56
|
+
" ctkit process data/raw --out data/processed --dataset tcga-kirc\n"
|
|
57
|
+
" ctkit filter data/raw --report qc.csv\n"
|
|
58
|
+
" ctkit radiomics data/processed --out features.csv\n"
|
|
59
|
+
" ctkit info data/processed/TCGA-KN-8424/imaging.nii.gz\n"
|
|
60
|
+
),
|
|
61
|
+
)
|
|
62
|
+
parser.add_argument("--version", action="version", version=f"%(prog)s {__version__}")
|
|
63
|
+
parser.add_argument(
|
|
64
|
+
"--log-level", default="INFO",
|
|
65
|
+
choices=["DEBUG", "INFO", "WARNING", "ERROR"],
|
|
66
|
+
help="verbosity (default: INFO)",
|
|
67
|
+
)
|
|
68
|
+
subparsers = parser.add_subparsers(dest="command")
|
|
69
|
+
|
|
70
|
+
_add_datasets_command(subparsers)
|
|
71
|
+
_add_download_command(subparsers)
|
|
72
|
+
_add_process_command(subparsers)
|
|
73
|
+
_add_filter_command(subparsers)
|
|
74
|
+
_add_radiomics_command(subparsers)
|
|
75
|
+
_add_info_command(subparsers)
|
|
76
|
+
_add_config_command(subparsers)
|
|
77
|
+
return parser
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
def _add_datasets_command(subparsers) -> None:
|
|
81
|
+
parser = subparsers.add_parser(
|
|
82
|
+
"datasets", help="list the datasets this package knows about"
|
|
83
|
+
)
|
|
84
|
+
parser.add_argument("--project", help="restrict to a project, e.g. tcga or cptac")
|
|
85
|
+
parser.add_argument(
|
|
86
|
+
"--tcia", action="store_true",
|
|
87
|
+
help="list every collection available from TCIA (queries the archive)",
|
|
88
|
+
)
|
|
89
|
+
parser.set_defaults(handler=_run_datasets)
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
def _add_download_command(subparsers) -> None:
|
|
93
|
+
parser = subparsers.add_parser("download", help="download a collection from TCIA")
|
|
94
|
+
parser.add_argument("dataset", help="catalog name or TCIA collection name")
|
|
95
|
+
parser.add_argument("--out", required=True, help="destination directory")
|
|
96
|
+
parser.add_argument("--modality", default="CT", help="modality filter (default: CT)")
|
|
97
|
+
parser.add_argument("--limit", type=int, help="stop after this many series")
|
|
98
|
+
parser.add_argument("--patients", nargs="+", help="restrict to these PatientIDs")
|
|
99
|
+
parser.add_argument(
|
|
100
|
+
"--no-convert", action="store_true",
|
|
101
|
+
help="keep DICOM slices instead of converting to NIfTI",
|
|
102
|
+
)
|
|
103
|
+
parser.add_argument(
|
|
104
|
+
"--no-filter", action="store_true",
|
|
105
|
+
help="download every series, including localizers and scouts",
|
|
106
|
+
)
|
|
107
|
+
parser.add_argument("--workers", type=int, default=4, help="parallel downloads")
|
|
108
|
+
parser.add_argument("--overwrite", action="store_true")
|
|
109
|
+
parser.add_argument(
|
|
110
|
+
"--supplementary", metavar="KIND",
|
|
111
|
+
help="also fetch supplementary files, e.g. 'segmentations'",
|
|
112
|
+
)
|
|
113
|
+
parser.set_defaults(handler=_run_download)
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
def _add_process_command(subparsers) -> None:
|
|
117
|
+
parser = subparsers.add_parser(
|
|
118
|
+
"process", help="run the full processing pipeline over a directory of images"
|
|
119
|
+
)
|
|
120
|
+
parser.add_argument("input", help="directory of images (or a single image file)")
|
|
121
|
+
parser.add_argument("--out", required=True, help="output directory")
|
|
122
|
+
parser.add_argument("--config", help="a processing config YAML (see `ctkit config`)")
|
|
123
|
+
parser.add_argument("--dataset", help="use the curated settings for this collection")
|
|
124
|
+
parser.add_argument(
|
|
125
|
+
"--dimensionality", choices=["2D", "3D"], help="3D volumes or the best 2D slice"
|
|
126
|
+
)
|
|
127
|
+
parser.add_argument("--workers", type=int, default=1, help="parallel worker processes")
|
|
128
|
+
parser.add_argument(
|
|
129
|
+
"--layout", choices=["case_dirs", "flat"], default="case_dirs",
|
|
130
|
+
help="output layout (default: case_dirs)",
|
|
131
|
+
)
|
|
132
|
+
parser.add_argument(
|
|
133
|
+
"--format", dest="output_format", choices=["nifti", "numpy"], default="nifti"
|
|
134
|
+
)
|
|
135
|
+
parser.add_argument("--skip-existing", action="store_true", help="resume a partial run")
|
|
136
|
+
parser.add_argument("--qc", action="store_true", help="run quality control first")
|
|
137
|
+
parser.add_argument(
|
|
138
|
+
"--image-pattern", help="glob for image files, e.g. '*_CT.nii.gz'"
|
|
139
|
+
)
|
|
140
|
+
parser.add_argument("--mask-pattern", help="glob for mask files")
|
|
141
|
+
|
|
142
|
+
steps = parser.add_argument_group("pipeline steps (override the config)")
|
|
143
|
+
for name, help_text in [
|
|
144
|
+
("orient", "reorient to canonical RAS"),
|
|
145
|
+
("segment", "segment organs with TotalSegmentator"),
|
|
146
|
+
("clip", "clip the intensity window"),
|
|
147
|
+
("crop-to-content", "crop away the air around the body"),
|
|
148
|
+
("resample", "resample to a common voxel size"),
|
|
149
|
+
("mask", "mask and crop to the region of interest"),
|
|
150
|
+
("standardize-size", "crop/pad to a common array shape"),
|
|
151
|
+
("normalize", "z-score the intensities"),
|
|
152
|
+
]:
|
|
153
|
+
steps.add_argument(f"--{name}", dest=name.replace("-", "_"),
|
|
154
|
+
action="store_true", default=None, help=help_text)
|
|
155
|
+
steps.add_argument(f"--no-{name}", dest=name.replace("-", "_"),
|
|
156
|
+
action="store_false", default=None,
|
|
157
|
+
help=f"skip: {help_text}")
|
|
158
|
+
|
|
159
|
+
steps.add_argument("--clip-min", type=float, help="lower intensity bound (HU)")
|
|
160
|
+
steps.add_argument(
|
|
161
|
+
"--crop-threshold", type=float, metavar="HU",
|
|
162
|
+
help="keep voxels above this when cropping to content "
|
|
163
|
+
"(default: the clipping minimum)",
|
|
164
|
+
)
|
|
165
|
+
steps.add_argument("--clip-max", type=float, help="upper intensity bound (HU)")
|
|
166
|
+
steps.add_argument(
|
|
167
|
+
"--spacing", nargs=3, type=float, metavar=("X", "Y", "Z"),
|
|
168
|
+
help="target voxel spacing in mm",
|
|
169
|
+
)
|
|
170
|
+
steps.add_argument(
|
|
171
|
+
"--shape", nargs=3, type=int, metavar=("X", "Y", "Z"),
|
|
172
|
+
help="target array shape (default: the 95th percentile of the cohort)",
|
|
173
|
+
)
|
|
174
|
+
steps.add_argument("--organs", nargs="+", help="TotalSegmentator structures to segment")
|
|
175
|
+
steps.add_argument(
|
|
176
|
+
"--segmentation-dir", metavar="DIR",
|
|
177
|
+
help="keep the TotalSegmentator files here, one subdirectory per series "
|
|
178
|
+
"(default: a temporary directory, deleted once the masks are read)",
|
|
179
|
+
)
|
|
180
|
+
steps.add_argument(
|
|
181
|
+
"--slice-mode", choices=["mask", "index"],
|
|
182
|
+
help="how 2D mode picks its slice: the one with the most mask (default) "
|
|
183
|
+
"or the one named by --slice-index",
|
|
184
|
+
)
|
|
185
|
+
steps.add_argument(
|
|
186
|
+
"--slice-index", type=int, help="which slice to keep, with --slice-mode index"
|
|
187
|
+
)
|
|
188
|
+
steps.add_argument(
|
|
189
|
+
"--slice-label", nargs="+", type=int, metavar="LABEL",
|
|
190
|
+
help="mask values to measure when picking the slice (default: the mask's "
|
|
191
|
+
"only non-zero value)",
|
|
192
|
+
)
|
|
193
|
+
parser.set_defaults(handler=_run_process)
|
|
194
|
+
|
|
195
|
+
|
|
196
|
+
def _add_filter_command(subparsers) -> None:
|
|
197
|
+
parser = subparsers.add_parser(
|
|
198
|
+
"filter",
|
|
199
|
+
help="run quality control and report which series are usable",
|
|
200
|
+
formatter_class=argparse.RawDescriptionHelpFormatter,
|
|
201
|
+
description=(
|
|
202
|
+
"Decide which series are worth processing, at two levels.\n\n"
|
|
203
|
+
"Metadata checks read headers only — modality, localizer/scout/MIP "
|
|
204
|
+
"keywords, slice thickness, slice count — so they run before any "
|
|
205
|
+
"pixels are read, and on a metadata CSV before anything is even "
|
|
206
|
+
"downloaded. Volume checks read the reconstructed image: 4D series, "
|
|
207
|
+
"too few slices, extreme voxel spacing, in-plane anisotropy that "
|
|
208
|
+
"marks a reformat rather than an axial acquisition.\n\n"
|
|
209
|
+
"INPUT is a directory of images or DICOM series, or a metadata CSV "
|
|
210
|
+
"(in which case only the metadata checks apply)."
|
|
211
|
+
),
|
|
212
|
+
epilog=(
|
|
213
|
+
"examples:\n"
|
|
214
|
+
" ctkit filter data/raw --report qc.csv\n"
|
|
215
|
+
" ctkit filter metadata.csv --metadata-out metadata_filtered.csv\n"
|
|
216
|
+
" ctkit filter data/raw --level metadata --preset radiomics\n"
|
|
217
|
+
" ctkit filter data/raw --metadata metadata.csv --rejected-out data/excluded\n"
|
|
218
|
+
),
|
|
219
|
+
)
|
|
220
|
+
parser.add_argument(
|
|
221
|
+
"input", help="directory of images or DICOM series, or a metadata CSV"
|
|
222
|
+
)
|
|
223
|
+
parser.add_argument(
|
|
224
|
+
"--level", choices=["metadata", "volume", "all"], default="all",
|
|
225
|
+
help="which checks to run (default: all)",
|
|
226
|
+
)
|
|
227
|
+
parser.add_argument(
|
|
228
|
+
"--preset", choices=["default", "radiomics", "permissive"], default="default",
|
|
229
|
+
help="starting thresholds: radiomics also drops sharp kernels, permissive "
|
|
230
|
+
"drops only what cannot be processed at all",
|
|
231
|
+
)
|
|
232
|
+
|
|
233
|
+
outputs = parser.add_argument_group("outputs")
|
|
234
|
+
outputs.add_argument("--report", help="write the full pass/fail table to this CSV")
|
|
235
|
+
outputs.add_argument("--out", help="write the series that pass into this directory")
|
|
236
|
+
outputs.add_argument(
|
|
237
|
+
"--metadata", help="a metadata CSV to annotate with the outcome (directory input)"
|
|
238
|
+
)
|
|
239
|
+
outputs.add_argument(
|
|
240
|
+
"--metadata-out",
|
|
241
|
+
help="where the annotated metadata goes (default: <input>_filtered.csv)",
|
|
242
|
+
)
|
|
243
|
+
outputs.add_argument(
|
|
244
|
+
"--metadata-key", default="series_id",
|
|
245
|
+
help="metadata column holding the series id (default: series_id)",
|
|
246
|
+
)
|
|
247
|
+
outputs.add_argument(
|
|
248
|
+
"--keep-rejected-rows", action="store_true",
|
|
249
|
+
help="annotate the metadata but keep the rows that failed",
|
|
250
|
+
)
|
|
251
|
+
outputs.add_argument(
|
|
252
|
+
"--rejected-out", metavar="DIR",
|
|
253
|
+
help="move the files of the series that fail here",
|
|
254
|
+
)
|
|
255
|
+
outputs.add_argument(
|
|
256
|
+
"--delete-rejected", action="store_true",
|
|
257
|
+
help="delete the files of the series that fail (asks for confirmation)",
|
|
258
|
+
)
|
|
259
|
+
outputs.add_argument(
|
|
260
|
+
"--yes", action="store_true", help="skip the confirmation prompt"
|
|
261
|
+
)
|
|
262
|
+
|
|
263
|
+
checks = parser.add_argument_group("criteria (override the preset)")
|
|
264
|
+
checks.add_argument("--modality", help="required modality, or 'any' (default: CT)")
|
|
265
|
+
checks.add_argument("--min-slices", type=int, help="fewest usable slices (default: 25)")
|
|
266
|
+
checks.add_argument(
|
|
267
|
+
"--max-slice-thickness", type=float, metavar="MM",
|
|
268
|
+
help="thickest acceptable slice from the headers (default: 10)",
|
|
269
|
+
)
|
|
270
|
+
checks.add_argument(
|
|
271
|
+
"--max-spacing", type=float, metavar="MM",
|
|
272
|
+
help="largest acceptable voxel dimension (default: 20)",
|
|
273
|
+
)
|
|
274
|
+
checks.add_argument(
|
|
275
|
+
"--max-anisotropy", type=float,
|
|
276
|
+
help="largest acceptable in-plane spacing ratio (default: 4)",
|
|
277
|
+
)
|
|
278
|
+
checks.add_argument("--allow-4d", action="store_true", help="keep 4D series")
|
|
279
|
+
checks.add_argument(
|
|
280
|
+
"--exclude-keywords", nargs="+", metavar="WORD",
|
|
281
|
+
help="series description keywords that disqualify a series "
|
|
282
|
+
"(default: localizer, scout, topogram, MIP, ...)",
|
|
283
|
+
)
|
|
284
|
+
checks.add_argument(
|
|
285
|
+
"--exclude-sharp-kernels", action="store_true",
|
|
286
|
+
help="also drop B50-B80 and bone kernels, whose noise dominates texture features",
|
|
287
|
+
)
|
|
288
|
+
checks.add_argument(
|
|
289
|
+
"--require-si-axis", action="store_true",
|
|
290
|
+
help="drop series whose thickest axis is not superior-inferior "
|
|
291
|
+
"(coronal/sagittal reformats)",
|
|
292
|
+
)
|
|
293
|
+
parser.set_defaults(handler=_run_filter)
|
|
294
|
+
|
|
295
|
+
|
|
296
|
+
def _add_radiomics_command(subparsers) -> None:
|
|
297
|
+
parser = subparsers.add_parser(
|
|
298
|
+
"radiomics", help="extract PyRadiomics features from processed images"
|
|
299
|
+
)
|
|
300
|
+
parser.add_argument("input", help="directory of processed images with masks")
|
|
301
|
+
parser.add_argument("--out", required=True, help="output CSV")
|
|
302
|
+
parser.add_argument(
|
|
303
|
+
"--labels", nargs="+", type=int, default=[1, 2],
|
|
304
|
+
help="mask labels making up the region of interest (default: 1 2)",
|
|
305
|
+
)
|
|
306
|
+
parser.add_argument("--params", help="a PyRadiomics parameter YAML")
|
|
307
|
+
parser.set_defaults(handler=_run_radiomics)
|
|
308
|
+
|
|
309
|
+
|
|
310
|
+
def _add_info_command(subparsers) -> None:
|
|
311
|
+
parser = subparsers.add_parser("info", help="describe an image or a directory of images")
|
|
312
|
+
parser.add_argument("input")
|
|
313
|
+
parser.add_argument("--mask", help="a mask to describe alongside it")
|
|
314
|
+
parser.add_argument("--json", action="store_true", help="emit JSON")
|
|
315
|
+
parser.set_defaults(handler=_run_info)
|
|
316
|
+
|
|
317
|
+
|
|
318
|
+
def _add_config_command(subparsers) -> None:
|
|
319
|
+
parser = subparsers.add_parser(
|
|
320
|
+
"config", help="print or save a processing config"
|
|
321
|
+
)
|
|
322
|
+
parser.add_argument("--dataset", help="start from the curated settings for a collection")
|
|
323
|
+
parser.add_argument("--preset", choices=["default", "radiomics", "minimal"],
|
|
324
|
+
default="default")
|
|
325
|
+
parser.add_argument("--out", help="write the YAML here instead of printing it")
|
|
326
|
+
parser.set_defaults(handler=_run_config)
|
|
327
|
+
|
|
328
|
+
|
|
329
|
+
# ----------------------------------------------------------------------
|
|
330
|
+
# handlers
|
|
331
|
+
# ----------------------------------------------------------------------
|
|
332
|
+
def _run_datasets(args) -> int:
|
|
333
|
+
if args.tcia:
|
|
334
|
+
from .tcia import list_collections
|
|
335
|
+
|
|
336
|
+
for name in list_collections():
|
|
337
|
+
print(name)
|
|
338
|
+
return 0
|
|
339
|
+
|
|
340
|
+
from .datasets import list_datasets
|
|
341
|
+
|
|
342
|
+
frame = list_datasets(project=args.project)
|
|
343
|
+
columns = ["name", "collection", "organ", "cancer_type", "curated_settings"]
|
|
344
|
+
with _wide_output():
|
|
345
|
+
print(frame[columns].to_string(index=False))
|
|
346
|
+
print(
|
|
347
|
+
f"\n{len(frame)} datasets. Curated entries carry a clipping window, organ list "
|
|
348
|
+
"and output size; the rest use protocol defaults.\n"
|
|
349
|
+
"Any TCIA collection also works: `ctkit datasets --tcia` lists them all."
|
|
350
|
+
)
|
|
351
|
+
return 0
|
|
352
|
+
|
|
353
|
+
|
|
354
|
+
def _run_download(args) -> int:
|
|
355
|
+
from .tcia import download, download_supplementary
|
|
356
|
+
|
|
357
|
+
dataset = download(
|
|
358
|
+
args.dataset,
|
|
359
|
+
args.out,
|
|
360
|
+
modality=args.modality,
|
|
361
|
+
limit=args.limit,
|
|
362
|
+
patients=args.patients,
|
|
363
|
+
convert=not args.no_convert,
|
|
364
|
+
filter_series=not args.no_filter,
|
|
365
|
+
overwrite=args.overwrite,
|
|
366
|
+
workers=args.workers,
|
|
367
|
+
)
|
|
368
|
+
print(f"Downloaded {len(dataset)} series to {args.out}")
|
|
369
|
+
|
|
370
|
+
if args.supplementary:
|
|
371
|
+
path = download_supplementary(args.dataset, args.out, kind=args.supplementary)
|
|
372
|
+
print(f"Supplementary {args.supplementary} in {path}")
|
|
373
|
+
return 0
|
|
374
|
+
|
|
375
|
+
|
|
376
|
+
def _run_process(args) -> int:
|
|
377
|
+
from .dataset import Dataset
|
|
378
|
+
from .image import RadiologyImage
|
|
379
|
+
|
|
380
|
+
config = _config_from_args(args)
|
|
381
|
+
|
|
382
|
+
if os.path.isfile(args.input):
|
|
383
|
+
image = RadiologyImage(args.input).process(config)
|
|
384
|
+
written = image.save(
|
|
385
|
+
args.out, output_format=config.output_format, compress=config.compress
|
|
386
|
+
)
|
|
387
|
+
print(f"Wrote {written}")
|
|
388
|
+
return 0
|
|
389
|
+
|
|
390
|
+
dataset = Dataset.from_directory(
|
|
391
|
+
args.input,
|
|
392
|
+
image_pattern=args.image_pattern,
|
|
393
|
+
mask_pattern=args.mask_pattern,
|
|
394
|
+
)
|
|
395
|
+
print(f"Found {len(dataset)} series in {args.input}")
|
|
396
|
+
print(config.describe())
|
|
397
|
+
|
|
398
|
+
if args.qc:
|
|
399
|
+
dataset = dataset.filter()
|
|
400
|
+
|
|
401
|
+
processed = dataset.process(
|
|
402
|
+
config,
|
|
403
|
+
out_dir=args.out,
|
|
404
|
+
workers=args.workers,
|
|
405
|
+
layout=args.layout,
|
|
406
|
+
skip_existing=args.skip_existing,
|
|
407
|
+
)
|
|
408
|
+
print(f"Processed {len(processed)} of {len(dataset)} series into {args.out}")
|
|
409
|
+
return 0
|
|
410
|
+
|
|
411
|
+
|
|
412
|
+
def _run_filter(args) -> int:
|
|
413
|
+
criteria = _criteria_from_args(args)
|
|
414
|
+
if args.input.lower().endswith((".csv", ".tsv")):
|
|
415
|
+
return _filter_metadata_table(args, criteria)
|
|
416
|
+
return _filter_directory(args, criteria)
|
|
417
|
+
|
|
418
|
+
|
|
419
|
+
def _filter_directory(args, criteria) -> int:
|
|
420
|
+
from .dataset import Dataset, discard
|
|
421
|
+
|
|
422
|
+
dataset = Dataset.from_directory(args.input)
|
|
423
|
+
kept = dataset.filter(criteria, level=args.level)
|
|
424
|
+
report = kept.qc_report
|
|
425
|
+
|
|
426
|
+
_report_outcome(report, len(kept), len(dataset))
|
|
427
|
+
if args.report:
|
|
428
|
+
_write_csv(report, args.report)
|
|
429
|
+
print(f"\nWrote the quality control report to {args.report}")
|
|
430
|
+
|
|
431
|
+
if args.metadata:
|
|
432
|
+
destination = args.metadata_out or _suffixed(args.metadata, "_filtered")
|
|
433
|
+
rows = _annotate_metadata(
|
|
434
|
+
args.metadata, report, destination,
|
|
435
|
+
key=args.metadata_key, drop_rejected=not args.keep_rejected_rows,
|
|
436
|
+
)
|
|
437
|
+
print(f"Wrote {rows} metadata rows to {destination}")
|
|
438
|
+
|
|
439
|
+
if args.out:
|
|
440
|
+
kept.save(args.out)
|
|
441
|
+
print(f"\nWrote the series that passed to {args.out}")
|
|
442
|
+
|
|
443
|
+
rejected = dataset.rejected
|
|
444
|
+
if rejected and (args.rejected_out or args.delete_rejected):
|
|
445
|
+
if args.delete_rejected and not _confirm(
|
|
446
|
+
f"Delete the files of {len(rejected)} rejected series under {args.input}?",
|
|
447
|
+
args.yes,
|
|
448
|
+
):
|
|
449
|
+
print("Left them in place.")
|
|
450
|
+
return 0
|
|
451
|
+
paths = discard(
|
|
452
|
+
rejected,
|
|
453
|
+
destination=None if args.delete_rejected else args.rejected_out,
|
|
454
|
+
delete=args.delete_rejected,
|
|
455
|
+
)
|
|
456
|
+
verb = "Deleted" if args.delete_rejected else f"Moved to {args.rejected_out}:"
|
|
457
|
+
print(f"\n{verb} {len(paths)} paths.")
|
|
458
|
+
return 0
|
|
459
|
+
|
|
460
|
+
|
|
461
|
+
def _filter_metadata_table(args, criteria) -> int:
|
|
462
|
+
"""Filter a metadata CSV — the pass that runs before anything is downloaded."""
|
|
463
|
+
import pandas as pd
|
|
464
|
+
|
|
465
|
+
from .qc import check_series_metadata
|
|
466
|
+
|
|
467
|
+
if args.level == "volume":
|
|
468
|
+
raise ValueError(
|
|
469
|
+
"--level volume needs images to read; give a directory instead of a "
|
|
470
|
+
"metadata CSV, or use --level metadata."
|
|
471
|
+
)
|
|
472
|
+
|
|
473
|
+
frame = pd.read_csv(args.input)
|
|
474
|
+
key = args.metadata_key if args.metadata_key in frame.columns else None
|
|
475
|
+
results = [
|
|
476
|
+
check_series_metadata(
|
|
477
|
+
row, criteria, series_id=None if key is None else str(row[key])
|
|
478
|
+
)
|
|
479
|
+
for _, row in frame.iterrows()
|
|
480
|
+
]
|
|
481
|
+
report = pd.DataFrame([result.to_dict() for result in results])
|
|
482
|
+
if key is not None:
|
|
483
|
+
report["series_id"] = frame[key].astype(str).values
|
|
484
|
+
|
|
485
|
+
passed = report["passed"].to_numpy()
|
|
486
|
+
_report_outcome(report, int(passed.sum()), len(frame))
|
|
487
|
+
if args.report:
|
|
488
|
+
_write_csv(report, args.report)
|
|
489
|
+
print(f"\nWrote the quality control report to {args.report}")
|
|
490
|
+
|
|
491
|
+
destination = args.metadata_out or _suffixed(args.input, "_filtered")
|
|
492
|
+
annotated = frame.drop(columns=["qc_passed", "qc_reason"], errors="ignore").copy()
|
|
493
|
+
annotated["qc_passed"] = passed
|
|
494
|
+
annotated["qc_reason"] = [result.reason for result in results]
|
|
495
|
+
if not args.keep_rejected_rows:
|
|
496
|
+
annotated = annotated[annotated["qc_passed"]]
|
|
497
|
+
_write_csv(annotated, destination)
|
|
498
|
+
print(f"Wrote {len(annotated)} metadata rows to {destination}")
|
|
499
|
+
return 0
|
|
500
|
+
|
|
501
|
+
|
|
502
|
+
def _report_outcome(report, n_kept: int, n_total: int) -> None:
|
|
503
|
+
print(f"{n_kept} of {n_total} series passed quality control.")
|
|
504
|
+
failed = report[~report["passed"].astype(bool)]
|
|
505
|
+
if len(failed):
|
|
506
|
+
columns = [
|
|
507
|
+
column for column in ("series_id", "reason") if column in failed.columns
|
|
508
|
+
]
|
|
509
|
+
print("\nExcluded:")
|
|
510
|
+
with _wide_output():
|
|
511
|
+
print(failed[columns].to_string(index=False))
|
|
512
|
+
|
|
513
|
+
|
|
514
|
+
def _annotate_metadata(
|
|
515
|
+
path: str, report, destination: str, key: str = "series_id",
|
|
516
|
+
drop_rejected: bool = True,
|
|
517
|
+
) -> int:
|
|
518
|
+
"""Merge the pass/fail outcome into a metadata table, keyed by series id.
|
|
519
|
+
|
|
520
|
+
Rows with no matching series in the input are left alone: they were never
|
|
521
|
+
measured, which is not the same as having failed.
|
|
522
|
+
"""
|
|
523
|
+
import pandas as pd
|
|
524
|
+
|
|
525
|
+
frame = pd.read_csv(path)
|
|
526
|
+
if key not in frame.columns:
|
|
527
|
+
raise ValueError(
|
|
528
|
+
f"{path} has no {key!r} column to match series on "
|
|
529
|
+
f"(columns: {', '.join(map(str, frame.columns))}). Use --metadata-key."
|
|
530
|
+
)
|
|
531
|
+
|
|
532
|
+
outcome = report[["series_id", "passed", "reason"]].rename(
|
|
533
|
+
columns={"series_id": key, "passed": "qc_passed", "reason": "qc_reason"}
|
|
534
|
+
)
|
|
535
|
+
outcome[key] = outcome[key].astype(str)
|
|
536
|
+
|
|
537
|
+
merged = frame.drop(columns=["qc_passed", "qc_reason"], errors="ignore").copy()
|
|
538
|
+
merged[key] = merged[key].astype(str)
|
|
539
|
+
merged = merged.merge(outcome, on=key, how="left")
|
|
540
|
+
|
|
541
|
+
unmatched = int(merged["qc_passed"].isna().sum())
|
|
542
|
+
if unmatched:
|
|
543
|
+
print(
|
|
544
|
+
f"{unmatched} metadata rows had no matching series in the input; "
|
|
545
|
+
"they were not checked and are kept."
|
|
546
|
+
)
|
|
547
|
+
if drop_rejected:
|
|
548
|
+
merged = merged[merged["qc_passed"].ne(False)] # unmeasured rows stay
|
|
549
|
+
|
|
550
|
+
_write_csv(merged, destination)
|
|
551
|
+
return len(merged)
|
|
552
|
+
|
|
553
|
+
|
|
554
|
+
def _criteria_from_args(args):
|
|
555
|
+
from .qc import QCCriteria
|
|
556
|
+
|
|
557
|
+
if args.preset == "radiomics":
|
|
558
|
+
criteria = QCCriteria.for_radiomics()
|
|
559
|
+
elif args.preset == "permissive":
|
|
560
|
+
criteria = QCCriteria.permissive()
|
|
561
|
+
else:
|
|
562
|
+
criteria = QCCriteria()
|
|
563
|
+
|
|
564
|
+
if args.modality is not None:
|
|
565
|
+
criteria.modality = None if args.modality.lower() == "any" else args.modality
|
|
566
|
+
for name, value in (
|
|
567
|
+
("min_slices", args.min_slices),
|
|
568
|
+
("max_slice_thickness", args.max_slice_thickness),
|
|
569
|
+
("max_spacing", args.max_spacing),
|
|
570
|
+
("max_in_plane_anisotropy", args.max_anisotropy),
|
|
571
|
+
):
|
|
572
|
+
if value is not None:
|
|
573
|
+
setattr(criteria, name, value)
|
|
574
|
+
if args.exclude_keywords:
|
|
575
|
+
criteria.exclude_keywords = tuple(args.exclude_keywords)
|
|
576
|
+
if args.exclude_sharp_kernels:
|
|
577
|
+
criteria.exclude_sharp_kernels = True
|
|
578
|
+
if args.allow_4d:
|
|
579
|
+
criteria.reject_4d = False
|
|
580
|
+
if args.require_si_axis:
|
|
581
|
+
criteria.require_thickest_axis_is_superior_inferior = True
|
|
582
|
+
return criteria
|
|
583
|
+
|
|
584
|
+
|
|
585
|
+
def _confirm(question: str, assume_yes: bool = False) -> bool:
|
|
586
|
+
if assume_yes:
|
|
587
|
+
return True
|
|
588
|
+
if not sys.stdin.isatty():
|
|
589
|
+
print(
|
|
590
|
+
f"{question} Refusing to delete without a terminal to ask in; "
|
|
591
|
+
"pass --yes to go ahead.",
|
|
592
|
+
file=sys.stderr,
|
|
593
|
+
)
|
|
594
|
+
return False
|
|
595
|
+
return input(f"{question} [y/N] ").strip().lower() in ("y", "yes")
|
|
596
|
+
|
|
597
|
+
|
|
598
|
+
def _write_csv(frame, path: str) -> None:
|
|
599
|
+
directory = os.path.dirname(os.path.abspath(path))
|
|
600
|
+
os.makedirs(directory, exist_ok=True)
|
|
601
|
+
frame.to_csv(path, index=False)
|
|
602
|
+
|
|
603
|
+
|
|
604
|
+
def _suffixed(path: str, suffix: str) -> str:
|
|
605
|
+
stem, extension = os.path.splitext(path)
|
|
606
|
+
return f"{stem}{suffix}{extension or '.csv'}"
|
|
607
|
+
|
|
608
|
+
|
|
609
|
+
def _run_radiomics(args) -> int:
|
|
610
|
+
from .dataset import Dataset
|
|
611
|
+
|
|
612
|
+
dataset = Dataset.from_directory(args.input)
|
|
613
|
+
frame = dataset.radiomics(labels=args.labels, params=args.params, out_csv=args.out)
|
|
614
|
+
feature_columns = [
|
|
615
|
+
column for column in frame.columns if not column.startswith("diagnostics_")
|
|
616
|
+
]
|
|
617
|
+
print(
|
|
618
|
+
f"Extracted {len(feature_columns) - 1} features for {len(frame)} series "
|
|
619
|
+
f"into {args.out}"
|
|
620
|
+
)
|
|
621
|
+
return 0
|
|
622
|
+
|
|
623
|
+
|
|
624
|
+
def _run_info(args) -> int:
|
|
625
|
+
from .dataset import Dataset
|
|
626
|
+
from .image import RadiologyImage
|
|
627
|
+
|
|
628
|
+
if os.path.isdir(args.input) and not _looks_like_dicom_dir(args.input):
|
|
629
|
+
dataset = Dataset.from_directory(args.input)
|
|
630
|
+
frame = dataset.statistics()
|
|
631
|
+
if args.json:
|
|
632
|
+
print(frame.to_json(orient="records", indent=2))
|
|
633
|
+
else:
|
|
634
|
+
with _wide_output():
|
|
635
|
+
print(frame.to_string(index=False))
|
|
636
|
+
return 0
|
|
637
|
+
|
|
638
|
+
image = RadiologyImage(args.input, mask=args.mask)
|
|
639
|
+
stats = image.statistics()
|
|
640
|
+
result = image.check()
|
|
641
|
+
stats["quality_control"] = "pass" if result.passed else f"fail: {result.reason}"
|
|
642
|
+
|
|
643
|
+
if args.json:
|
|
644
|
+
print(json.dumps(stats, indent=2, default=str))
|
|
645
|
+
else:
|
|
646
|
+
width = max(len(key) for key in stats)
|
|
647
|
+
for key, value in stats.items():
|
|
648
|
+
print(f"{key:<{width}} {value}")
|
|
649
|
+
return 0
|
|
650
|
+
|
|
651
|
+
|
|
652
|
+
def _run_config(args) -> int:
|
|
653
|
+
from .config import ProcessingConfig
|
|
654
|
+
|
|
655
|
+
if args.preset == "radiomics":
|
|
656
|
+
config = ProcessingConfig.radiomics(args.dataset)
|
|
657
|
+
elif args.preset == "minimal":
|
|
658
|
+
config = ProcessingConfig.minimal()
|
|
659
|
+
elif args.dataset:
|
|
660
|
+
config = ProcessingConfig.for_dataset(args.dataset)
|
|
661
|
+
else:
|
|
662
|
+
config = ProcessingConfig()
|
|
663
|
+
|
|
664
|
+
if args.out:
|
|
665
|
+
config.to_yaml(args.out)
|
|
666
|
+
print(f"Wrote {args.out}")
|
|
667
|
+
else:
|
|
668
|
+
print(f"# {config.describe()}".replace("\n", "\n# "))
|
|
669
|
+
print(config.to_yaml())
|
|
670
|
+
return 0
|
|
671
|
+
|
|
672
|
+
|
|
673
|
+
# ----------------------------------------------------------------------
|
|
674
|
+
# helpers
|
|
675
|
+
# ----------------------------------------------------------------------
|
|
676
|
+
def _config_from_args(args):
|
|
677
|
+
from .config import ProcessingConfig
|
|
678
|
+
|
|
679
|
+
if args.config:
|
|
680
|
+
config = ProcessingConfig.load(args.config)
|
|
681
|
+
elif args.dataset:
|
|
682
|
+
config = ProcessingConfig.for_dataset(args.dataset)
|
|
683
|
+
else:
|
|
684
|
+
config = ProcessingConfig()
|
|
685
|
+
|
|
686
|
+
overrides = {}
|
|
687
|
+
for name in (
|
|
688
|
+
"orient", "segment", "clip", "crop_to_content", "resample", "mask",
|
|
689
|
+
"normalize", "standardize_size",
|
|
690
|
+
):
|
|
691
|
+
value = getattr(args, name, None)
|
|
692
|
+
if value is not None:
|
|
693
|
+
overrides[name] = value
|
|
694
|
+
if args.dimensionality:
|
|
695
|
+
overrides["dimensionality"] = args.dimensionality
|
|
696
|
+
if args.slice_mode:
|
|
697
|
+
overrides["slice_selection_mode"] = args.slice_mode
|
|
698
|
+
if args.slice_index is not None:
|
|
699
|
+
overrides["slice_index"] = args.slice_index
|
|
700
|
+
overrides.setdefault("slice_selection_mode", "index")
|
|
701
|
+
if args.slice_label:
|
|
702
|
+
overrides["slice_selection_label"] = (
|
|
703
|
+
args.slice_label[0] if len(args.slice_label) == 1 else list(args.slice_label)
|
|
704
|
+
)
|
|
705
|
+
if args.clip_min is not None:
|
|
706
|
+
overrides["clip_min"] = args.clip_min
|
|
707
|
+
if args.crop_threshold is not None:
|
|
708
|
+
overrides["crop_content_threshold"] = args.crop_threshold
|
|
709
|
+
overrides.setdefault("crop_to_content", True)
|
|
710
|
+
if args.clip_max is not None:
|
|
711
|
+
overrides["clip_max"] = args.clip_max
|
|
712
|
+
if args.spacing:
|
|
713
|
+
overrides["target_spacing"] = tuple(args.spacing)
|
|
714
|
+
if args.shape:
|
|
715
|
+
overrides["target_shape"] = tuple(args.shape)
|
|
716
|
+
if args.organs:
|
|
717
|
+
overrides["organs"] = list(args.organs)
|
|
718
|
+
overrides.setdefault("segment", True)
|
|
719
|
+
if args.segmentation_dir:
|
|
720
|
+
overrides["segmentation_dir"] = args.segmentation_dir
|
|
721
|
+
overrides.setdefault("segment", True)
|
|
722
|
+
if args.output_format:
|
|
723
|
+
overrides["output_format"] = args.output_format
|
|
724
|
+
|
|
725
|
+
return config.replace(**overrides) if overrides else config
|
|
726
|
+
|
|
727
|
+
|
|
728
|
+
def _looks_like_dicom_dir(path: str) -> bool:
|
|
729
|
+
for _, _, names in os.walk(path):
|
|
730
|
+
if any(name.lower().endswith((".dcm", ".ima")) for name in names):
|
|
731
|
+
return True
|
|
732
|
+
return False
|
|
733
|
+
|
|
734
|
+
|
|
735
|
+
class _wide_output:
|
|
736
|
+
"""Let pandas use the full terminal width for a moment."""
|
|
737
|
+
|
|
738
|
+
def __enter__(self):
|
|
739
|
+
import pandas as pd
|
|
740
|
+
|
|
741
|
+
self._context = pd.option_context(
|
|
742
|
+
"display.max_rows", 200,
|
|
743
|
+
"display.max_colwidth", 60,
|
|
744
|
+
"display.width", 200,
|
|
745
|
+
)
|
|
746
|
+
self._context.__enter__()
|
|
747
|
+
return self
|
|
748
|
+
|
|
749
|
+
def __exit__(self, *exc_info):
|
|
750
|
+
self._context.__exit__(*exc_info)
|
|
751
|
+
|
|
752
|
+
|
|
753
|
+
if __name__ == "__main__": # pragma: no cover
|
|
754
|
+
sys.exit(main())
|