dichotomise 2.0.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- dichotomise-2.0.0/LICENSE +21 -0
- dichotomise-2.0.0/PKG-INFO +369 -0
- dichotomise-2.0.0/README.md +339 -0
- dichotomise-2.0.0/pyproject.toml +78 -0
- dichotomise-2.0.0/setup.cfg +4 -0
- dichotomise-2.0.0/src/dichotomise/__init__.py +1 -0
- dichotomise-2.0.0/src/dichotomise/cli.py +414 -0
- dichotomise-2.0.0/src/dichotomise/errors.py +51 -0
- dichotomise-2.0.0/src/dichotomise/pipeline.py +257 -0
- dichotomise-2.0.0/src/dichotomise/pydcm/__init__.py +1 -0
- dichotomise-2.0.0/src/dichotomise/pydcm/identify.py +107 -0
- dichotomise-2.0.0/src/dichotomise/pydcm/names.py +1160 -0
- dichotomise-2.0.0/src/dichotomise/pydcm/naming.py +24 -0
- dichotomise-2.0.0/src/dichotomise/pydcm/policies/custom.json +35 -0
- dichotomise-2.0.0/src/dichotomise/pydcm/policies/full.json +56 -0
- dichotomise-2.0.0/src/dichotomise/pydcm/policies/minimal.json +17 -0
- dichotomise-2.0.0/src/dichotomise/pydcm/policies/retain.json +6 -0
- dichotomise-2.0.0/src/dichotomise/pydcm/policies/standard.json +47 -0
- dichotomise-2.0.0/src/dichotomise/pydcm/read.py +92 -0
- dichotomise-2.0.0/src/dichotomise/pydcm/relabel.py +395 -0
- dichotomise-2.0.0/src/dichotomise/run.py +62 -0
- dichotomise-2.0.0/src/dichotomise/stages/__init__.py +1 -0
- dichotomise-2.0.0/src/dichotomise/stages/audit.py +85 -0
- dichotomise-2.0.0/src/dichotomise/stages/capture.py +80 -0
- dichotomise-2.0.0/src/dichotomise/stages/finalise.py +124 -0
- dichotomise-2.0.0/src/dichotomise/stages/rectify.py +52 -0
- dichotomise-2.0.0/src/dichotomise/stages/reports.py +207 -0
- dichotomise-2.0.0/src/dichotomise/stages/sanitise.py +54 -0
- dichotomise-2.0.0/src/dichotomise/stages/sift.py +73 -0
- dichotomise-2.0.0/src/dichotomise/stages/source_archive.py +33 -0
- dichotomise-2.0.0/src/dichotomise/utils/__init__.py +1 -0
- dichotomise-2.0.0/src/dichotomise/utils/archive.py +61 -0
- dichotomise-2.0.0/src/dichotomise/utils/console.py +29 -0
- dichotomise-2.0.0/src/dichotomise/utils/fs.py +28 -0
- dichotomise-2.0.0/src/dichotomise/utils/text.py +13 -0
- dichotomise-2.0.0/src/dichotomise.egg-info/PKG-INFO +369 -0
- dichotomise-2.0.0/src/dichotomise.egg-info/SOURCES.txt +61 -0
- dichotomise-2.0.0/src/dichotomise.egg-info/dependency_links.txt +1 -0
- dichotomise-2.0.0/src/dichotomise.egg-info/entry_points.txt +2 -0
- dichotomise-2.0.0/src/dichotomise.egg-info/requires.txt +7 -0
- dichotomise-2.0.0/src/dichotomise.egg-info/top_level.txt +1 -0
- dichotomise-2.0.0/tests/test_cli.py +313 -0
- dichotomise-2.0.0/tests/test_pipeline.py +185 -0
- dichotomise-2.0.0/tests/test_pipeline_real_data.py +59 -0
- dichotomise-2.0.0/tests/test_pydcm_identify.py +132 -0
- dichotomise-2.0.0/tests/test_pydcm_names.py +31 -0
- dichotomise-2.0.0/tests/test_pydcm_naming.py +57 -0
- dichotomise-2.0.0/tests/test_pydcm_read.py +55 -0
- dichotomise-2.0.0/tests/test_pydcm_read_patient_name.py +16 -0
- dichotomise-2.0.0/tests/test_pydcm_relabel.py +294 -0
- dichotomise-2.0.0/tests/test_run.py +38 -0
- dichotomise-2.0.0/tests/test_stages_audit.py +84 -0
- dichotomise-2.0.0/tests/test_stages_capture.py +90 -0
- dichotomise-2.0.0/tests/test_stages_finalise.py +66 -0
- dichotomise-2.0.0/tests/test_stages_rectify.py +102 -0
- dichotomise-2.0.0/tests/test_stages_reports.py +222 -0
- dichotomise-2.0.0/tests/test_stages_sanitise.py +83 -0
- dichotomise-2.0.0/tests/test_stages_sift.py +76 -0
- dichotomise-2.0.0/tests/test_stages_source_archive.py +36 -0
- dichotomise-2.0.0/tests/test_utils_archive.py +66 -0
- dichotomise-2.0.0/tests/test_utils_console.py +23 -0
- dichotomise-2.0.0/tests/test_utils_fs.py +34 -0
- dichotomise-2.0.0/tests/test_utils_text.py +19 -0
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Sriranga Kashyap
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,369 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: dichotomise
|
|
3
|
+
Version: 2.0.0
|
|
4
|
+
Summary: Audit, sort, rectify, and archive DICOM exports from Siemens XA60+ systems.
|
|
5
|
+
Author: DIChOtoMise contributors
|
|
6
|
+
License-Expression: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/srikash/dichotomise
|
|
8
|
+
Project-URL: Repository, https://github.com/srikash/dichotomise
|
|
9
|
+
Project-URL: Issues, https://github.com/srikash/dichotomise/issues
|
|
10
|
+
Keywords: DICOM,medical-imaging,Siemens,anonymisation,audit
|
|
11
|
+
Classifier: Development Status :: 4 - Beta
|
|
12
|
+
Classifier: Environment :: Console
|
|
13
|
+
Classifier: Intended Audience :: Science/Research
|
|
14
|
+
Classifier: Programming Language :: Python :: 3
|
|
15
|
+
Classifier: Programming Language :: Python :: 3 :: Only
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
19
|
+
Classifier: Topic :: Scientific/Engineering :: Medical Science Apps.
|
|
20
|
+
Requires-Python: >=3.11
|
|
21
|
+
Description-Content-Type: text/markdown
|
|
22
|
+
License-File: LICENSE
|
|
23
|
+
Requires-Dist: pydicom<4,>=3.0
|
|
24
|
+
Requires-Dist: click<9,>=8.1
|
|
25
|
+
Requires-Dist: rich<15,>=13
|
|
26
|
+
Requires-Dist: rich-click<2,>=1.9
|
|
27
|
+
Provides-Extra: test
|
|
28
|
+
Requires-Dist: pytest<10,>=8; extra == "test"
|
|
29
|
+
Dynamic: license-file
|
|
30
|
+
|
|
31
|
+
# dichotomise [](https://github.com/srikash/dichotomise/releases/tag/v2.0.0) [](LICENSE) [](https://github.com/srikash/dichotomise/actions/workflows/ci.yml)
|
|
32
|
+
<b><ins>DIC</ins></b>h<b><ins>O</ins></b>to<b><ins>M</ins></b>ise is a command-line tool for checking, sorting, renaming,
|
|
33
|
+
de-identifying, and archiving DICOM exports from Siemens XA60+ systems.
|
|
34
|
+
|
|
35
|
+
### The default export
|
|
36
|
+
|
|
37
|
+
For a given study, the XA60+ export places DICOMs from each series into sub-folders named:
|
|
38
|
+
```text
|
|
39
|
+
<ProtocolName>_<SeriesNumber>_MR/
|
|
40
|
+
```
|
|
41
|
+
|
|
42
|
+
The DICOMs inside are named `1.dcm`, `2.dcm`, and so on. The numbering starts again in
|
|
43
|
+
every series folder and recurs across subject exports.
|
|
44
|
+
|
|
45
|
+
```text
|
|
46
|
+
subject-export/
|
|
47
|
+
DWI_21_MR/
|
|
48
|
+
1.dcm
|
|
49
|
+
2.dcm
|
|
50
|
+
fMRI_22_MR/
|
|
51
|
+
1.dcm
|
|
52
|
+
2.dcm
|
|
53
|
+
```
|
|
54
|
+
`1.dcm` in one series may be entirely unrelated to `1.dcm` in another. The
|
|
55
|
+
files can contain different acquisitions and have different file sizes.
|
|
56
|
+
|
|
57
|
+
## Why repeated filenames are positively dreadful
|
|
58
|
+
|
|
59
|
+
#### TL;DR
|
|
60
|
+
|
|
61
|
+
Naming files `1.dcm`, `2.dcm`, `3.dcm` is the worst possible choice. It is a
|
|
62
|
+
regression from something that previously worked well.
|
|
63
|
+
|
|
64
|
+
| Export | File naming |
|
|
65
|
+
|---|---|
|
|
66
|
+
| Older exports | Distinct `.IMA` filenames |
|
|
67
|
+
| XA30 exports | Distinct `.dcm` filenames |
|
|
68
|
+
| XA60+ exports | `1.dcm`, `2.dcm`, … repeated in every series folder |
|
|
69
|
+
|
|
70
|
+
Every DICOM carries a globally unique SOP Instance UID. The XA60 export retains
|
|
71
|
+
that identifier in the header but does not use it to distinguish the exported
|
|
72
|
+
filename. The file's name identifies it only while its surrounding folder
|
|
73
|
+
structure remains intact. That is a data-integrity problem at the point of
|
|
74
|
+
export, before a researcher runs a pipeline or moves a file.
|
|
75
|
+
|
|
76
|
+
FAIR principle **F1** calls for globally unique, persistent identifiers for
|
|
77
|
+
data. The DICOM header supplies one; the exported filename hides it from
|
|
78
|
+
ordinary file operations. BIDS takes the opposite approach to filenames: its
|
|
79
|
+
applicable entities identify the data within the filename itself.
|
|
80
|
+
|
|
81
|
+
### The new export format is the antithesis of good data-handling practice
|
|
82
|
+
|
|
83
|
+
1. **Give each file a usable identity.** XA60+ assigns the same name to
|
|
84
|
+
unrelated DICOMs. A file separated from its folder cannot be identified by
|
|
85
|
+
name, although its SOP Instance UID remains in its header. This undermines
|
|
86
|
+
findability at the file-system level and makes safe handling depend on
|
|
87
|
+
reading DICOM metadata every time.
|
|
88
|
+
|
|
89
|
+
2. **Keep the file and its context in agreement.** The export uses a series
|
|
90
|
+
folder to provide context that the filename lacks. If that folder
|
|
91
|
+
contradicts the DICOM header, the name offers no independent clue.
|
|
92
|
+
|
|
93
|
+
[Example Incident 1](#incident-1-dicom-placed-in-the-wrong-series) shows that this
|
|
94
|
+
disagreement occurred in a single export with no user intervention.
|
|
95
|
+
|
|
96
|
+
3. **Make collisions visible before they cost data.** When exports are
|
|
97
|
+
combined, flattened or restored into one directory, repeated names
|
|
98
|
+
collide. Depending on the operation and its settings, a collision can
|
|
99
|
+
overwrite a file or stop the transfer. Either way, the filename cannot
|
|
100
|
+
distinguish the DICOMs involved. Safe reuse demands an explicit identity
|
|
101
|
+
check, not trust in the exported names.
|
|
102
|
+
|
|
103
|
+
4. **Preserve the evidence needed to audit and reuse data.** Matching
|
|
104
|
+
filenames do not establish that two DICOMs contain the same instance or
|
|
105
|
+
the same scan content. Counters alone cannot verify completeness, detect a
|
|
106
|
+
misplaced instance or establish where a file came from. FAIR **R1.2**
|
|
107
|
+
calls for detailed provenance; losing a file's folder context makes that
|
|
108
|
+
provenance harder to recover from the export, even though metadata remains
|
|
109
|
+
in the DICOM.
|
|
110
|
+
|
|
111
|
+
[Example Incident 2](#incident-2-interleaved-dwi-duplication) shows
|
|
112
|
+
why a plausible filename sequence cannot serve as an audit.
|
|
113
|
+
|
|
114
|
+
### Two example incidents (from amongst several)
|
|
115
|
+
|
|
116
|
+
We noticed inconsistencies in scanner exports and inspected the DICOM metadata
|
|
117
|
+
manually. Both incidents below were present before anyone copied, moved or
|
|
118
|
+
processed the files. They are examples of the intermittent, inconsistent
|
|
119
|
+
errors we have encountered, not an exhaustive list. These occur in product sequences
|
|
120
|
+
and C2Ps alike.
|
|
121
|
+
|
|
122
|
+
#### Incident 1: DICOM placed in the wrong series
|
|
123
|
+
|
|
124
|
+
`300.dcm` appeared in the folder for Series 21. We checked its DICOM Series
|
|
125
|
+
Number and Instance UID and found that the file belonged to Series 22. The
|
|
126
|
+
export's folder contradicted the file's metadata.
|
|
127
|
+
|
|
128
|
+
#### Incident 2: Interleaved DWI duplication
|
|
129
|
+
|
|
130
|
+
We expected 81 DICOMs in a DWI series, but the export contained 162. Manual
|
|
131
|
+
inspection found that duplicates were randomly interleaved with the original
|
|
132
|
+
files, rather than appended as a second sequence. The filename range and file
|
|
133
|
+
count alone could not show which files were duplicated.
|
|
134
|
+
|
|
135
|
+
# dichotomise to the rescue
|
|
136
|
+
|
|
137
|
+
`dichotomise` addresses this failure at ingestion. It reads identity from
|
|
138
|
+
DICOM metadata rather than trusting exported filenames or series folders. It
|
|
139
|
+
preserves the original files, audits for duplicate content and inconsistent
|
|
140
|
+
placement or metadata, separates files needing review, and writes clearly
|
|
141
|
+
named, verified output.
|
|
142
|
+
|
|
143
|
+
The tool is intended for any DICOM export that needs structured auditing,
|
|
144
|
+
sorting and validation. Its immediate motivation is the XA60+ export: a
|
|
145
|
+
globally unique identifier is already inside each DICOM, yet the scanner
|
|
146
|
+
hands researchers filenames that cannot distinguish one file from another
|
|
147
|
+
outside a fragile folder hierarchy.
|
|
148
|
+
|
|
149
|
+
## What it does
|
|
150
|
+
|
|
151
|
+
One command runs the full pipeline, in order:
|
|
152
|
+
|
|
153
|
+
1. **capture** — copies readable DICOM files into a working folder, grouping files
|
|
154
|
+
by `(PatientID, StudyInstanceUID)` from their own headers, never from
|
|
155
|
+
folder names.
|
|
156
|
+
2. **source_archive** — archives each captured study's untouched DICOM files
|
|
157
|
+
with a checksum before further processing.
|
|
158
|
+
3. **audit** — flags duplicate scan content (by comparing everything except
|
|
159
|
+
each file's own unique ID), files sitting in the wrong series folder, and
|
|
160
|
+
inconsistent patient, study, or series metadata within a physical DICOM
|
|
161
|
+
folder. The CLI prints a Rich inventory table, and the reports include a
|
|
162
|
+
spreadsheet-friendly CSV.
|
|
163
|
+
4. **sift** — splits files into `retained` and `review`, physically, so a
|
|
164
|
+
flagged file is never silently included.
|
|
165
|
+
5. **rectify** — copies retained files into a sorted, clearly named tree:
|
|
166
|
+
`<series number>-<series description>/<series>_<series ID>_<instance>_e<echo>.dcm`.
|
|
167
|
+
6. **sanitise** *(optional, `--sanitise`)* — replaces patient identity
|
|
168
|
+
according to a chosen policy; see [Sanitisation](#sanitisation) below.
|
|
169
|
+
7. **finalise** — independently re-reads the finished output tree (not a
|
|
170
|
+
cached record from earlier stages) to catch any corruption introduced by
|
|
171
|
+
copying or renaming, then archives it with a checksum.
|
|
172
|
+
|
|
173
|
+
This is a from-scratch, simplified rewrite of the original `dichotomise`,
|
|
174
|
+
aimed at being easy to read, debug, and extend, including for someone new to
|
|
175
|
+
Python. It implements a single end-to-end pipeline; there is no separate
|
|
176
|
+
expert/stage-by-stage command.
|
|
177
|
+
|
|
178
|
+
## Installation
|
|
179
|
+
|
|
180
|
+
`dichotomise` requires Python 3.11+ and [uv](https://docs.astral.sh/uv/).
|
|
181
|
+
|
|
182
|
+
If you already have Python and `pip`:
|
|
183
|
+
|
|
184
|
+
```bash
|
|
185
|
+
python -m pip install --user uv
|
|
186
|
+
```
|
|
187
|
+
|
|
188
|
+
Otherwise, install uv directly:
|
|
189
|
+
|
|
190
|
+
```bash
|
|
191
|
+
curl -LsSf https://astral.sh/uv/install.sh | sh
|
|
192
|
+
```
|
|
193
|
+
|
|
194
|
+
### Install from PyPI (after release)
|
|
195
|
+
|
|
196
|
+
```bash
|
|
197
|
+
uv tool install --python 3.11 dichotomise
|
|
198
|
+
dichotomise --help
|
|
199
|
+
```
|
|
200
|
+
|
|
201
|
+
This creates an isolated environment for `dichotomise`. If Python 3.11 is not
|
|
202
|
+
available, uv downloads it automatically.
|
|
203
|
+
|
|
204
|
+
### Install from a source checkout
|
|
205
|
+
|
|
206
|
+
```bash
|
|
207
|
+
git clone https://github.com/srikash/dichotomise.git
|
|
208
|
+
cd dichotomise
|
|
209
|
+
uv venv --python 3.11
|
|
210
|
+
uv sync --locked
|
|
211
|
+
uv run dichotomise --help
|
|
212
|
+
```
|
|
213
|
+
|
|
214
|
+
## Usage
|
|
215
|
+
|
|
216
|
+
```bash
|
|
217
|
+
dichotomise --source-dir ./study/sub-001 --out-dir ./dichotomise-runs
|
|
218
|
+
```
|
|
219
|
+
|
|
220
|
+
`--source-dir` accepts either one subject/session export or a scanner export
|
|
221
|
+
containing multiple subjects — they are discovered from DICOM metadata, so
|
|
222
|
+
folder names do not need to be sensible.
|
|
223
|
+
|
|
224
|
+
| Flag | Purpose |
|
|
225
|
+
|---|---|
|
|
226
|
+
| `--source-dir` (required) | Raw DICOM directory to process. |
|
|
227
|
+
| `--out-dir` (required) | Parent directory for the timestamped output folder. |
|
|
228
|
+
| `--sanitise` | Replace patient identity before final archiving. This uses `minimal` unless a policy is chosen. |
|
|
229
|
+
| `--sanitise-policy` | JSON policy name: `minimal` (the default), `standard`, `full`, `retain`, `custom`, or a policy you add yourself. Both `custom` and `custom.json` are accepted. Implies `--sanitise`. |
|
|
230
|
+
| `--subj-id` | First numerical replacement ID. For one study, `6` produces `sub-0006`. With several studies, IDs are enumerated automatically as `sub-0006`, `sub-0007`, `sub-0008`, and so on; the CLI logs a warning. |
|
|
231
|
+
| `--new-id` | Replacement ID for one study. `ADNC0751` becomes `sub-ADNC0751`; an existing `sub-` prefix is retained. |
|
|
232
|
+
| `--random-name` | Generate a random replacement name using the `minimal` sanitisation policy. |
|
|
233
|
+
| `--mapping` | One or more `PatientID:replacement_id` pairs. Repeat the flag or separate pairs with commas. The PatientID must exactly match the DICOM `PatientID`. |
|
|
234
|
+
| `--mapping-file` | JSON object mapping DICOM PatientID to replacement ID, for example `{"source-01": "sub-0001"}`. A full path may be given with or without the `.json` suffix. |
|
|
235
|
+
| `--keep-working-files` | Keep the copied and processed DICOM files (`working/`) instead of deleting them once the archives are verified. |
|
|
236
|
+
|
|
237
|
+
For one study, use `--subj-id`, `--random-name`, or `--new-id`. For several
|
|
238
|
+
studies, use `--subj-id` (automatic enumeration), `--random-name`, `--mapping`,
|
|
239
|
+
or `--mapping-file`; `--new-id` remains intentionally limited to one study.
|
|
240
|
+
|
|
241
|
+
[`docs/example-mapping.json`](docs/example-mapping.json) is a ready-to-copy
|
|
242
|
+
mapping-file example. Its keys must match the source DICOM `PatientID` values;
|
|
243
|
+
its values are the exact replacement labels to write. Its `instructions` field
|
|
244
|
+
is ignored by the CLI.
|
|
245
|
+
|
|
246
|
+
## Output structure
|
|
247
|
+
|
|
248
|
+
Every run creates one timestamped, UTC output folder beneath `--out-dir`:
|
|
249
|
+
|
|
250
|
+
```text
|
|
251
|
+
<run-timestamp>_dichotomise_outputs/
|
|
252
|
+
source/
|
|
253
|
+
<patient-id>_<6char-hex>_source-archive_<run-timestamp>.tar.gz
|
|
254
|
+
<patient-id>_<6char-hex>_source-archive_<run-timestamp>.sha256
|
|
255
|
+
archives/
|
|
256
|
+
<subject-label>_<6char-hex>_dichotomised-archive_<run-timestamp>.tar.gz
|
|
257
|
+
<subject-label>_<6char-hex>_dichotomised-archive_<run-timestamp>.sha256
|
|
258
|
+
working/ # temporary, removed unless --keep-working-files
|
|
259
|
+
reports/
|
|
260
|
+
<subject-label>_<6char-hex>/
|
|
261
|
+
stage-01-report.json
|
|
262
|
+
stage-01-audit.csv
|
|
263
|
+
stage-02-report.json
|
|
264
|
+
stage-03-report.json
|
|
265
|
+
run-status.json # in_progress, complete, or failed
|
|
266
|
+
```
|
|
267
|
+
|
|
268
|
+
`run-status.json` lets you distinguish a complete result from one left by a
|
|
269
|
+
failed or interrupted run. It contains no patient details.
|
|
270
|
+
|
|
271
|
+
A multi-subject run produces one `source/` archive and one `archives/` archive
|
|
272
|
+
per study. Each source archive name uses the original patient ID plus a random
|
|
273
|
+
six-character hexadecimal suffix. The source archive contains untouched DICOM
|
|
274
|
+
files and their original headers, including patient identity; use the
|
|
275
|
+
sanitised `archives/` output for sharing.
|
|
276
|
+
|
|
277
|
+
Processed archive and report names use the study's `<subject-label>` plus a
|
|
278
|
+
random six-character hexadecimal suffix. `<subject-label>` is the real
|
|
279
|
+
`PatientID`, unless `--sanitise` was used, in which case it is the replacement
|
|
280
|
+
label. A sanitised archive's filename and DICOM headers therefore do not
|
|
281
|
+
expose the original identifier.
|
|
282
|
+
|
|
283
|
+
### Reports
|
|
284
|
+
|
|
285
|
+
Each study has a directory under `reports/` containing:
|
|
286
|
+
|
|
287
|
+
- `stage-01-report.json` — audit totals, duplicate and misfiled folders, and
|
|
288
|
+
one summary per physical DICOM folder.
|
|
289
|
+
- `stage-01-audit.csv` — the same per-folder audit summary for spreadsheets,
|
|
290
|
+
including duplicate/misfiled counts and differing metadata fields.
|
|
291
|
+
- `stage-02-report.json` — files retained versus copied to `review/`.
|
|
292
|
+
- `stage-03-report.json` — archive checksum and a `series_inventory` with the
|
|
293
|
+
first original and final DICOM filename for every series.
|
|
294
|
+
|
|
295
|
+
The audit table and CSV flag differences in Patient ID/name, study UID/date/
|
|
296
|
+
time, and series UID/number/description/protocol within a DICOM folder.
|
|
297
|
+
|
|
298
|
+
Compression is always `.tar.gz`.
|
|
299
|
+
|
|
300
|
+
## Sanitisation
|
|
301
|
+
|
|
302
|
+
`--sanitise` applies a named policy — a small JSON file describing what
|
|
303
|
+
happens to each DICOM field: kept, removed, replaced with a fixed or
|
|
304
|
+
run-specific value, given a newly generated identifier (consistent across
|
|
305
|
+
every file for that subject), a birth date scrambled to an approximate but
|
|
306
|
+
different year, or a randomly generated placeholder name.
|
|
307
|
+
|
|
308
|
+
Five policies ship in `src/dichotomise/pydcm/policies/`:
|
|
309
|
+
|
|
310
|
+
| Policy | What happens |
|
|
311
|
+
|---|---|
|
|
312
|
+
| `retain` | Nothing changed. |
|
|
313
|
+
| `standard` | Identity, patient address, accession number, institution, and device-operator fields removed; every UID reissued; birth date scrambled by ±1 year (day/month randomised too); demographic fields (e.g. sex) and all scan-descriptive text (protocol name, series/study description, etc.) kept. |
|
|
314
|
+
| `full` | Everything `standard` does, plus scan-descriptive text and the device serial number also removed. |
|
|
315
|
+
| `minimal` *(the default level)* | Everything `standard` does, but a generated pseudonym is used for both patient name (`Abrahall^Gracious`) and patient ID/output name (`abrahall_gracious`); the birth date is the scan date rather than scrambled. |
|
|
316
|
+
| `custom` | A worked, commented example for building your own — not used automatically. |
|
|
317
|
+
|
|
318
|
+
Full detail — including the real scanner-export comparison these were
|
|
319
|
+
built from, the policy file schema, and how to write your own — is in
|
|
320
|
+
[`docs/sanitise-policies.md`](docs/sanitise-policies.md). In short: copy
|
|
321
|
+
`custom.json` to `<your-policy-name>.json` in the same folder, edit it, and
|
|
322
|
+
run with `--sanitise-policy <your-policy-name>`.
|
|
323
|
+
|
|
324
|
+
## Layout
|
|
325
|
+
|
|
326
|
+
```text
|
|
327
|
+
src/dichotomise/
|
|
328
|
+
cli.py # the `dichotomise` command (Click + Rich)
|
|
329
|
+
pipeline.py # runs every stage in order
|
|
330
|
+
run.py # output paths for one run
|
|
331
|
+
errors.py # error types
|
|
332
|
+
stages/ # one file per pipeline stage
|
|
333
|
+
pydcm/ # DICOM-specific metadata and sanitisation logic
|
|
334
|
+
policies/ # the sanitisation policy JSON files
|
|
335
|
+
names.py # a self-contained adjective+surname placeholder-name generator
|
|
336
|
+
utils/ # generic filesystem/archive/console helpers
|
|
337
|
+
```
|
|
338
|
+
|
|
339
|
+
## Development
|
|
340
|
+
|
|
341
|
+
```bash
|
|
342
|
+
uv run pytest
|
|
343
|
+
uv run ruff format .
|
|
344
|
+
uv run ruff check .
|
|
345
|
+
uv run mypy src
|
|
346
|
+
```
|
|
347
|
+
|
|
348
|
+
`tests/data/` (real, non-synthetic scan exports used for some tests) is
|
|
349
|
+
gitignored and never committed — it may contain identifying information.
|
|
350
|
+
|
|
351
|
+
## The dichotomise workflow
|
|
352
|
+
|
|
353
|
+
```mermaid
|
|
354
|
+
flowchart TD
|
|
355
|
+
SOURCE["Raw scanner export"] --> CAPTURE["capture<br/>Copies readable DICOM into working/"]
|
|
356
|
+
CAPTURE --> ARCHIVE["source_archive<br/>Per-study verified tarball + checksum (source/)"]
|
|
357
|
+
ARCHIVE --> AUDIT["audit<br/>Structural and metadata QA per subject"]
|
|
358
|
+
AUDIT --> REPORT1["stage-01-report.json<br/>stage-01-audit.csv"]
|
|
359
|
+
AUDIT --> SIFT["sift<br/>Splits retained vs review files"]
|
|
360
|
+
SIFT --> REPORT2["stage-02-report.json"]
|
|
361
|
+
SIFT --> RECTIFY["rectify<br/>Metadata-sorted, renamed DICOM tree"]
|
|
362
|
+
RECTIFY --> SANITISE["sanitise (optional, --sanitise)<br/>Replaces patient identity per policy"]
|
|
363
|
+
RECTIFY --> FINALISE["finalise<br/>Re-verifies output, archives it (archives/)"]
|
|
364
|
+
SANITISE --> FINALISE
|
|
365
|
+
FINALISE --> REPORT3["stage-03-report.json"]
|
|
366
|
+
FINALISE --> ARCHIVES["archives/dichotomised-archive.tar.gz"]
|
|
367
|
+
```
|
|
368
|
+
|
|
369
|
+
The project is licensed under the MIT License.
|