dichotomise 2.0.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (63) hide show
  1. dichotomise-2.0.0/LICENSE +21 -0
  2. dichotomise-2.0.0/PKG-INFO +369 -0
  3. dichotomise-2.0.0/README.md +339 -0
  4. dichotomise-2.0.0/pyproject.toml +78 -0
  5. dichotomise-2.0.0/setup.cfg +4 -0
  6. dichotomise-2.0.0/src/dichotomise/__init__.py +1 -0
  7. dichotomise-2.0.0/src/dichotomise/cli.py +414 -0
  8. dichotomise-2.0.0/src/dichotomise/errors.py +51 -0
  9. dichotomise-2.0.0/src/dichotomise/pipeline.py +257 -0
  10. dichotomise-2.0.0/src/dichotomise/pydcm/__init__.py +1 -0
  11. dichotomise-2.0.0/src/dichotomise/pydcm/identify.py +107 -0
  12. dichotomise-2.0.0/src/dichotomise/pydcm/names.py +1160 -0
  13. dichotomise-2.0.0/src/dichotomise/pydcm/naming.py +24 -0
  14. dichotomise-2.0.0/src/dichotomise/pydcm/policies/custom.json +35 -0
  15. dichotomise-2.0.0/src/dichotomise/pydcm/policies/full.json +56 -0
  16. dichotomise-2.0.0/src/dichotomise/pydcm/policies/minimal.json +17 -0
  17. dichotomise-2.0.0/src/dichotomise/pydcm/policies/retain.json +6 -0
  18. dichotomise-2.0.0/src/dichotomise/pydcm/policies/standard.json +47 -0
  19. dichotomise-2.0.0/src/dichotomise/pydcm/read.py +92 -0
  20. dichotomise-2.0.0/src/dichotomise/pydcm/relabel.py +395 -0
  21. dichotomise-2.0.0/src/dichotomise/run.py +62 -0
  22. dichotomise-2.0.0/src/dichotomise/stages/__init__.py +1 -0
  23. dichotomise-2.0.0/src/dichotomise/stages/audit.py +85 -0
  24. dichotomise-2.0.0/src/dichotomise/stages/capture.py +80 -0
  25. dichotomise-2.0.0/src/dichotomise/stages/finalise.py +124 -0
  26. dichotomise-2.0.0/src/dichotomise/stages/rectify.py +52 -0
  27. dichotomise-2.0.0/src/dichotomise/stages/reports.py +207 -0
  28. dichotomise-2.0.0/src/dichotomise/stages/sanitise.py +54 -0
  29. dichotomise-2.0.0/src/dichotomise/stages/sift.py +73 -0
  30. dichotomise-2.0.0/src/dichotomise/stages/source_archive.py +33 -0
  31. dichotomise-2.0.0/src/dichotomise/utils/__init__.py +1 -0
  32. dichotomise-2.0.0/src/dichotomise/utils/archive.py +61 -0
  33. dichotomise-2.0.0/src/dichotomise/utils/console.py +29 -0
  34. dichotomise-2.0.0/src/dichotomise/utils/fs.py +28 -0
  35. dichotomise-2.0.0/src/dichotomise/utils/text.py +13 -0
  36. dichotomise-2.0.0/src/dichotomise.egg-info/PKG-INFO +369 -0
  37. dichotomise-2.0.0/src/dichotomise.egg-info/SOURCES.txt +61 -0
  38. dichotomise-2.0.0/src/dichotomise.egg-info/dependency_links.txt +1 -0
  39. dichotomise-2.0.0/src/dichotomise.egg-info/entry_points.txt +2 -0
  40. dichotomise-2.0.0/src/dichotomise.egg-info/requires.txt +7 -0
  41. dichotomise-2.0.0/src/dichotomise.egg-info/top_level.txt +1 -0
  42. dichotomise-2.0.0/tests/test_cli.py +313 -0
  43. dichotomise-2.0.0/tests/test_pipeline.py +185 -0
  44. dichotomise-2.0.0/tests/test_pipeline_real_data.py +59 -0
  45. dichotomise-2.0.0/tests/test_pydcm_identify.py +132 -0
  46. dichotomise-2.0.0/tests/test_pydcm_names.py +31 -0
  47. dichotomise-2.0.0/tests/test_pydcm_naming.py +57 -0
  48. dichotomise-2.0.0/tests/test_pydcm_read.py +55 -0
  49. dichotomise-2.0.0/tests/test_pydcm_read_patient_name.py +16 -0
  50. dichotomise-2.0.0/tests/test_pydcm_relabel.py +294 -0
  51. dichotomise-2.0.0/tests/test_run.py +38 -0
  52. dichotomise-2.0.0/tests/test_stages_audit.py +84 -0
  53. dichotomise-2.0.0/tests/test_stages_capture.py +90 -0
  54. dichotomise-2.0.0/tests/test_stages_finalise.py +66 -0
  55. dichotomise-2.0.0/tests/test_stages_rectify.py +102 -0
  56. dichotomise-2.0.0/tests/test_stages_reports.py +222 -0
  57. dichotomise-2.0.0/tests/test_stages_sanitise.py +83 -0
  58. dichotomise-2.0.0/tests/test_stages_sift.py +76 -0
  59. dichotomise-2.0.0/tests/test_stages_source_archive.py +36 -0
  60. dichotomise-2.0.0/tests/test_utils_archive.py +66 -0
  61. dichotomise-2.0.0/tests/test_utils_console.py +23 -0
  62. dichotomise-2.0.0/tests/test_utils_fs.py +34 -0
  63. dichotomise-2.0.0/tests/test_utils_text.py +19 -0
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Sriranga Kashyap
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,369 @@
1
+ Metadata-Version: 2.4
2
+ Name: dichotomise
3
+ Version: 2.0.0
4
+ Summary: Audit, sort, rectify, and archive DICOM exports from Siemens XA60+ systems.
5
+ Author: DIChOtoMise contributors
6
+ License-Expression: MIT
7
+ Project-URL: Homepage, https://github.com/srikash/dichotomise
8
+ Project-URL: Repository, https://github.com/srikash/dichotomise
9
+ Project-URL: Issues, https://github.com/srikash/dichotomise/issues
10
+ Keywords: DICOM,medical-imaging,Siemens,anonymisation,audit
11
+ Classifier: Development Status :: 4 - Beta
12
+ Classifier: Environment :: Console
13
+ Classifier: Intended Audience :: Science/Research
14
+ Classifier: Programming Language :: Python :: 3
15
+ Classifier: Programming Language :: Python :: 3 :: Only
16
+ Classifier: Programming Language :: Python :: 3.11
17
+ Classifier: Programming Language :: Python :: 3.12
18
+ Classifier: Programming Language :: Python :: 3.13
19
+ Classifier: Topic :: Scientific/Engineering :: Medical Science Apps.
20
+ Requires-Python: >=3.11
21
+ Description-Content-Type: text/markdown
22
+ License-File: LICENSE
23
+ Requires-Dist: pydicom<4,>=3.0
24
+ Requires-Dist: click<9,>=8.1
25
+ Requires-Dist: rich<15,>=13
26
+ Requires-Dist: rich-click<2,>=1.9
27
+ Provides-Extra: test
28
+ Requires-Dist: pytest<10,>=8; extra == "test"
29
+ Dynamic: license-file
30
+
31
+ # dichotomise [![Version](https://img.shields.io/badge/version-2.0.0-purple.svg)](https://github.com/srikash/dichotomise/releases/tag/v2.0.0) [![License](https://img.shields.io/badge/license-MIT-orange.svg)](LICENSE) [![CI](https://github.com/srikash/dichotomise/actions/workflows/ci.yml/badge.svg?branch=main)](https://github.com/srikash/dichotomise/actions/workflows/ci.yml)
32
+ <b><ins>DIC</ins></b>h<b><ins>O</ins></b>to<b><ins>M</ins></b>ise is a command-line tool for checking, sorting, renaming,
33
+ de-identifying, and archiving DICOM exports from Siemens XA60+ systems.
34
+
35
+ ### The default export
36
+
37
+ For a given study, the XA60+ export places DICOMs from each series into sub-folders named:
38
+ ```text
39
+ <ProtocolName>_<SeriesNumber>_MR/
40
+ ```
41
+
42
+ The DICOMs inside are named `1.dcm`, `2.dcm`, and so on. The numbering starts again in
43
+ every series folder and recurs across subject exports.
44
+
45
+ ```text
46
+ subject-export/
47
+ DWI_21_MR/
48
+ 1.dcm
49
+ 2.dcm
50
+ fMRI_22_MR/
51
+ 1.dcm
52
+ 2.dcm
53
+ ```
54
+ `1.dcm` in one series may be entirely unrelated to `1.dcm` in another. The
55
+ files can contain different acquisitions and have different file sizes.
56
+
57
+ ## Why repeated filenames are positively dreadful
58
+
59
+ #### TL;DR
60
+
61
+ Naming files `1.dcm`, `2.dcm`, `3.dcm` is the worst possible choice. It is a
62
+ regression from something that previously worked well.
63
+
64
+ | Export | File naming |
65
+ |---|---|
66
+ | Older exports | Distinct `.IMA` filenames |
67
+ | XA30 exports | Distinct `.dcm` filenames |
68
+ | XA60+ exports | `1.dcm`, `2.dcm`, … repeated in every series folder |
69
+
70
+ Every DICOM carries a globally unique SOP Instance UID. The XA60 export retains
71
+ that identifier in the header but does not use it to distinguish the exported
72
+ filename. The file's name identifies it only while its surrounding folder
73
+ structure remains intact. That is a data-integrity problem at the point of
74
+ export, before a researcher runs a pipeline or moves a file.
75
+
76
+ FAIR principle **F1** calls for globally unique, persistent identifiers for
77
+ data. The DICOM header supplies one; the exported filename hides it from
78
+ ordinary file operations. BIDS takes the opposite approach to filenames: its
79
+ applicable entities identify the data within the filename itself.
80
+
81
+ ### The new export format is the antithesis of good data-handling practice
82
+
83
+ 1. **Give each file a usable identity.** XA60+ assigns the same name to
84
+ unrelated DICOMs. A file separated from its folder cannot be identified by
85
+ name, although its SOP Instance UID remains in its header. This undermines
86
+ findability at the file-system level and makes safe handling depend on
87
+ reading DICOM metadata every time.
88
+
89
+ 2. **Keep the file and its context in agreement.** The export uses a series
90
+ folder to provide context that the filename lacks. If that folder
91
+ contradicts the DICOM header, the name offers no independent clue.
92
+
93
+ [Example Incident 1](#incident-1-dicom-placed-in-the-wrong-series) shows that this
94
+ disagreement occurred in a single export with no user intervention.
95
+
96
+ 3. **Make collisions visible before they cost data.** When exports are
97
+ combined, flattened or restored into one directory, repeated names
98
+ collide. Depending on the operation and its settings, a collision can
99
+ overwrite a file or stop the transfer. Either way, the filename cannot
100
+ distinguish the DICOMs involved. Safe reuse demands an explicit identity
101
+ check, not trust in the exported names.
102
+
103
+ 4. **Preserve the evidence needed to audit and reuse data.** Matching
104
+ filenames do not establish that two DICOMs contain the same instance or
105
+ the same scan content. Counters alone cannot verify completeness, detect a
106
+ misplaced instance or establish where a file came from. FAIR **R1.2**
107
+ calls for detailed provenance; losing a file's folder context makes that
108
+ provenance harder to recover from the export, even though metadata remains
109
+ in the DICOM.
110
+
111
+ [Example Incident 2](#incident-2-interleaved-dwi-duplication) shows
112
+ why a plausible filename sequence cannot serve as an audit.
113
+
114
+ ### Two example incidents (from amongst several)
115
+
116
+ We noticed inconsistencies in scanner exports and inspected the DICOM metadata
117
+ manually. Both incidents below were present before anyone copied, moved or
118
+ processed the files. They are examples of the intermittent, inconsistent
119
+ errors we have encountered, not an exhaustive list. These occur in product sequences
120
+ and C2Ps alike.
121
+
122
+ #### Incident 1: DICOM placed in the wrong series
123
+
124
+ `300.dcm` appeared in the folder for Series 21. We checked its DICOM Series
125
+ Number and Instance UID and found that the file belonged to Series 22. The
126
+ export's folder contradicted the file's metadata.
127
+
128
+ #### Incident 2: Interleaved DWI duplication
129
+
130
+ We expected 81 DICOMs in a DWI series, but the export contained 162. Manual
131
+ inspection found that duplicates were randomly interleaved with the original
132
+ files, rather than appended as a second sequence. The filename range and file
133
+ count alone could not show which files were duplicated.
134
+
135
+ # dichotomise to the rescue
136
+
137
+ `dichotomise` addresses this failure at ingestion. It reads identity from
138
+ DICOM metadata rather than trusting exported filenames or series folders. It
139
+ preserves the original files, audits for duplicate content and inconsistent
140
+ placement or metadata, separates files needing review, and writes clearly
141
+ named, verified output.
142
+
143
+ The tool is intended for any DICOM export that needs structured auditing,
144
+ sorting and validation. Its immediate motivation is the XA60+ export: a
145
+ globally unique identifier is already inside each DICOM, yet the scanner
146
+ hands researchers filenames that cannot distinguish one file from another
147
+ outside a fragile folder hierarchy.
148
+
149
+ ## What it does
150
+
151
+ One command runs the full pipeline, in order:
152
+
153
+ 1. **capture** — copies readable DICOM files into a working folder, grouping files
154
+ by `(PatientID, StudyInstanceUID)` from their own headers, never from
155
+ folder names.
156
+ 2. **source_archive** — archives each captured study's untouched DICOM files
157
+ with a checksum before further processing.
158
+ 3. **audit** — flags duplicate scan content (by comparing everything except
159
+ each file's own unique ID), files sitting in the wrong series folder, and
160
+ inconsistent patient, study, or series metadata within a physical DICOM
161
+ folder. The CLI prints a Rich inventory table, and the reports include a
162
+ spreadsheet-friendly CSV.
163
+ 4. **sift** — splits files into `retained` and `review`, physically, so a
164
+ flagged file is never silently included.
165
+ 5. **rectify** — copies retained files into a sorted, clearly named tree:
166
+ `<series number>-<series description>/<series>_<series ID>_<instance>_e<echo>.dcm`.
167
+ 6. **sanitise** *(optional, `--sanitise`)* — replaces patient identity
168
+ according to a chosen policy; see [Sanitisation](#sanitisation) below.
169
+ 7. **finalise** — independently re-reads the finished output tree (not a
170
+ cached record from earlier stages) to catch any corruption introduced by
171
+ copying or renaming, then archives it with a checksum.
172
+
173
+ This is a from-scratch, simplified rewrite of the original `dichotomise`,
174
+ aimed at being easy to read, debug, and extend, including for someone new to
175
+ Python. It implements a single end-to-end pipeline; there is no separate
176
+ expert/stage-by-stage command.
177
+
178
+ ## Installation
179
+
180
+ `dichotomise` requires Python 3.11+ and [uv](https://docs.astral.sh/uv/).
181
+
182
+ If you already have Python and `pip`:
183
+
184
+ ```bash
185
+ python -m pip install --user uv
186
+ ```
187
+
188
+ Otherwise, install uv directly:
189
+
190
+ ```bash
191
+ curl -LsSf https://astral.sh/uv/install.sh | sh
192
+ ```
193
+
194
+ ### Install from PyPI (after release)
195
+
196
+ ```bash
197
+ uv tool install --python 3.11 dichotomise
198
+ dichotomise --help
199
+ ```
200
+
201
+ This creates an isolated environment for `dichotomise`. If Python 3.11 is not
202
+ available, uv downloads it automatically.
203
+
204
+ ### Install from a source checkout
205
+
206
+ ```bash
207
+ git clone https://github.com/srikash/dichotomise.git
208
+ cd dichotomise
209
+ uv venv --python 3.11
210
+ uv sync --locked
211
+ uv run dichotomise --help
212
+ ```
213
+
214
+ ## Usage
215
+
216
+ ```bash
217
+ dichotomise --source-dir ./study/sub-001 --out-dir ./dichotomise-runs
218
+ ```
219
+
220
+ `--source-dir` accepts either one subject/session export or a scanner export
221
+ containing multiple subjects — they are discovered from DICOM metadata, so
222
+ folder names do not need to be sensible.
223
+
224
+ | Flag | Purpose |
225
+ |---|---|
226
+ | `--source-dir` (required) | Raw DICOM directory to process. |
227
+ | `--out-dir` (required) | Parent directory for the timestamped output folder. |
228
+ | `--sanitise` | Replace patient identity before final archiving. This uses `minimal` unless a policy is chosen. |
229
+ | `--sanitise-policy` | JSON policy name: `minimal` (the default), `standard`, `full`, `retain`, `custom`, or a policy you add yourself. Both `custom` and `custom.json` are accepted. Implies `--sanitise`. |
230
+ | `--subj-id` | First numerical replacement ID. For one study, `6` produces `sub-0006`. With several studies, IDs are enumerated automatically as `sub-0006`, `sub-0007`, `sub-0008`, and so on; the CLI logs a warning. |
231
+ | `--new-id` | Replacement ID for one study. `ADNC0751` becomes `sub-ADNC0751`; an existing `sub-` prefix is retained. |
232
+ | `--random-name` | Generate a random replacement name using the `minimal` sanitisation policy. |
233
+ | `--mapping` | One or more `PatientID:replacement_id` pairs. Repeat the flag or separate pairs with commas. The PatientID must exactly match the DICOM `PatientID`. |
234
+ | `--mapping-file` | JSON object mapping DICOM PatientID to replacement ID, for example `{"source-01": "sub-0001"}`. A full path may be given with or without the `.json` suffix. |
235
+ | `--keep-working-files` | Keep the copied and processed DICOM files (`working/`) instead of deleting them once the archives are verified. |
236
+
237
+ For one study, use `--subj-id`, `--random-name`, or `--new-id`. For several
238
+ studies, use `--subj-id` (automatic enumeration), `--random-name`, `--mapping`,
239
+ or `--mapping-file`; `--new-id` remains intentionally limited to one study.
240
+
241
+ [`docs/example-mapping.json`](docs/example-mapping.json) is a ready-to-copy
242
+ mapping-file example. Its keys must match the source DICOM `PatientID` values;
243
+ its values are the exact replacement labels to write. Its `instructions` field
244
+ is ignored by the CLI.
245
+
246
+ ## Output structure
247
+
248
+ Every run creates one timestamped, UTC output folder beneath `--out-dir`:
249
+
250
+ ```text
251
+ <run-timestamp>_dichotomise_outputs/
252
+ source/
253
+ <patient-id>_<6char-hex>_source-archive_<run-timestamp>.tar.gz
254
+ <patient-id>_<6char-hex>_source-archive_<run-timestamp>.sha256
255
+ archives/
256
+ <subject-label>_<6char-hex>_dichotomised-archive_<run-timestamp>.tar.gz
257
+ <subject-label>_<6char-hex>_dichotomised-archive_<run-timestamp>.sha256
258
+ working/ # temporary, removed unless --keep-working-files
259
+ reports/
260
+ <subject-label>_<6char-hex>/
261
+ stage-01-report.json
262
+ stage-01-audit.csv
263
+ stage-02-report.json
264
+ stage-03-report.json
265
+ run-status.json # in_progress, complete, or failed
266
+ ```
267
+
268
+ `run-status.json` lets you distinguish a complete result from one left by a
269
+ failed or interrupted run. It contains no patient details.
270
+
271
+ A multi-subject run produces one `source/` archive and one `archives/` archive
272
+ per study. Each source archive name uses the original patient ID plus a random
273
+ six-character hexadecimal suffix. The source archive contains untouched DICOM
274
+ files and their original headers, including patient identity; use the
275
+ sanitised `archives/` output for sharing.
276
+
277
+ Processed archive and report names use the study's `<subject-label>` plus a
278
+ random six-character hexadecimal suffix. `<subject-label>` is the real
279
+ `PatientID`, unless `--sanitise` was used, in which case it is the replacement
280
+ label. A sanitised archive's filename and DICOM headers therefore do not
281
+ expose the original identifier.
282
+
283
+ ### Reports
284
+
285
+ Each study has a directory under `reports/` containing:
286
+
287
+ - `stage-01-report.json` — audit totals, duplicate and misfiled folders, and
288
+ one summary per physical DICOM folder.
289
+ - `stage-01-audit.csv` — the same per-folder audit summary for spreadsheets,
290
+ including duplicate/misfiled counts and differing metadata fields.
291
+ - `stage-02-report.json` — files retained versus copied to `review/`.
292
+ - `stage-03-report.json` — archive checksum and a `series_inventory` with the
293
+ first original and final DICOM filename for every series.
294
+
295
+ The audit table and CSV flag differences in Patient ID/name, study UID/date/
296
+ time, and series UID/number/description/protocol within a DICOM folder.
297
+
298
+ Compression is always `.tar.gz`.
299
+
300
+ ## Sanitisation
301
+
302
+ `--sanitise` applies a named policy — a small JSON file describing what
303
+ happens to each DICOM field: kept, removed, replaced with a fixed or
304
+ run-specific value, given a newly generated identifier (consistent across
305
+ every file for that subject), a birth date scrambled to an approximate but
306
+ different year, or a randomly generated placeholder name.
307
+
308
+ Five policies ship in `src/dichotomise/pydcm/policies/`:
309
+
310
+ | Policy | What happens |
311
+ |---|---|
312
+ | `retain` | Nothing changed. |
313
+ | `standard` | Identity, patient address, accession number, institution, and device-operator fields removed; every UID reissued; birth date scrambled by ±1 year (day/month randomised too); demographic fields (e.g. sex) and all scan-descriptive text (protocol name, series/study description, etc.) kept. |
314
+ | `full` | Everything `standard` does, plus scan-descriptive text and the device serial number also removed. |
315
+ | `minimal` *(the default level)* | Everything `standard` does, but a generated pseudonym is used for both patient name (`Abrahall^Gracious`) and patient ID/output name (`abrahall_gracious`); the birth date is the scan date rather than scrambled. |
316
+ | `custom` | A worked, commented example for building your own — not used automatically. |
317
+
318
+ Full detail — including the real scanner-export comparison these were
319
+ built from, the policy file schema, and how to write your own — is in
320
+ [`docs/sanitise-policies.md`](docs/sanitise-policies.md). In short: copy
321
+ `custom.json` to `<your-policy-name>.json` in the same folder, edit it, and
322
+ run with `--sanitise-policy <your-policy-name>`.
323
+
324
+ ## Layout
325
+
326
+ ```text
327
+ src/dichotomise/
328
+ cli.py # the `dichotomise` command (Click + Rich)
329
+ pipeline.py # runs every stage in order
330
+ run.py # output paths for one run
331
+ errors.py # error types
332
+ stages/ # one file per pipeline stage
333
+ pydcm/ # DICOM-specific metadata and sanitisation logic
334
+ policies/ # the sanitisation policy JSON files
335
+ names.py # a self-contained adjective+surname placeholder-name generator
336
+ utils/ # generic filesystem/archive/console helpers
337
+ ```
338
+
339
+ ## Development
340
+
341
+ ```bash
342
+ uv run pytest
343
+ uv run ruff format .
344
+ uv run ruff check .
345
+ uv run mypy src
346
+ ```
347
+
348
+ `tests/data/` (real, non-synthetic scan exports used for some tests) is
349
+ gitignored and never committed — it may contain identifying information.
350
+
351
+ ## The dichotomise workflow
352
+
353
+ ```mermaid
354
+ flowchart TD
355
+ SOURCE["Raw scanner export"] --> CAPTURE["capture<br/>Copies readable DICOM into working/"]
356
+ CAPTURE --> ARCHIVE["source_archive<br/>Per-study verified tarball + checksum (source/)"]
357
+ ARCHIVE --> AUDIT["audit<br/>Structural and metadata QA per subject"]
358
+ AUDIT --> REPORT1["stage-01-report.json<br/>stage-01-audit.csv"]
359
+ AUDIT --> SIFT["sift<br/>Splits retained vs review files"]
360
+ SIFT --> REPORT2["stage-02-report.json"]
361
+ SIFT --> RECTIFY["rectify<br/>Metadata-sorted, renamed DICOM tree"]
362
+ RECTIFY --> SANITISE["sanitise (optional, --sanitise)<br/>Replaces patient identity per policy"]
363
+ RECTIFY --> FINALISE["finalise<br/>Re-verifies output, archives it (archives/)"]
364
+ SANITISE --> FINALISE
365
+ FINALISE --> REPORT3["stage-03-report.json"]
366
+ FINALISE --> ARCHIVES["archives/dichotomised-archive.tar.gz"]
367
+ ```
368
+
369
+ The project is licensed under the MIT License.