xnatbidscli 2.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,890 @@
1
+ import argparse
2
+ import csv
3
+ import io
4
+ import json
5
+ import os
6
+ import re
7
+ import shutil
8
+ import subprocess
9
+ import sys
10
+ import threading
11
+ import time
12
+ from concurrent.futures import ThreadPoolExecutor, as_completed
13
+ from datetime import datetime
14
+ from pathlib import Path
15
+
16
+ from .archive import (
17
+ OK_STATUSES as ARCHIVE_OK_STATUSES,
18
+ archive_experiment,
19
+ delete_experiment_dir,
20
+ )
21
+ from .sysinfo import get_system_username
22
+
23
+ STATUS_COMPLETE = "COMPLETE"
24
+ STATUS_FAILURE = "FAILURE"
25
+ STATUS_NONEXISTENT = "NONEXISTENT"
26
+ STATUS_EMPTY = "EMPTY"
27
+
28
+ _OK_STATUSES = {STATUS_COMPLETE, STATUS_EMPTY}
29
+ _NON_ALNUM = re.compile(r"[^A-Za-z0-9]")
30
+ _DICOM_EXTS = {".dcm", ".ima"}
31
+ _DATE_TAGS = (
32
+ "StudyDate",
33
+ "SeriesDate",
34
+ "AcquisitionDate",
35
+ "ContentDate",
36
+ "InstanceCreationDate",
37
+ )
38
+ _NON_DIGIT = re.compile(r"\D")
39
+
40
+ # Root-level mriconvert_qc.tsv generation (run at the end of mriconvert).
41
+ _SCANS_COLUMNS = [
42
+ "filename",
43
+ "acq_time",
44
+ "series_number",
45
+ "dimensions",
46
+ "size_bytes",
47
+ "participant_id",
48
+ "session_id",
49
+ "datatype",
50
+ "suffix",
51
+ "bids_name",
52
+ "rename",
53
+ "physio",
54
+ "recommend_for_use",
55
+ "complete",
56
+ "usable",
57
+ "qc_rating",
58
+ "rating_reason",
59
+ "qc_notes",
60
+ ]
61
+ # Columns the generator always writes empty; user-entered text in any of them
62
+ # means the file has been reviewed and must not be regenerated over.
63
+ _SCANS_USER_COLUMNS = (
64
+ "rename",
65
+ "physio",
66
+ "recommend_for_use",
67
+ "complete",
68
+ "usable",
69
+ "qc_rating",
70
+ "rating_reason",
71
+ "qc_notes",
72
+ )
73
+ _SUB_RE = re.compile(r"(sub-[A-Za-z0-9]+)")
74
+ _SES_RE = re.compile(r"(ses-[A-Za-z0-9]+)")
75
+ # A SUBJECT/EXPERIMENT directory name matching these exactly (not just
76
+ # containing a substring like it) is one `xnatbidscli download` wrote via
77
+ # --rename-subject/--rename-experiment (or the CSV's SUBJECT_BIDS_RENAME/
78
+ # EXPERIMENT_BIDS_RENAME columns): already `sub-<alnum>`/`ses-<alnum>`, so it
79
+ # must be used as the BIDS label verbatim rather than sanitized and
80
+ # re-prefixed.
81
+ _SUB_RENAMED_RE = re.compile(r"^sub-[A-Za-z0-9]+$")
82
+ _SES_RENAMED_RE = re.compile(r"^ses-[A-Za-z0-9]+$")
83
+ _BIDS_NAME_RE = re.compile(r"^sub-[A-Za-z0-9]+_ses-[A-Za-z0-9]+_(.+)\.nii\.gz$")
84
+
85
+ _print_lock = threading.Lock()
86
+
87
+ # In-place "elapsed time" progress block shown while conversions are running
88
+ # (see _render_progress). Every concurrent session gets its own line, all
89
+ # redrawn together every 5 seconds (or whenever a status line is printed).
90
+ _active_lock = threading.Lock()
91
+ _active: dict[str, float] = {} # label -> start time (time.monotonic())
92
+ _progress_lines = 0 # number of progress lines currently drawn on-screen
93
+
94
+
95
+ def _register_active(label: str) -> None:
96
+ with _active_lock:
97
+ _active[label] = time.monotonic()
98
+ _render_progress()
99
+
100
+
101
+ def _unregister_active(label: str) -> None:
102
+ with _active_lock:
103
+ _active.pop(label, None)
104
+
105
+
106
+ def _progress_snapshot() -> list[str]:
107
+ """Current 'LABEL: converting... elapsed Ns' lines, one per active session."""
108
+ with _active_lock:
109
+ items = list(_active.items())
110
+ now = time.monotonic()
111
+ return [
112
+ f"{label}: converting... elapsed {int(now - start)}s"
113
+ for label, start in items
114
+ ]
115
+
116
+
117
+ def _render_progress() -> None:
118
+ """Redraw the live progress block in place, replacing whatever was drawn
119
+ on the previous call (see _progress_lines)."""
120
+ global _progress_lines
121
+ lines = _progress_snapshot()
122
+ with _print_lock:
123
+ if _progress_lines:
124
+ sys.stdout.write(f"\x1b[{_progress_lines}A\x1b[0J")
125
+ if lines:
126
+ sys.stdout.write("\n".join(lines) + "\n")
127
+ sys.stdout.flush()
128
+ _progress_lines = len(lines)
129
+
130
+
131
+ def _safe_print(msg: str) -> None:
132
+ """Print msg above the live progress block, then redraw the block."""
133
+ global _progress_lines
134
+ lines = _progress_snapshot()
135
+ with _print_lock:
136
+ if _progress_lines:
137
+ sys.stdout.write(f"\x1b[{_progress_lines}A\x1b[0J")
138
+ print(msg)
139
+ if lines:
140
+ sys.stdout.write("\n".join(lines) + "\n")
141
+ sys.stdout.flush()
142
+ _progress_lines = len(lines)
143
+
144
+
145
+ def _logging_now() -> str:
146
+ now = datetime.now()
147
+ return f"{now.strftime('%Y-%m-%d %H:%M:%S')},{now.microsecond // 1000:03d}"
148
+
149
+
150
+ class _LogWriter:
151
+ def __init__(self, path: Path | None):
152
+ self._path = path
153
+ self._lock = threading.Lock()
154
+ self._user = get_system_username()
155
+ if path is not None:
156
+ path.parent.mkdir(parents=True, exist_ok=True)
157
+ with path.open("w", newline="") as f:
158
+ csv.writer(f).writerow(
159
+ [
160
+ "DATESTAMP",
161
+ "USER",
162
+ "PROJECT",
163
+ "SUBJECT",
164
+ "EXPERIMENT",
165
+ "STATUS",
166
+ ]
167
+ )
168
+
169
+ def write(
170
+ self,
171
+ datestamp: str,
172
+ project: str,
173
+ subject: str,
174
+ experiment: str,
175
+ status: str,
176
+ ) -> None:
177
+ if self._path is None:
178
+ return
179
+ with self._lock, self._path.open("a", newline="") as f:
180
+ csv.writer(f).writerow(
181
+ [datestamp, self._user, project, subject, experiment, status]
182
+ )
183
+
184
+
185
+ def _extract_session_date(ds) -> str | None:
186
+ """YYYYMMDD from the first non-empty DICOM date tag in priority order."""
187
+ for tag in _DATE_TAGS:
188
+ value = getattr(ds, tag, None)
189
+ if value is None or value == "":
190
+ continue
191
+ digits = _NON_DIGIT.sub("", str(value))
192
+ if len(digits) >= 8:
193
+ return digits[:8]
194
+ return None
195
+
196
+
197
+ def _scan_dicoms(scans_dir: Path) -> tuple[bool, str | None]:
198
+ """Walk DICOMs under scans_dir, returning (any_readable, session_date).
199
+
200
+ session_date is YYYYMMDD pulled from the first DICOM with a usable date
201
+ tag, or None if no readable DICOM has one. After a DICOM parses but
202
+ yields no date, sibling files in that directory are skipped so a series
203
+ with empty date tags does not block dates available in other series.
204
+ """
205
+ import pydicom
206
+
207
+ any_readable = False
208
+ skip_dirs: set[Path] = set()
209
+ for path in scans_dir.rglob("*"):
210
+ if not path.is_file():
211
+ continue
212
+ if path.suffix.lower() not in _DICOM_EXTS:
213
+ continue
214
+ if path.parent in skip_dirs:
215
+ continue
216
+ try:
217
+ ds = pydicom.dcmread(str(path), stop_before_pixels=True)
218
+ except Exception:
219
+ continue
220
+ any_readable = True
221
+ date = _extract_session_date(ds)
222
+ if date:
223
+ return True, date
224
+ skip_dirs.add(path.parent)
225
+ return any_readable, None
226
+
227
+
228
+ def _convert_one(
229
+ input_root: Path,
230
+ project: str,
231
+ subject: str,
232
+ experiment: str,
233
+ output_dir: Path,
234
+ config_path: Path,
235
+ dcm2bids_path: str,
236
+ ) -> tuple[str, str | None]:
237
+ exp_dir = input_root / project / subject / experiment
238
+ if not exp_dir.is_dir():
239
+ return STATUS_NONEXISTENT, f"session directory not found: {exp_dir}"
240
+
241
+ scans_dir = exp_dir / "scans"
242
+ if not scans_dir.is_dir():
243
+ return STATUS_EMPTY, f"no 'scans/' subdirectory under {exp_dir}"
244
+
245
+ any_readable, session_date = _scan_dicoms(scans_dir)
246
+ if not any_readable:
247
+ return STATUS_EMPTY, "no readable .dcm/.IMA DICOM files under scans/"
248
+
249
+ # A directory already named sub-X/ses-Y came from a download-stage
250
+ # rename: use the label verbatim (no stripping, no re-prefixing, and no
251
+ # DICOM-date override for the session) since the point of an explicit
252
+ # rename is that the user controls the final label.
253
+ if _SUB_RENAMED_RE.match(subject):
254
+ participant = subject[len("sub-"):]
255
+ else:
256
+ participant = _NON_ALNUM.sub("", subject)
257
+
258
+ if _SES_RENAMED_RE.match(experiment):
259
+ session = experiment[len("ses-"):]
260
+ else:
261
+ session = session_date or _NON_ALNUM.sub("", experiment)
262
+
263
+ if not participant or not session:
264
+ return STATUS_FAILURE, (
265
+ f"empty PARTICIPANT or SESSION after sanitizing "
266
+ f"SUBJECT={subject!r} EXPERIMENT={experiment!r}"
267
+ )
268
+
269
+ bids_root = output_dir / project
270
+ sub_dir = bids_root / f"sub-{participant}" / f"ses-{session}"
271
+ if sub_dir.exists() and any(sub_dir.iterdir()):
272
+ _safe_print(f"WARNING: overwriting existing {sub_dir}")
273
+
274
+ bids_root.mkdir(parents=True, exist_ok=True)
275
+
276
+ cmd = [
277
+ dcm2bids_path,
278
+ "-d", str(scans_dir),
279
+ "-p", participant,
280
+ "-s", session,
281
+ "-c", str(config_path),
282
+ "-o", str(bids_root),
283
+ "--clobber",
284
+ "--force_dcm2bids",
285
+ ]
286
+ label = f"{project}/{subject}/{experiment}"
287
+ _register_active(label)
288
+ try:
289
+ proc = subprocess.Popen(
290
+ cmd, stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True
291
+ )
292
+ stdout, stderr = proc.communicate()
293
+ finally:
294
+ _unregister_active(label)
295
+
296
+ if proc.returncode != 0:
297
+ stderr_tail = (stderr or "").strip().splitlines()[-1:] or [""]
298
+ return STATUS_FAILURE, (
299
+ f"dcm2bids exited with code {proc.returncode}"
300
+ + (f" — {stderr_tail[0]}" if stderr_tail[0] else "")
301
+ )
302
+ return STATUS_COMPLETE, None
303
+
304
+
305
+ def _discover_sessions(
306
+ input_root: Path, args: argparse.Namespace
307
+ ) -> list[tuple[str, str, str]]:
308
+ if args.triplet is not None:
309
+ p, s, e = args.triplet
310
+ return [(p, s, e)]
311
+
312
+ if args.subject is not None:
313
+ project, subject = args.subject
314
+ subject_dir = input_root / project / subject
315
+ if not subject_dir.is_dir():
316
+ sys.exit(f"Error: subject directory not found: {subject_dir}")
317
+ sessions = [
318
+ (project, subject, exp.name)
319
+ for exp in sorted(subject_dir.iterdir())
320
+ if exp.is_dir()
321
+ ]
322
+ if not sessions:
323
+ sys.exit(f"Error: no sessions found under {subject_dir}")
324
+ return sessions
325
+
326
+ project = args.project
327
+ project_dir = input_root / project
328
+ if not project_dir.is_dir():
329
+ sys.exit(f"Error: project directory not found: {project_dir}")
330
+ sessions: list[tuple[str, str, str]] = []
331
+ for subject_dir in sorted(project_dir.iterdir()):
332
+ if not subject_dir.is_dir():
333
+ continue
334
+ for exp_dir in sorted(subject_dir.iterdir()):
335
+ if exp_dir.is_dir():
336
+ sessions.append(
337
+ (project, subject_dir.name, exp_dir.name)
338
+ )
339
+ if not sessions:
340
+ sys.exit(f"Error: no sessions found under {project_dir}")
341
+ return sessions
342
+
343
+
344
+ def _read_sidecar(nii_path: Path) -> dict:
345
+ """Load the JSON sidecar that accompanies a .nii.gz, or {} if absent."""
346
+ json_path = nii_path.with_name(nii_path.name[:-7] + ".json")
347
+ if not json_path.is_file():
348
+ return {}
349
+ try:
350
+ with json_path.open(encoding="utf-8") as f:
351
+ data = json.load(f)
352
+ except (OSError, ValueError):
353
+ return {}
354
+ return data if isinstance(data, dict) else {}
355
+
356
+
357
+ def _nii_dimensions(nii_path: Path, nib) -> str:
358
+ """x-joined image shape from nibabel, padded to always include a 4th dim."""
359
+ if nib is None:
360
+ return ""
361
+ try:
362
+ shape = list(nib.load(str(nii_path)).shape)
363
+ except Exception:
364
+ return ""
365
+ while len(shape) < 4:
366
+ shape.append(1)
367
+ return "x".join(str(int(s)) for s in shape)
368
+
369
+
370
+ def _scans_row(nii_path: Path, bids_root: Path, nib) -> list:
371
+ basename = nii_path.name
372
+ sidecar = _read_sidecar(nii_path)
373
+ try:
374
+ size_bytes: object = nii_path.stat().st_size
375
+ except OSError:
376
+ size_bytes = ""
377
+
378
+ name_match = _BIDS_NAME_RE.match(basename)
379
+ sub_match = _SUB_RE.search(basename)
380
+ ses_match = _SES_RE.search(basename)
381
+ stem = basename[:-7] if basename.endswith(".nii.gz") else basename
382
+
383
+ return [
384
+ nii_path.relative_to(bids_root).as_posix(),
385
+ sidecar.get("AcquisitionTime", ""),
386
+ sidecar.get("SeriesNumber", ""),
387
+ _nii_dimensions(nii_path, nib),
388
+ size_bytes,
389
+ sub_match.group(1) if sub_match else "",
390
+ ses_match.group(1) if ses_match else "",
391
+ nii_path.parent.name,
392
+ stem.rsplit("_", 1)[-1],
393
+ name_match.group(1) if name_match else "",
394
+ "", # rename — for the end-user
395
+ "", # physio — for the end-user
396
+ "", # recommend_for_use — for the end-user
397
+ "", # complete — for the end-user
398
+ "", # usable — for the end-user
399
+ "", # qc_rating — for the end-user
400
+ "", # rating_reason — for the end-user
401
+ "", # qc_notes — for the end-user
402
+ ]
403
+
404
+
405
+ def _find_scans_json() -> Path | None:
406
+ """Locate the static mriconvert_qc.json data dictionary (dev tree or wheel)."""
407
+ here = Path(__file__).resolve().parent
408
+ for candidate in (
409
+ here.parent / "assets" / "mriconvert_qc.json", # dev: src/assets/
410
+ here / "assets" / "mriconvert_qc.json", # wheel: xnatbidscli/assets/
411
+ ):
412
+ if candidate.is_file():
413
+ return candidate
414
+ return None
415
+
416
+
417
+ def _qc_tsv_path(bids_root: Path) -> Path:
418
+ """``OUTPUT_DIR/PROJECT-<PROJECT_ID>_mriconvert_qc.tsv`` for ``bids_root``
419
+ (``OUTPUT_DIR/PROJECT_ID``)."""
420
+ return bids_root.parent / f"PROJECT-{bids_root.name}_mriconvert_qc.tsv"
421
+
422
+
423
+ def _backup_qc_tsv(tsv_path: Path, new_content: str) -> Path | None:
424
+ """Lazily back up ``tsv_path`` before it is overwritten with ``new_content``.
425
+
426
+ Copies the current file to
427
+ ``OUTPUT_DIR/mriconvert_qc_backups/<stem>_YYYYMMDD_HHMMSS.tsv`` only when it
428
+ exists and its content differs from ``new_content``, so unchanged reruns
429
+ don't accumulate duplicate backups. Backups are never pruned.
430
+
431
+ Returns
432
+ -------
433
+ Path or None
434
+ The backup path, or None when no backup was needed.
435
+ """
436
+ try:
437
+ with tsv_path.open(newline="", encoding="utf-8") as f:
438
+ if f.read() == new_content:
439
+ return None
440
+ except FileNotFoundError:
441
+ return None
442
+
443
+ backup_dir = tsv_path.parent / "mriconvert_qc_backups"
444
+ backup_dir.mkdir(parents=True, exist_ok=True)
445
+ ts = datetime.now().strftime("%Y%m%d_%H%M%S")
446
+ backup_path = backup_dir / f"{tsv_path.stem}_{ts}{tsv_path.suffix}"
447
+ shutil.copy2(tsv_path, backup_path)
448
+ return backup_path
449
+
450
+
451
+ def _qc_json_path(bids_root: Path) -> Path:
452
+ """``OUTPUT_DIR/PROJECT-<PROJECT_ID>_mriconvert_qc.json`` for ``bids_root``
453
+ (``OUTPUT_DIR/PROJECT_ID``)."""
454
+ return bids_root.parent / f"PROJECT-{bids_root.name}_mriconvert_qc.json"
455
+
456
+
457
+ def _write_scans_json(
458
+ bids_root: Path,
459
+ physio_parent: Path | None,
460
+ dcm2bids_config: Path | None = None,
461
+ ) -> None:
462
+ """Write mriconvert_qc.json from the static asset, injecting
463
+ ``PhysioParent`` and ``Dcm2BidsConfigPath``.
464
+
465
+ Loads the static data dictionary (never mutated in place) and sets
466
+ ``PhysioParent.Value``/``Dcm2BidsConfigPath.Value`` to ``physio_parent``/
467
+ ``dcm2bids_config`` when given. When ``dcm2bids_config`` is given,
468
+ ``Dcm2BidsConfigPath.LastModified`` is also stamped with the current time
469
+ (in Python logging's default ``asctime`` format). When either is None,
470
+ preserves whatever value (and, for ``Dcm2BidsConfigPath``, ``LastModified``)
471
+ was already recorded for it in the destination mriconvert_qc.json from a
472
+ prior run, if any, so omitting ``-y``/``-c`` on a rerun (e.g. under
473
+ ``-m/--maps``) doesn't erase it.
474
+ """
475
+ src_json = _find_scans_json()
476
+ if src_json is None:
477
+ _safe_print(
478
+ "WARNING: mriconvert_qc.json data dictionary not found; skipping the sidecar write."
479
+ )
480
+ return
481
+
482
+ try:
483
+ with src_json.open(encoding="utf-8") as f:
484
+ data = json.load(f)
485
+ except (OSError, ValueError) as exc:
486
+ _safe_print(f"WARNING: could not read {src_json}: {exc}; skipping the sidecar write.")
487
+ return
488
+
489
+ dest_json = _qc_json_path(bids_root)
490
+ prior: dict | None = None
491
+ if (physio_parent is None or dcm2bids_config is None) and dest_json.is_file():
492
+ try:
493
+ with dest_json.open(encoding="utf-8") as f:
494
+ prior = json.load(f)
495
+ except (OSError, ValueError):
496
+ prior = None
497
+
498
+ if physio_parent is not None:
499
+ data.setdefault("PhysioParent", {})["Value"] = str(physio_parent)
500
+ elif prior is not None:
501
+ prior_value = prior.get("PhysioParent", {}).get("Value", "")
502
+ if prior_value:
503
+ data.setdefault("PhysioParent", {})["Value"] = prior_value
504
+
505
+ if dcm2bids_config is not None:
506
+ entry = data.setdefault("Dcm2BidsConfigPath", {})
507
+ entry["Value"] = str(dcm2bids_config)
508
+ entry["LastModified"] = datetime.now().strftime("%Y-%m-%d %H:%M:%S,%f")[:-3]
509
+ elif prior is not None:
510
+ prior_entry = prior.get("Dcm2BidsConfigPath", {})
511
+ prior_value = prior_entry.get("Value", "")
512
+ if prior_value:
513
+ entry = data.setdefault("Dcm2BidsConfigPath", {})
514
+ entry["Value"] = prior_value
515
+ prior_last_modified = prior_entry.get("LastModified", "")
516
+ if prior_last_modified:
517
+ entry["LastModified"] = prior_last_modified
518
+
519
+ with dest_json.open("w", encoding="utf-8") as f:
520
+ json.dump(data, f, indent=4)
521
+ f.write("\n")
522
+
523
+
524
+ def _read_existing_scans(
525
+ tsv_path: Path,
526
+ ) -> tuple[list[str], list[dict]] | None:
527
+ """Parse an existing mriconvert_qc.tsv into (fieldnames, rows), or None if it
528
+ cannot be read."""
529
+ try:
530
+ with tsv_path.open(newline="", encoding="utf-8") as f:
531
+ reader = csv.DictReader(f, delimiter="\t")
532
+ return list(reader.fieldnames or []), list(reader)
533
+ except OSError:
534
+ return None
535
+
536
+
537
+ def _scans_deviations(
538
+ existing_rows: list[dict],
539
+ new_rows: list[list],
540
+ compare_columns: list[str],
541
+ ) -> list[tuple[str, str]]:
542
+ """Deviations in non-user fields between the current mriconvert_qc.tsv and a
543
+ freshly generated set of rows, keyed by ``filename``.
544
+
545
+ Returns a list of ``(kind, message)`` pairs, where ``kind`` is one of
546
+ ``"new"`` (file newly present on disk), ``"missing"`` (a mriconvert_qc.tsv row
547
+ whose file is no longer found on disk — the row is preserved, not
548
+ dropped), or ``"changed"`` (a generator-owned field drifted).
549
+ """
550
+ name_idx = _SCANS_COLUMNS.index("filename")
551
+ col_idx = {c: _SCANS_COLUMNS.index(c) for c in compare_columns}
552
+ existing_by_name = {r.get("filename", ""): r for r in existing_rows}
553
+ new_by_name = {r[name_idx]: r for r in new_rows}
554
+
555
+ deviations: list[tuple[str, str]] = []
556
+ for name in sorted(set(existing_by_name) | set(new_by_name)):
557
+ if name not in existing_by_name:
558
+ deviations.append((
559
+ "new",
560
+ f"{name}: newly present on disk (absent from current mriconvert_qc.tsv)",
561
+ ))
562
+ continue
563
+ if name not in new_by_name:
564
+ deviations.append((
565
+ "missing",
566
+ f"{name}: present in current mriconvert_qc.tsv but no longer found "
567
+ "on disk (row preserved)",
568
+ ))
569
+ continue
570
+ old_row, new_row = existing_by_name[name], new_by_name[name]
571
+ for col in compare_columns:
572
+ new_val = new_row[col_idx[col]]
573
+ new_val = "" if new_val is None else str(new_val)
574
+ old_val = old_row.get(col) or ""
575
+ if new_val != old_val:
576
+ deviations.append((
577
+ "changed",
578
+ f"{name}: {col} changed from {old_val!r} to {new_val!r}",
579
+ ))
580
+ return deviations
581
+
582
+
583
+ def _report_scans_deviations(
584
+ bids_root: Path, deviations: list[tuple[str, str]]
585
+ ) -> None:
586
+ """Print one WARNING per deviation to stdout and write them to a log
587
+ file under <output>/PROJECT_ID/log/."""
588
+ project = bids_root.name
589
+ lines = [
590
+ f"WARNING: mriconvert_qc.tsv deviation [{project}] {msg}"
591
+ for _, msg in deviations
592
+ ]
593
+ for line in lines:
594
+ _safe_print(line)
595
+
596
+ log_dir = bids_root / "log"
597
+ log_dir.mkdir(parents=True, exist_ok=True)
598
+ ts = datetime.now().strftime("%Y%m%d_%H%M%S")
599
+ log_path = log_dir / f"scans_deviations_{ts}.log"
600
+ with log_path.open("w", encoding="utf-8") as f:
601
+ f.write("\n".join(lines) + "\n")
602
+ _safe_print(
603
+ f"{len(deviations)} mriconvert_qc.tsv deviation(s) for {project} "
604
+ f"logged to {log_path}"
605
+ )
606
+
607
+
608
+ def _generate_scans_tsv(
609
+ bids_root: Path,
610
+ physio_parent: Path | None = None,
611
+ dcm2bids_config: Path | None = None,
612
+ ) -> int:
613
+ """Write <output_dir>/PROJECT-<bids_root.name>_mriconvert_qc.tsv from
614
+ every .nii.gz under the dataset.
615
+
616
+ Walks bids_root with os.walk (skipping the dcm2bids ``tmp_dcm2bids``
617
+ scratch directory) and emits one row per .nii.gz, then writes the
618
+ mriconvert_qc.json sidecar alongside it (see ``_write_scans_json``). When a
619
+ mriconvert_qc.tsv is already present, rows are merged by ``filename``: a row
620
+ already present in mriconvert_qc.tsv is always kept exactly as-is (preserving
621
+ any reviewer edits), even if its generator-owned fields have since
622
+ drifted — such drift is only reported as a WARNING, never applied. Only
623
+ rows for files that are newly present on disk (their filename absent from
624
+ the current mriconvert_qc.tsv) are appended. This lets separate sessions be
625
+ converted at different times, appending to mriconvert_qc.tsv without
626
+ disturbing already-reviewed rows. Before an existing mriconvert_qc.tsv is
627
+ overwritten with different content, it is backed up (see
628
+ ``_backup_qc_tsv``).
629
+
630
+ ``physio_parent`` and ``dcm2bids_config``, when given, are recorded as the
631
+ ``PhysioParent``/``Dcm2BidsConfigPath`` values in mriconvert_qc.json;
632
+ when omitted, a value already recorded there is preserved (see
633
+ ``_write_scans_json``).
634
+
635
+ Returns the number of preserved rows whose file is no longer found on
636
+ disk, for the caller to fold into an end-of-run summary.
637
+ """
638
+ if not bids_root.is_dir():
639
+ return 0
640
+
641
+ tsv_path = _qc_tsv_path(bids_root)
642
+
643
+ try:
644
+ import nibabel as nib
645
+ except ImportError:
646
+ nib = None
647
+ _safe_print(
648
+ "WARNING: nibabel not installed; the 'dimensions' column in "
649
+ "mriconvert_qc.tsv will be empty. Install nibabel to populate it."
650
+ )
651
+
652
+ rows: list[list] = []
653
+ for dirpath, dirnames, filenames in os.walk(bids_root):
654
+ dirnames[:] = [d for d in dirnames if d != "tmp_dcm2bids"]
655
+ for fname in filenames:
656
+ if fname.endswith(".nii.gz"):
657
+ rows.append(
658
+ _scans_row(Path(dirpath) / fname, bids_root, nib)
659
+ )
660
+ rows.sort(key=lambda r: r[0])
661
+
662
+ existing = _read_existing_scans(tsv_path) if tsv_path.is_file() else None
663
+ missing_count = 0
664
+ if existing is None:
665
+ merged_rows = rows
666
+ added = len(rows)
667
+ else:
668
+ _, existing_rows = existing
669
+ # Non-user, non-key columns the generator owns. 'dimensions' is
670
+ # excluded when nibabel is missing so its empty values are not
671
+ # reported as spurious deviations.
672
+ compare_columns = [
673
+ c
674
+ for c in _SCANS_COLUMNS
675
+ if c not in _SCANS_USER_COLUMNS
676
+ and c != "filename"
677
+ and not (c == "dimensions" and nib is None)
678
+ ]
679
+ deviations = _scans_deviations(existing_rows, rows, compare_columns)
680
+ if deviations:
681
+ _report_scans_deviations(bids_root, deviations)
682
+ missing_count = sum(1 for kind, _ in deviations if kind == "missing")
683
+
684
+ name_idx = _SCANS_COLUMNS.index("filename")
685
+ existing_by_name = {r.get("filename", ""): r for r in existing_rows}
686
+ new_by_name = {r[name_idx]: r for r in rows}
687
+
688
+ merged_rows = []
689
+ added = 0
690
+ for name in sorted(set(existing_by_name) | set(new_by_name)):
691
+ if name in existing_by_name:
692
+ # Already present (whether or not it's a "partially matching"
693
+ # row, i.e. identical besides the reviewer columns): keep it
694
+ # untouched rather than overwriting it.
695
+ old_row = existing_by_name[name]
696
+ merged_rows.append([old_row.get(c, "") for c in _SCANS_COLUMNS])
697
+ else:
698
+ # Newly present on disk and absent from mriconvert_qc.tsv: add it.
699
+ merged_rows.append(new_by_name[name])
700
+ added += 1
701
+
702
+ buf = io.StringIO(newline="")
703
+ writer = csv.writer(buf, delimiter="\t")
704
+ writer.writerow(_SCANS_COLUMNS)
705
+ writer.writerows(merged_rows)
706
+ content = buf.getvalue()
707
+
708
+ backup_path = _backup_qc_tsv(tsv_path, content)
709
+ if backup_path is not None:
710
+ _safe_print(f"Backed up previous {tsv_path.name} to {backup_path}")
711
+
712
+ with tsv_path.open("w", newline="", encoding="utf-8") as f:
713
+ f.write(content)
714
+ if existing is None:
715
+ _safe_print(f"Wrote {tsv_path} ({len(merged_rows)} scan(s))")
716
+ else:
717
+ _safe_print(
718
+ f"Updated {tsv_path} ({len(merged_rows)} scan(s), {added} newly "
719
+ "added; existing rows preserved)"
720
+ )
721
+
722
+ _write_scans_json(bids_root, physio_parent, dcm2bids_config)
723
+
724
+ return missing_count
725
+
726
+
727
+ def mriconvert_cmd(args: argparse.Namespace) -> int:
728
+ if args.nconvert < 1:
729
+ sys.exit("Error: -n/--nconvert must be >= 1.")
730
+
731
+ input_root = Path(args.input).resolve()
732
+ if not input_root.is_dir():
733
+ sys.exit(f"Error: input directory not found: {input_root}")
734
+
735
+ output_dir = Path(args.output).resolve()
736
+
737
+ physio_parent = Path(args.physio_parent).resolve() if args.physio_parent else None
738
+ if physio_parent is not None and not physio_parent.is_dir():
739
+ _safe_print(f"WARNING: -y/--physio directory not found: {physio_parent}")
740
+
741
+ dcm2bids_config = Path(args.config).resolve() if args.config else None
742
+
743
+ # --maps: skip the dcm2bids conversion entirely and only (re)generate the
744
+ # mriconvert_qc.tsv/mriconvert_qc.json tabular outputs for every project in scope from the
745
+ # already-converted BIDS data under OUTPUT_DIR. The config, pydicom, and the
746
+ # dcm2bids/dcm2niix tools are unused on this path, so none are required.
747
+ if args.maps:
748
+ if not output_dir.is_dir():
749
+ sys.exit(
750
+ f"Error: output directory not found: {output_dir}; run "
751
+ "mriconvert without -m/--maps first."
752
+ )
753
+ sessions = _discover_sessions(input_root, args)
754
+ missing_total = 0
755
+ for project in sorted({p for p, _, _ in sessions}):
756
+ missing_total += _generate_scans_tsv(
757
+ output_dir / project, physio_parent, dcm2bids_config
758
+ )
759
+ if missing_total:
760
+ print(
761
+ f"WARNING: {missing_total} mriconvert_qc.tsv row(s) across all "
762
+ "project(s) in scope reference file(s) no longer found on "
763
+ "disk (rows preserved; see WARNING(s) above)."
764
+ )
765
+ return 0
766
+
767
+ if dcm2bids_config is None:
768
+ sys.exit("Error: -c/--config is required unless -m/--maps is given.")
769
+ config_path = dcm2bids_config
770
+ if not config_path.is_file():
771
+ sys.exit(f"Error: config file not found: {config_path}")
772
+
773
+ output_dir.mkdir(parents=True, exist_ok=True)
774
+
775
+ try:
776
+ import pydicom # noqa: F401
777
+ except ImportError:
778
+ sys.exit(
779
+ "Error: pydicom is required for mriconvert. "
780
+ "Install it via 'uv sync' or 'pip install pydicom'."
781
+ )
782
+
783
+ dcm2bids = shutil.which("dcm2bids")
784
+ dcm2niix = shutil.which("dcm2niix")
785
+ missing = [
786
+ name for name, path in (("dcm2bids", dcm2bids), ("dcm2niix", dcm2niix))
787
+ if path is None
788
+ ]
789
+ if missing:
790
+ sys.exit(
791
+ f"Error: required tool(s) not found on PATH: {', '.join(missing)}."
792
+ )
793
+
794
+ sessions = _discover_sessions(input_root, args)
795
+ project = sessions[0][0]
796
+
797
+ log_path: Path | None = None
798
+ if args.log:
799
+ while True:
800
+ ts = datetime.now().strftime("%Y%m%d_%H%M%S")
801
+ log_path = output_dir / "log" / f"mriconvert_{ts}_log.csv"
802
+ if not log_path.exists():
803
+ break
804
+ time.sleep(1)
805
+ log_writer = _LogWriter(log_path)
806
+
807
+ counts = {
808
+ STATUS_COMPLETE: 0,
809
+ STATUS_FAILURE: 0,
810
+ STATUS_NONEXISTENT: 0,
811
+ STATUS_EMPTY: 0,
812
+ }
813
+
814
+ def _one(triplet: tuple[str, str, str]) -> str:
815
+ p, s, e = triplet
816
+ start = _logging_now()
817
+ status, detail = _convert_one(
818
+ input_root, p, s, e, output_dir, config_path, dcm2bids,
819
+ )
820
+ line = f"{p}/{s}/{e}: {status}"
821
+ if detail:
822
+ line += f" — {detail}"
823
+ _safe_print(line)
824
+ log_writer.write(start, p, s, e, status)
825
+
826
+ archive_ok = False
827
+ if args.archive:
828
+ a_status, a_detail = archive_experiment(input_root, p, s, e)
829
+ a_line = f" archive {p}/{s}/{e}: {a_status}"
830
+ if a_detail:
831
+ a_line += f" — {a_detail}"
832
+ _safe_print(a_line)
833
+ archive_ok = a_status in ARCHIVE_OK_STATUSES
834
+
835
+ if args.delete:
836
+ if args.archive:
837
+ if archive_ok:
838
+ delete_experiment_dir(input_root, p, s, e)
839
+ elif status in _OK_STATUSES:
840
+ delete_experiment_dir(input_root, p, s, e)
841
+ return status
842
+
843
+ stop_progress = threading.Event()
844
+
845
+ def _progress_loop() -> None:
846
+ while not stop_progress.wait(5):
847
+ _render_progress()
848
+
849
+ progress_thread = threading.Thread(target=_progress_loop, daemon=True)
850
+ progress_thread.start()
851
+ try:
852
+ if args.nconvert <= 1:
853
+ for triplet in sessions:
854
+ counts[_one(triplet)] += 1
855
+ else:
856
+ with ThreadPoolExecutor(max_workers=args.nconvert) as ex:
857
+ futures = [ex.submit(_one, t) for t in sessions]
858
+ for fut in as_completed(futures):
859
+ counts[fut.result()] += 1
860
+ finally:
861
+ stop_progress.set()
862
+ progress_thread.join()
863
+ _render_progress() # clear any leftover progress lines (all sessions are done)
864
+
865
+ total = sum(counts.values())
866
+ print(f"\nProcessed {total} session(s):")
867
+ for status in (
868
+ STATUS_COMPLETE,
869
+ STATUS_FAILURE,
870
+ STATUS_NONEXISTENT,
871
+ STATUS_EMPTY,
872
+ ):
873
+ print(f" {status}: {counts[status]}")
874
+ if log_path is not None:
875
+ print(f"Log written to {log_path}")
876
+
877
+ missing_total = 0
878
+ for project in sorted({p for p, _, _ in sessions}):
879
+ missing_total += _generate_scans_tsv(
880
+ output_dir / project, physio_parent, dcm2bids_config
881
+ )
882
+ if missing_total:
883
+ print(
884
+ f"WARNING: {missing_total} mriconvert_qc.tsv row(s) across all "
885
+ "project(s) in scope reference file(s) no longer found on disk "
886
+ "(rows preserved; see WARNING(s) above)."
887
+ )
888
+
889
+ bad = counts[STATUS_FAILURE] + counts[STATUS_NONEXISTENT]
890
+ return 0 if bad == 0 else 1