xnatbidscli 2.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
xnatbidscli/bidsmap.py ADDED
@@ -0,0 +1,820 @@
1
+ """Backs the ``xnatbidscli bidsmap`` command: participant/session mapping and a
2
+ rename-copy engine for a BIDS-shaped project directory produced by
3
+ ``xnatbidscli mriconvert``.
4
+
5
+ The xnatbidscli workflow is "convert" (raw source data -> BIDS via
6
+ ``mriconvert``, with associated physio placed by ``physioconvert``) then
7
+ "map" (raw BIDS -> renamed/mapped BIDS output, this module) — so the source
8
+ data and the unmapped BIDS data are both preserved, and only the final copy
9
+ carries participant/session and file-level renames. The generic
10
+ scan/blank/merge/copy-plan helpers below work for any ``sub-*/[ses-*/]``
11
+ tree; the mri-specific pieces (QC exclusion, rename columns, root manifest
12
+ patching) live at the bottom alongside ``bidsmap_cmd``.
13
+ """
14
+
15
+ import argparse
16
+ import csv
17
+ import json
18
+ import os
19
+ import shutil
20
+ import sys
21
+ from pathlib import Path
22
+
23
+ import pandas as pd
24
+
25
+ PARTICIPANT_COLS = ["participant_id", "participant_rename"]
26
+ SESSION_COLS = ["session_id", "session_rename"]
27
+
28
+ # Default directories skipped when walking a source tree during copy-with-rename.
29
+ # tmp_dcm2bids holds dcm2bids scratch data with no sub-*/ses-* home to rename;
30
+ # log holds diagnostic run logs, not BIDS data.
31
+ DEFAULT_SKIP_DIRS = frozenset({"tmp_dcm2bids", "log"})
32
+
33
+ # Main file extension and sidecars for the mri-shaped BIDS tree bidsmap maps.
34
+ # A physio _physio.tsv.gz/.json pair co-located under sub-*/ses-*/<datatype>/
35
+ # doesn't match main_ext, so it only gets participant/session label
36
+ # substitution below (it was already written under its final name by
37
+ # physioconvert, so no bids_name override is needed).
38
+ _MAIN_EXT = ".nii.gz"
39
+ _SIDECAR_EXTS = (".json", ".bval", ".bvec")
40
+
41
+
42
+ def scan_pairs(bids_dir: Path) -> tuple[list[dict[str, str]], bool, list[str]]:
43
+ """Walk a BIDS-shaped dataset for sub-*/ses-* directories.
44
+
45
+ Returns (rows, has_sessions, skipped). ``rows`` is a list of dicts with
46
+ ``participant_id`` (and ``session_id`` when the dataset uses sessions).
47
+ ``has_sessions`` is True when at least one participant has a ``ses-*``
48
+ subdirectory; in that case participants without any session directory are
49
+ omitted and listed in ``skipped``.
50
+ """
51
+ participants = sorted(
52
+ p.name for p in bids_dir.iterdir() if p.is_dir() and p.name.startswith("sub-")
53
+ )
54
+
55
+ sessions_by_participant: dict[str, list[str]] = {}
56
+ for participant in participants:
57
+ sessions_by_participant[participant] = sorted(
58
+ s.name
59
+ for s in (bids_dir / participant).iterdir()
60
+ if s.is_dir() and s.name.startswith("ses-")
61
+ )
62
+
63
+ has_sessions = any(sessions_by_participant.values())
64
+
65
+ rows: list[dict[str, str]] = []
66
+ skipped: list[str] = []
67
+ if has_sessions:
68
+ for participant in participants:
69
+ sessions = sessions_by_participant[participant]
70
+ if not sessions:
71
+ skipped.append(participant)
72
+ continue
73
+ for session in sessions:
74
+ rows.append({"participant_id": participant, "session_id": session})
75
+ else:
76
+ for participant in participants:
77
+ rows.append({"participant_id": participant})
78
+
79
+ return rows, has_sessions, skipped
80
+
81
+
82
+ def blank_map(rows: list[dict[str, str]], has_sessions: bool) -> pd.DataFrame:
83
+ columns = list(PARTICIPANT_COLS)
84
+ key_cols = ["participant_id"]
85
+ if has_sessions:
86
+ columns = PARTICIPANT_COLS + SESSION_COLS
87
+ key_cols = ["participant_id", "session_id"]
88
+
89
+ fresh = pd.DataFrame(rows, columns=columns).fillna("")
90
+ fresh = fresh.sort_values(key_cols, ignore_index=True)
91
+ return fresh
92
+
93
+
94
+ def merge_existing(fresh: pd.DataFrame, existing_path: Path) -> tuple[pd.DataFrame, int]:
95
+ existing = pd.read_csv(existing_path, sep="\t", dtype=str).fillna("")
96
+
97
+ # Key columns are whichever of the fresh key columns the existing file
98
+ # also has; this keeps the merge robust if the dataset gained or lost
99
+ # sessions since the file was first generated.
100
+ key_cols = [c for c in ("participant_id", "session_id") if c in fresh.columns]
101
+ shared_keys = [c for c in key_cols if c in existing.columns]
102
+ if not shared_keys:
103
+ sys.exit(
104
+ f"Error: existing map file {existing_path} has no participant_id "
105
+ "column to merge on. Move or delete it and re-run."
106
+ )
107
+
108
+ existing_keys = set(map(tuple, existing[shared_keys].to_numpy()))
109
+ fresh_key_tuples = fresh[shared_keys].apply(tuple, axis=1)
110
+ new_rows = fresh[~fresh_key_tuples.isin(existing_keys)]
111
+
112
+ merged = pd.concat([existing, new_rows], ignore_index=True).fillna("")
113
+ merged = merged.sort_values(key_cols, ignore_index=True)
114
+ return merged, len(new_rows)
115
+
116
+
117
+ def load_participant_session_map(
118
+ map_path: Path,
119
+ ) -> tuple[dict[str, str], dict[tuple[str, str], str]]:
120
+ """Read a participant/session map TSV and return (participant_map, session_map).
121
+
122
+ ``participant_map`` is ``{sub_old: sub_new}`` for rows where
123
+ ``participant_rename`` is non-empty. ``session_map`` is
124
+ ``{(sub_old, ses_old): ses_new}`` for rows where ``session_rename`` is
125
+ non-empty. Blank rename columns are treated as "keep the original label".
126
+ """
127
+ participant_map: dict[str, str] = {}
128
+ session_map: dict[tuple[str, str], str] = {}
129
+
130
+ try:
131
+ with map_path.open(newline="", encoding="utf-8") as f:
132
+ for row in csv.DictReader(f, delimiter="\t"):
133
+ sub_old = (row.get("participant_id") or "").strip()
134
+ sub_new = (row.get("participant_rename") or "").strip()
135
+ ses_old = (row.get("session_id") or "").strip()
136
+ ses_new = (row.get("session_rename") or "").strip()
137
+ if sub_old and sub_new:
138
+ participant_map[sub_old] = sub_new
139
+ if sub_old and ses_old and ses_new:
140
+ session_map[(sub_old, ses_old)] = ses_new
141
+ except OSError as exc:
142
+ sys.exit(f"Error reading map file {map_path}: {exc}")
143
+
144
+ return participant_map, session_map
145
+
146
+
147
+ def build_excluded_stems(excluded: set[str], main_ext: str = ".nii.gz") -> set[tuple[str, str]]:
148
+ """Convert ``{main_rel_posix}`` to ``{(parent_dir_posix, old_stem)}``.
149
+
150
+ Allows sidecar files to be excluded by the same key as their main file
151
+ (e.g. a ``.nii.gz`` or a ``.tsv.gz``).
152
+ """
153
+ result: set[tuple[str, str]] = set()
154
+ for rel in excluded:
155
+ if rel.endswith(main_ext):
156
+ p = Path(rel)
157
+ result.add((p.parent.as_posix(), p.name[: -len(main_ext)]))
158
+ return result
159
+
160
+
161
+ def build_stem_rename(
162
+ rename_map: dict[str, str], main_ext: str = ".nii.gz"
163
+ ) -> dict[tuple[str, str], str]:
164
+ """Pivot ``{main_rel_posix: new_bids_name}`` to
165
+ ``{(parent_dir_posix, old_stem): new_bids_name}``.
166
+
167
+ The pivot lets sidecar files (e.g. ``.json``, ``.bval``, ``.bvec``) be
168
+ looked up by the same key as their main-file sibling without
169
+ reconstructing the main file's path.
170
+ """
171
+ stem_rename: dict[tuple[str, str], str] = {}
172
+ for rel, new_bids_name in rename_map.items():
173
+ p = Path(rel)
174
+ stem_rename[(p.parent.as_posix(), p.name[: -len(main_ext)])] = new_bids_name
175
+ return stem_rename
176
+
177
+
178
+ def new_filename(
179
+ filename: str,
180
+ sub_old: str | None,
181
+ ses_old: str | None,
182
+ sub_new: str | None,
183
+ ses_new: str | None,
184
+ stem_rename: dict[tuple[str, str], str],
185
+ rel_parent_posix: str,
186
+ main_ext: str = ".nii.gz",
187
+ sidecar_exts: tuple[str, ...] = (".json", ".bval", ".bvec"),
188
+ ) -> str:
189
+ """Return the renamed filename for a single file.
190
+
191
+ Applies participant/session label substitution, then overrides the
192
+ ``bids_name`` portion for main files (``main_ext``) and their sidecars
193
+ (``sidecar_exts``) when ``stem_rename`` has an entry for the file.
194
+ """
195
+ if filename.endswith(main_ext):
196
+ old_stem, ext = filename[: -len(main_ext)], main_ext
197
+ else:
198
+ ext = next((e for e in sidecar_exts if filename.endswith(e)), None)
199
+ if ext is not None:
200
+ old_stem = filename[: -len(ext)]
201
+ else:
202
+ # Non-sidecar file: apply sub/ses label substitution only.
203
+ result = filename
204
+ if sub_old and sub_new and sub_old != sub_new:
205
+ result = result.replace(sub_old, sub_new)
206
+ if ses_old and ses_new and ses_old != ses_new:
207
+ result = result.replace(ses_old, ses_new)
208
+ return result
209
+
210
+ # Check for a bids_name override from the rename column in scans.tsv.
211
+ if (rel_parent_posix, old_stem) in stem_rename:
212
+ new_bids_name = stem_rename[(rel_parent_posix, old_stem)]
213
+ if sub_new and ses_new:
214
+ prefix = f"{sub_new}_{ses_new}_"
215
+ elif sub_new:
216
+ prefix = f"{sub_new}_"
217
+ else:
218
+ prefix = ""
219
+ return prefix + new_bids_name + ext
220
+
221
+ # No bids_name override: substitute sub/ses labels in the stem only.
222
+ new_stem = old_stem
223
+ if sub_old and sub_new and sub_old != sub_new:
224
+ new_stem = new_stem.replace(sub_old, sub_new)
225
+ if ses_old and ses_new and ses_old != ses_new:
226
+ new_stem = new_stem.replace(ses_old, ses_new)
227
+ return new_stem + ext
228
+
229
+
230
+ def build_rename_copy_plan(
231
+ source_root: Path,
232
+ participant_map: dict[str, str],
233
+ session_map: dict[tuple[str, str], str],
234
+ rename_map: dict[str, str],
235
+ excluded_stems: set[tuple[str, str]],
236
+ main_ext: str = ".nii.gz",
237
+ sidecar_exts: tuple[str, ...] = (".json", ".bval", ".bvec"),
238
+ skip_dirs: frozenset[str] | None = None,
239
+ skip_root_files: frozenset[str] = frozenset(),
240
+ ) -> list[tuple[Path, str]]:
241
+ """Walk source_root and return ``[(src_abs_path, dest_rel_posix)]``.
242
+
243
+ Does not touch the filesystem beyond reading directory entries. Skips
244
+ ``skip_dirs`` (default ``DEFAULT_SKIP_DIRS``), omits any main file (and
245
+ its sidecars) whose stem is in ``excluded_stems``, and omits any
246
+ root-level file named in ``skip_root_files`` (e.g. a source manifest that
247
+ the caller will write under a different name in the destination instead
248
+ of copying verbatim).
249
+ """
250
+ if skip_dirs is None:
251
+ skip_dirs = DEFAULT_SKIP_DIRS
252
+
253
+ stem_rename = build_stem_rename(rename_map, main_ext)
254
+ plan: list[tuple[Path, str]] = []
255
+
256
+ for dirpath_str, dirnames, filenames in os.walk(source_root):
257
+ dirnames[:] = sorted(d for d in dirnames if d not in skip_dirs)
258
+
259
+ dirpath = Path(dirpath_str)
260
+ rel_dir = dirpath.relative_to(source_root)
261
+ parts = rel_dir.parts # e.g. ("sub-A", "ses-X", "anat") or ()
262
+ is_root = parts == ()
263
+
264
+ sub_old = next((p for p in parts if p.startswith("sub-")), None)
265
+ ses_old = next((p for p in parts if p.startswith("ses-")), None)
266
+ sub_new = participant_map.get(sub_old, sub_old) if sub_old else None
267
+ ses_new = (
268
+ session_map.get((sub_old, ses_old), ses_old)
269
+ if (sub_old and ses_old)
270
+ else ses_old
271
+ )
272
+
273
+ new_parts: list[str] = []
274
+ for part in parts:
275
+ if part == sub_old and sub_new and sub_new != sub_old:
276
+ new_parts.append(sub_new)
277
+ elif part == ses_old and ses_new and ses_new != ses_old:
278
+ new_parts.append(ses_new)
279
+ else:
280
+ new_parts.append(part)
281
+
282
+ new_rel_dir = Path(*new_parts) if new_parts else Path(".")
283
+ rel_parent_posix = rel_dir.as_posix()
284
+
285
+ for filename in filenames:
286
+ if is_root and filename in skip_root_files:
287
+ continue
288
+
289
+ # Derive the check stem: non-None only for main files and sidecars.
290
+ if filename.endswith(main_ext):
291
+ check_stem: str | None = filename[: -len(main_ext)]
292
+ else:
293
+ ext = next((e for e in sidecar_exts if filename.endswith(e)), None)
294
+ check_stem = filename[: -len(ext)] if ext else None
295
+
296
+ if check_stem is not None and (rel_parent_posix, check_stem) in excluded_stems:
297
+ continue
298
+
299
+ new_fname = new_filename(
300
+ filename, sub_old, ses_old, sub_new, ses_new,
301
+ stem_rename, rel_parent_posix, main_ext, sidecar_exts,
302
+ )
303
+ dest_rel = (
304
+ new_fname
305
+ if new_rel_dir == Path(".")
306
+ else (new_rel_dir / new_fname).as_posix()
307
+ )
308
+ plan.append((dirpath / filename, dest_rel))
309
+
310
+ return plan
311
+
312
+
313
+ def existing_session_dirs(dest_bids: Path) -> set[str]:
314
+ """Relative posix keys (``sub-X`` or ``sub-X/ses-Y``) for every session
315
+ directory already present under ``dest_bids`` before this run.
316
+
317
+ Used to detect when a copy plan is about to add files into a session
318
+ that was already mapped by a previous run, so that can be flagged with a
319
+ WARNING instead of happening silently.
320
+ """
321
+ keys: set[str] = set()
322
+ if not dest_bids.is_dir():
323
+ return keys
324
+ for sub_dir in dest_bids.iterdir():
325
+ if not sub_dir.is_dir() or not sub_dir.name.startswith("sub-"):
326
+ continue
327
+ ses_dirs = [
328
+ d for d in sub_dir.iterdir() if d.is_dir() and d.name.startswith("ses-")
329
+ ]
330
+ if ses_dirs:
331
+ keys.update(f"{sub_dir.name}/{ses_dir.name}" for ses_dir in ses_dirs)
332
+ else:
333
+ keys.add(sub_dir.name)
334
+ return keys
335
+
336
+
337
+ def session_key(dest_rel: str) -> str | None:
338
+ """``sub-X`` or ``sub-X/ses-Y`` prefix of a plan destination path, or
339
+ ``None`` for a root-level file (e.g. ``scans.tsv``)."""
340
+ parts = dest_rel.split("/")
341
+ if not parts or not parts[0].startswith("sub-"):
342
+ return None
343
+ if len(parts) > 1 and parts[1].startswith("ses-"):
344
+ return f"{parts[0]}/{parts[1]}"
345
+ return parts[0]
346
+
347
+
348
+ def partition_plan_for_incremental(
349
+ plan: list[tuple[Path, str]],
350
+ dest_root: Path,
351
+ ) -> tuple[list[tuple[Path, str]], list[tuple[Path, str]]]:
352
+ """Split a copy plan into ``(to_copy, already_mapped)``.
353
+
354
+ A file under ``sub-*/`` whose destination already exists on disk is
355
+ treated as already mapped by a previous run and left untouched. Root-level
356
+ metadata files (``scans.tsv``, ``participants.tsv``,
357
+ ``dataset_description.json``, ...) are always re-copied, since they
358
+ reflect the fully merged state already maintained on the source side, and
359
+ are patched in place afterward by the caller.
360
+ """
361
+ to_copy: list[tuple[Path, str]] = []
362
+ already_mapped: list[tuple[Path, str]] = []
363
+ for src, dest_rel in plan:
364
+ is_root_level = "/" not in dest_rel
365
+ if not is_root_level and (dest_root / dest_rel).exists():
366
+ already_mapped.append((src, dest_rel))
367
+ else:
368
+ to_copy.append((src, dest_rel))
369
+ return to_copy, already_mapped
370
+
371
+
372
+ def check_collisions(
373
+ plan: list[tuple[Path, str]],
374
+ ) -> tuple[list[tuple[Path, str]], list[str]]:
375
+ """Detect destination collisions in the copy plan.
376
+
377
+ Returns ``(clean_plan, warnings)`` where ``clean_plan`` excludes any entry
378
+ whose destination is shared by more than one source, and ``warnings``
379
+ names each collision. Colliding files are omitted entirely rather than
380
+ having one silently win.
381
+ """
382
+ dest_to_sources: dict[str, list[Path]] = {}
383
+ for src, dest_rel in plan:
384
+ dest_to_sources.setdefault(dest_rel, []).append(src)
385
+
386
+ colliding: set[str] = {d for d, srcs in dest_to_sources.items() if len(srcs) > 1}
387
+
388
+ warnings: list[str] = []
389
+ for dest_rel in sorted(colliding):
390
+ srcs_str = ", ".join(str(s) for s in dest_to_sources[dest_rel])
391
+ warnings.append(
392
+ f"WARNING: destination collision — {srcs_str} all map to "
393
+ f"{dest_rel!r}; none will be copied."
394
+ )
395
+
396
+ clean_plan = [(src, d) for src, d in plan if d not in colliding]
397
+ return clean_plan, warnings
398
+
399
+
400
+ def execute_copy_plan(
401
+ plan: list[tuple[Path, str]],
402
+ dest_root: Path,
403
+ ) -> list[str]:
404
+ """Copy each (src, dest_rel) into dest_root, creating directories as needed.
405
+
406
+ Returns a list of warning strings for any files that fail to copy.
407
+ """
408
+ warnings: list[str] = []
409
+ for src, dest_rel in plan:
410
+ dest = dest_root / dest_rel
411
+ dest.parent.mkdir(parents=True, exist_ok=True)
412
+ try:
413
+ shutil.copy2(src, dest)
414
+ except OSError as exc:
415
+ warnings.append(f"WARNING: could not copy {src} to {dest}: {exc}")
416
+ return warnings
417
+
418
+
419
+ def rename_lookup_from_plan(
420
+ plan: list[tuple[Path, str]], source_root: Path
421
+ ) -> dict[str, str]:
422
+ """Map each file's original path (relative to ``source_root``, POSIX) to
423
+ its renamed destination path, from an already-built copy plan.
424
+
425
+ Lets a root-level manifest's path-valued columns (e.g. ``scans.tsv``'s
426
+ ``filename``) be patched by lookup instead of re-deriving the rename
427
+ independently. A path with no entry (its file was excluded from the copy,
428
+ e.g. by a QC filter or a collision) is simply absent — callers leave such
429
+ values unchanged.
430
+ """
431
+ return {
432
+ src.relative_to(source_root).as_posix(): dest_rel for src, dest_rel in plan
433
+ }
434
+
435
+
436
+ def patch_id_columns(
437
+ rows: list[dict[str, str]],
438
+ participant_map: dict[str, str],
439
+ session_map: dict[tuple[str, str], str],
440
+ participant_col: str = "participant_id",
441
+ session_col: str | None = "session_id",
442
+ ) -> None:
443
+ """Rewrite participant/session id columns of ``rows`` in place.
444
+
445
+ ``session_col`` may be ``None`` (or absent from a row) for a manifest with
446
+ no session granularity (e.g. ``participants.tsv``) — only
447
+ ``participant_col`` is then patched. The session lookup happens before
448
+ ``participant_col`` is overwritten, since ``session_map`` is keyed by the
449
+ *original* participant label.
450
+ """
451
+ for row in rows:
452
+ old_sub = row.get(participant_col, "")
453
+ if session_col is not None and session_col in row:
454
+ old_ses = row.get(session_col, "")
455
+ if old_sub and old_ses:
456
+ row[session_col] = session_map.get((old_sub, old_ses), old_ses)
457
+ if old_sub:
458
+ row[participant_col] = participant_map.get(old_sub, old_sub)
459
+
460
+
461
+ def patch_path_column(
462
+ rows: list[dict[str, str]], column: str, rename_lookup: dict[str, str]
463
+ ) -> None:
464
+ """Rewrite a single relative-path column of ``rows`` in place via
465
+ ``rename_lookup``. A path with no entry is left unchanged."""
466
+ for row in rows:
467
+ old = row.get(column, "")
468
+ if old:
469
+ row[column] = rename_lookup.get(old, old)
470
+
471
+
472
+ # ---------------------------------------------------------------------------
473
+ # Rename column, QC exclusion, and scans.tsv manifest patching for the mri
474
+ # modality bidsmap maps.
475
+ # ---------------------------------------------------------------------------
476
+
477
+ # QC filter: boolean columns where "FALSE" means exclude from copy.
478
+ _EXCLUDE_IF_FALSE = ("recommend_for_use", "complete", "usable")
479
+
480
+ # QC filter: qc_rating values that exclude a file from copy.
481
+ _EXCLUDE_QC_RATINGS = {"FAIL", "UNCERTAIN"}
482
+
483
+ # Columns dropped from the output scans.tsv produced by bidsmap -o.
484
+ # rename/physio: both have already been applied (rename to bids_name, physio
485
+ # by xnatbidscli physioconvert, which must run before bidsmap -o).
486
+ _SCANS_DROP_COLS = frozenset({"rename", "physio"})
487
+
488
+
489
+ def _load_rename_map(scans_tsv: Path) -> tuple[dict[str, str], list[str]]:
490
+ """Read mriconvert_qc.tsv-style TSV and return
491
+ ``({rel_posix_path: new_bids_name}, warnings)`` for rows with a
492
+ non-empty ``rename`` column, keyed by ``filename``. Returns an empty map
493
+ if the file does not exist or cannot be read.
494
+ """
495
+ rename_map: dict[str, str] = {}
496
+ warnings: list[str] = []
497
+ if not scans_tsv.is_file():
498
+ return rename_map, warnings
499
+ try:
500
+ with scans_tsv.open(newline="", encoding="utf-8") as f:
501
+ for row in csv.DictReader(f, delimiter="\t"):
502
+ raw_path = (row.get("filename") or "").strip()
503
+ rename = (row.get("rename") or "").strip()
504
+ if raw_path and rename:
505
+ rename_map[raw_path] = rename
506
+ except OSError as exc:
507
+ warnings.append(f"WARNING: could not read {scans_tsv}: {exc}")
508
+ return rename_map, warnings
509
+
510
+
511
+ def _load_scans_levels(scans_json: Path) -> dict[str, set[str]]:
512
+ """Read scans.json and return ``{column: {valid_level, ...}}`` for columns
513
+ that define a ``Levels`` dict. Returns an empty dict if the file is absent
514
+ or unreadable.
515
+ """
516
+ if not scans_json.is_file():
517
+ return {}
518
+ try:
519
+ with scans_json.open(encoding="utf-8") as f:
520
+ data = json.load(f)
521
+ except (OSError, json.JSONDecodeError):
522
+ return {}
523
+ return {
524
+ col: set(info["Levels"].keys())
525
+ for col, info in data.items()
526
+ if isinstance(info, dict) and "Levels" in info
527
+ }
528
+
529
+
530
+ def _load_exclusion_info(
531
+ scans_tsv: Path,
532
+ scans_json: Path,
533
+ ) -> tuple[set[str], list[str]]:
534
+ """Read mriconvert_qc.tsv and return ``(excluded, warnings)``.
535
+
536
+ ``excluded`` is the set of ``filename`` values whose row triggers a QC
537
+ exclusion: ``recommend_for_use``, ``complete``, or ``usable`` ==
538
+ ``"FALSE"`` (exact, case-sensitive), or ``qc_rating`` in
539
+ ``{"FAIL", "UNCERTAIN"}``.
540
+
541
+ ``warnings`` contains:
542
+ - One notice per excluded row listing all triggered criteria.
543
+ - One notice per non-empty cell whose value is not a valid Level for that
544
+ column (case mismatch or typo), since such entries are silently ignored
545
+ by the exclusion check.
546
+ """
547
+ excluded: set[str] = set()
548
+ warnings: list[str] = []
549
+
550
+ levels = _load_scans_levels(scans_json)
551
+
552
+ if not scans_tsv.is_file():
553
+ return excluded, warnings
554
+
555
+ try:
556
+ with scans_tsv.open(newline="", encoding="utf-8") as f:
557
+ rows = list(csv.DictReader(f, delimiter="\t"))
558
+ except OSError as exc:
559
+ warnings.append(f"WARNING: could not read {scans_tsv}: {exc}")
560
+ return excluded, warnings
561
+
562
+ for row in rows:
563
+ label = (row.get("filename") or "").strip()
564
+
565
+ # Warn about non-empty values that do not match any valid Level.
566
+ for col, valid_vals in levels.items():
567
+ val = (row.get(col) or "").strip()
568
+ if val and val not in valid_vals:
569
+ warnings.append(
570
+ f"WARNING: {label}: column {col!r} has unrecognized value "
571
+ f"{val!r} (valid Levels: {sorted(valid_vals)}); "
572
+ "this entry will be ignored for QC filtering."
573
+ )
574
+
575
+ if not label:
576
+ continue
577
+
578
+ # Collect all triggered exclusion criteria for this row.
579
+ reasons: list[str] = []
580
+ for col in _EXCLUDE_IF_FALSE:
581
+ if (row.get(col) or "").strip() == "FALSE":
582
+ reasons.append(f"{col}=FALSE")
583
+ qc = (row.get("qc_rating") or "").strip()
584
+ if qc in _EXCLUDE_QC_RATINGS:
585
+ reasons.append(f"qc_rating={qc!r}")
586
+
587
+ if reasons:
588
+ excluded.add(label)
589
+ warnings.append(
590
+ f"WARNING: {label} excluded from copy "
591
+ f"({', '.join(reasons)})."
592
+ )
593
+
594
+ return excluded, warnings
595
+
596
+
597
+ def _update_participants_tsv(
598
+ dest_participants: Path,
599
+ participant_map: dict[str, str],
600
+ ) -> None:
601
+ """Rewrite participant_id values in the output participants.tsv."""
602
+ if not dest_participants.is_file() or not participant_map:
603
+ return
604
+ try:
605
+ with dest_participants.open(newline="", encoding="utf-8") as f:
606
+ reader = csv.DictReader(f, delimiter="\t")
607
+ fieldnames = list(reader.fieldnames or [])
608
+ rows = list(reader)
609
+ except OSError as exc:
610
+ print(f"WARNING: could not read {dest_participants}: {exc}", file=sys.stderr)
611
+ return
612
+ if "participant_id" not in fieldnames:
613
+ return
614
+ patch_id_columns(rows, participant_map, {}, session_col=None)
615
+ try:
616
+ with dest_participants.open("w", newline="", encoding="utf-8") as f:
617
+ writer = csv.DictWriter(
618
+ f, fieldnames=fieldnames, delimiter="\t", extrasaction="ignore",
619
+ )
620
+ writer.writeheader()
621
+ writer.writerows(rows)
622
+ except OSError as exc:
623
+ print(f"WARNING: could not update {dest_participants}: {exc}", file=sys.stderr)
624
+
625
+
626
+ def _update_output_scans_tsv(
627
+ source_scans: Path,
628
+ dest_scans: Path,
629
+ participant_map: dict[str, str],
630
+ session_map: dict[tuple[str, str], str],
631
+ rename_map: dict[str, str],
632
+ excluded: set[str],
633
+ rename_lookup: dict[str, str],
634
+ ) -> None:
635
+ """Read ``source_scans`` (e.g. ``INPUT_DIR/PROJECT-<PROJECT>_mriconvert_qc.tsv``) and
636
+ write the renamed/QC-filtered table to ``dest_scans`` (BIDS's canonical
637
+ ``OUTPUT_DIR/PROJECT/scans.tsv``), promoting the manifest to its final
638
+ name in the mapped output.
639
+
640
+ Updates ``filename`` (via ``rename_lookup``), ``bids_name``,
641
+ ``participant_id``, and ``session_id`` to their renamed values, omits rows
642
+ for files in ``excluded`` (those files were not copied), and drops the
643
+ columns in ``_SCANS_DROP_COLS`` from the output. All other reviewer
644
+ columns are preserved.
645
+ """
646
+ if not source_scans.is_file():
647
+ return
648
+ try:
649
+ with source_scans.open(newline="", encoding="utf-8") as f:
650
+ reader = csv.DictReader(f, delimiter="\t")
651
+ fieldnames = list(reader.fieldnames or [])
652
+ rows = list(reader)
653
+ except OSError:
654
+ return
655
+
656
+ stem_rename = build_stem_rename(rename_map, _MAIN_EXT)
657
+
658
+ out_rows: list[dict] = []
659
+ for row in rows:
660
+ old_filename = (row.get("filename") or "").strip()
661
+ if not old_filename or old_filename in excluded:
662
+ continue
663
+
664
+ old_path = Path(old_filename)
665
+ old_file = old_path.name
666
+ old_stem = old_file[: -len(_MAIN_EXT)] if old_file.endswith(_MAIN_EXT) else old_file
667
+ old_parent_posix = old_path.parent.as_posix()
668
+
669
+ if (old_parent_posix, old_stem) in stem_rename:
670
+ row["bids_name"] = stem_rename[(old_parent_posix, old_stem)]
671
+
672
+ row["filename"] = rename_lookup.get(old_filename, old_filename)
673
+ out_rows.append(row)
674
+
675
+ patch_id_columns(out_rows, participant_map, session_map)
676
+
677
+ out_fieldnames = [f for f in fieldnames if f not in _SCANS_DROP_COLS]
678
+ try:
679
+ with dest_scans.open("w", newline="", encoding="utf-8") as f:
680
+ writer = csv.DictWriter(
681
+ f, fieldnames=out_fieldnames, delimiter="\t", extrasaction="ignore",
682
+ )
683
+ writer.writeheader()
684
+ writer.writerows(out_rows)
685
+ except OSError as exc:
686
+ print(f"WARNING: could not update {dest_scans}: {exc}", file=sys.stderr)
687
+
688
+
689
+ def bidsmap_cmd(args: argparse.Namespace) -> int:
690
+ input_root = Path(args.input).resolve()
691
+ if not input_root.is_dir():
692
+ sys.exit(f"Error: input directory not found: {input_root}")
693
+
694
+ project = args.project
695
+ bids_dir = input_root / project
696
+ if not bids_dir.is_dir():
697
+ sys.exit(f"Error: BIDS dataset for project {project!r} not found at {bids_dir}")
698
+
699
+ # --- Step 1: always generate/update the map TSV ---
700
+ rows, has_sessions, skipped = scan_pairs(bids_dir)
701
+ if not rows:
702
+ sys.exit(f"Error: no sub-* directories found in {bids_dir}; nothing to map.")
703
+ if skipped:
704
+ print(
705
+ "Warning: the dataset uses sessions but these participants have no "
706
+ f"ses-* subdirectory and were skipped: {', '.join(skipped)}",
707
+ file=sys.stderr,
708
+ )
709
+
710
+ fresh = blank_map(rows, has_sessions)
711
+
712
+ map_path = input_root / f"PROJECT-{project}_bidsmap.tsv"
713
+ if map_path.exists():
714
+ merged, added = merge_existing(fresh, map_path)
715
+ merged.to_csv(map_path, sep="\t", index=False, na_rep="")
716
+ print(
717
+ f"Updated {map_path} ({added} new row{'s' if added != 1 else ''} "
718
+ f"added, {len(merged)} total)."
719
+ )
720
+ else:
721
+ fresh.to_csv(map_path, sep="\t", index=False, na_rep="")
722
+ print(f"Wrote {map_path} ({len(fresh)} rows).")
723
+
724
+ if not getattr(args, "output", None):
725
+ return 0
726
+
727
+ # --- Step 2: copy-with-rename to the output directory ---
728
+ output_root = Path(args.output).resolve()
729
+ dest_bids = output_root / project
730
+ existing_sessions = existing_session_dirs(dest_bids)
731
+
732
+ participant_map, session_map = load_participant_session_map(map_path)
733
+
734
+ all_warnings: list[str] = []
735
+
736
+ source_scans = input_root / f"PROJECT-{project}_mriconvert_qc.tsv"
737
+ source_scans_json = input_root / f"PROJECT-{project}_mriconvert_qc.json"
738
+ skip_root_files = frozenset({
739
+ "mriconvert_qc.tsv", "mriconvert_qc.json", "physioconvert_qc.tsv", "physioconvert_qc.json",
740
+ })
741
+
742
+ rename_map, rename_warnings = _load_rename_map(source_scans)
743
+ all_warnings.extend(rename_warnings)
744
+ for w in rename_warnings:
745
+ print(w)
746
+
747
+ # QC filter: determine which files to exclude before building the copy plan.
748
+ excluded, qc_warnings = _load_exclusion_info(source_scans, source_scans_json)
749
+ all_warnings.extend(qc_warnings)
750
+ for w in qc_warnings:
751
+ print(w)
752
+
753
+ excluded_stems = build_excluded_stems(excluded, _MAIN_EXT)
754
+
755
+ plan = build_rename_copy_plan(
756
+ bids_dir, participant_map, session_map, rename_map, excluded_stems,
757
+ main_ext=_MAIN_EXT, sidecar_exts=_SIDECAR_EXTS, skip_root_files=skip_root_files,
758
+ )
759
+ plan, collision_warnings = check_collisions(plan)
760
+ all_warnings.extend(collision_warnings)
761
+ for w in collision_warnings:
762
+ print(w)
763
+
764
+ to_copy, already_mapped = partition_plan_for_incremental(plan, dest_bids)
765
+
766
+ # Loudly flag any session that already exists in the output and is
767
+ # about to receive additional, previously-unmapped files.
768
+ touched_existing_sessions = sorted(
769
+ {
770
+ key
771
+ for _, dest_rel in to_copy
772
+ if (key := session_key(dest_rel)) is not None
773
+ and key in existing_sessions
774
+ }
775
+ )
776
+ session_warnings = [
777
+ f"WARNING: session {dest_bids / key} already exists; mapping in "
778
+ f"{sum(1 for _, d in to_copy if session_key(d) == key)} new file(s) "
779
+ "to it."
780
+ for key in touched_existing_sessions
781
+ ]
782
+ all_warnings.extend(session_warnings)
783
+ for w in session_warnings:
784
+ print(w)
785
+
786
+ copy_warnings = execute_copy_plan(to_copy, dest_bids)
787
+ all_warnings.extend(copy_warnings)
788
+ for w in copy_warnings:
789
+ print(w)
790
+
791
+ rename_lookup = rename_lookup_from_plan(plan, bids_dir)
792
+ # Promote mriconvert_qc.tsv/mriconvert_qc.json to BIDS's canonical scans.tsv/scans.json
793
+ # in the mapped output (mriconvert_qc.tsv itself is not copied verbatim).
794
+ _update_output_scans_tsv(
795
+ source_scans, dest_bids / "scans.tsv", participant_map, session_map,
796
+ rename_map, excluded, rename_lookup,
797
+ )
798
+ if source_scans_json.is_file():
799
+ shutil.copyfile(source_scans_json, dest_bids / "scans.json")
800
+ _update_participants_tsv(dest_bids / "participants.tsv", participant_map)
801
+
802
+ print(
803
+ f"Copied {len(to_copy)} file(s) to {dest_bids} "
804
+ f"({len(already_mapped)} file(s) already mapped and left untouched, "
805
+ f"{len(participant_map)} participant rename(s), "
806
+ f"{len(session_map)} session rename(s), "
807
+ f"{len(rename_map)} file rename(s), "
808
+ f"{len(excluded)} file(s) excluded by QC filter)."
809
+ )
810
+
811
+ if all_warnings:
812
+ sep = "=" * 60
813
+ print(f"\n{sep}")
814
+ print(f" {len(all_warnings)} WARNING(s) from bidsmap -o:")
815
+ print(sep)
816
+ for w in all_warnings:
817
+ print(f" {w}")
818
+ print(sep)
819
+
820
+ return 0