xnatbidscli 2.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,845 @@
1
+ import argparse
2
+ import contextlib
3
+ import csv
4
+ import gzip
5
+ import io
6
+ import json
7
+ import logging
8
+ import re
9
+ import shutil
10
+ import sys
11
+ import tempfile
12
+ import time
13
+ from concurrent.futures import ProcessPoolExecutor, as_completed
14
+ from datetime import datetime
15
+ from pathlib import Path
16
+
17
+ from .sysinfo import get_system_username
18
+
19
+ # QC/status manifest written to OUTPUT_DIR, alongside mriconvert's
20
+ # PROJECT-<PROJECT>_mriconvert_qc.tsv. Fully regenerated every run from the
21
+ # current mriconvert_qc.tsv content -- a visual reference for an expert reviewer,
22
+ # not a store consulted by a later run.
23
+ _QC_FILENAME = "physioconvert_qc.tsv"
24
+
25
+ # Static data-dictionary sidecar copied next to the TSV on each run.
26
+ _QC_DICT_FILENAME = "physioconvert_qc.json"
27
+
28
+ # BIDS suffix for continuous physiological recordings.
29
+ _SUFFIX = "physio"
30
+
31
+ # physioconvert_qc.tsv holds: the physio basename used, keyed first, and the
32
+ # regenerated status/metrics for its conversion.
33
+ _QC_COLUMNS = [
34
+ "physio", "status", "n_channels", "sampling_frequencies",
35
+ "sample_count", "duration_seconds",
36
+ ]
37
+
38
+ _NON_ALNUM = re.compile(r"[^A-Za-z0-9]")
39
+
40
+ # A physio recording captures the whole run, not one echo of a multi-echo
41
+ # scan, so an ``echo-<N>`` entity inherited from the paired scan's bids_name
42
+ # must not carry over into the physio output's filename.
43
+ _ECHO_ENTITY = re.compile(r"_echo-\d+")
44
+
45
+ STATUS_CONVERTED = "CONVERTED"
46
+ STATUS_NOT_PHYSIO = "NOT_PHYSIO"
47
+ STATUS_READER_MISSING = "READER_MISSING"
48
+ STATUS_CONVERT_ERROR = "CONVERT_ERROR"
49
+ STATUS_SOURCE_MISSING = "SOURCE_MISSING"
50
+ STATUS_COLLISION = "COLLISION"
51
+ STATUS_SKIPPED = "SKIPPED"
52
+
53
+ # Optional reader packages phys2bids imports lazily per format. A missing one
54
+ # is an environment problem, not a sign the file is not physiological data.
55
+ _READER_PACKAGES = {
56
+ ".acq": "bioread",
57
+ ".mat": "scipy",
58
+ ".smr": "sonpy",
59
+ }
60
+
61
+
62
+ def _logging_now() -> str:
63
+ """Local timestamp matching the other subcommands' log ``DATESTAMP``."""
64
+ now = datetime.now()
65
+ return f"{now.strftime('%Y-%m-%d %H:%M:%S')},{now.microsecond // 1000:03d}"
66
+
67
+
68
+ class _LogWriter:
69
+ """Append per-association rows to a CSV log, mirroring the other
70
+ subcommands.
71
+
72
+ A no-op when ``path`` is ``None`` (logging disabled). The header is
73
+ ``DATESTAMP,USER,STATUS,MRI_FILENAME,PHYSIO_SOURCE,DESTINATION_PATH``;
74
+ physioconvert places files serially in the main process, so no lock is
75
+ needed. An association that produced several outputs (a multi-frequency
76
+ split) emits one row per destination; one with no output emits a single
77
+ row with a blank ``DESTINATION_PATH``.
78
+ """
79
+
80
+ def __init__(self, path: Path | None):
81
+ self._path = path
82
+ self._user = get_system_username()
83
+ if path is not None:
84
+ path.parent.mkdir(parents=True, exist_ok=True)
85
+ with path.open("w", newline="") as f:
86
+ csv.writer(f).writerow(
87
+ ["DATESTAMP", "USER", "STATUS", "MRI_FILENAME", "PHYSIO_SOURCE", "DESTINATION_PATH"]
88
+ )
89
+
90
+ def write(
91
+ self,
92
+ datestamp: str,
93
+ filename: str,
94
+ status: str,
95
+ physio_source: str,
96
+ destinations: list[str] | None = None,
97
+ ) -> None:
98
+ """Append one row per destination path (or a single blank-dest row)."""
99
+ if self._path is None:
100
+ return
101
+ dests = destinations or [""]
102
+ with self._path.open("a", newline="") as f:
103
+ writer = csv.writer(f)
104
+ for dest in dests:
105
+ writer.writerow(
106
+ [datestamp, self._user, status, filename, physio_source, dest]
107
+ )
108
+
109
+
110
+ class _StdioTee:
111
+ """Duplicate writes to both an underlying stream and a log file, the
112
+ Python equivalent of piping through ``tee``."""
113
+
114
+ def __init__(self, stream, log_file):
115
+ self._stream = stream
116
+ self._log_file = log_file
117
+
118
+ def write(self, data: str) -> int:
119
+ self._log_file.write(data)
120
+ return self._stream.write(data)
121
+
122
+ def flush(self) -> None:
123
+ self._stream.flush()
124
+ self._log_file.flush()
125
+
126
+ def isatty(self) -> bool:
127
+ return self._stream.isatty()
128
+
129
+
130
+ def _fmt_freq(freq: float) -> str:
131
+ """Format a sampling frequency without trailing zeros (e.g. ``1000``)."""
132
+ return f"{freq:g}"
133
+
134
+
135
+ def _load_blueprint(path: Path):
136
+ """Load a physio file with the phys2bids loader for its extension.
137
+
138
+ Returns ``(blueprint, error, reader_missing)``: a phys2bids
139
+ ``BlueprintInput`` with ``(None, False)`` on success, or ``None`` with a
140
+ message. ``reader_missing`` is True when the failure is an ``ImportError``
141
+ for the optional reader phys2bids needs for this format (e.g. ``bioread``
142
+ for ``.acq``) — an environment problem, not a sign the file is not physio.
143
+ """
144
+ from phys2bids import io as p2b_io
145
+
146
+ loaders = {
147
+ ".acq": p2b_io.load_acq,
148
+ ".txt": p2b_io.load_txt,
149
+ ".mat": p2b_io.load_mat,
150
+ ".gep": p2b_io.load_gep,
151
+ ".smr": p2b_io.load_smr,
152
+ }
153
+ ext = path.suffix.lower()
154
+ loader = loaders.get(ext)
155
+ if loader is None:
156
+ return None, f"unsupported extension {path.suffix}", False
157
+ try:
158
+ blueprint = loader(str(path))
159
+ except ImportError as exc:
160
+ package = _READER_PACKAGES.get(ext, "the required reader")
161
+ return (
162
+ None,
163
+ f"reader package not installed ({exc}); install {package} to read "
164
+ f"{ext} files",
165
+ True,
166
+ )
167
+ except Exception as exc: # noqa: BLE001 - any other load failure = not physio
168
+ return None, str(exc), False
169
+ return blueprint, None, False
170
+
171
+
172
+ def _blueprint_info(blueprint) -> tuple[str, str, str, str, bool]:
173
+ """Channel count, per-frequency metrics, and an is-physio flag.
174
+
175
+ phys2bids stores one 1-D timeseries per channel (the time channel first),
176
+ each tagged with its own sampling frequency, so channels may differ in both
177
+ frequency and length. The sampling frequencies, sample counts, and durations
178
+ are returned as comma-separated lists aligned by unique frequency (ascending):
179
+ each entry's ``sample_count`` is the longest channel recorded at that
180
+ frequency and its ``duration_seconds`` is ``sample_count / frequency`` in
181
+ seconds at 0.001 s precision. A successfully loaded file is treated as
182
+ physiological when it has at least one channel and a positive sampling
183
+ frequency.
184
+
185
+ Returns ``(n_channels, sampling_frequencies, sample_count, duration_seconds,
186
+ is_physio)`` with the three metric strings sharing the same order/length.
187
+ """
188
+ # Pair each channel's frequency with the longest sample count seen at it.
189
+ # Channels sharing a frequency should share a length; keep the max if not.
190
+ counts_by_freq: dict[float, int] = {}
191
+ timeseries = getattr(blueprint, "timeseries", None) or []
192
+ freq_list = getattr(blueprint, "freq", None) or []
193
+ for series, freq in zip(timeseries, freq_list):
194
+ try:
195
+ f, n = float(freq), int(len(series))
196
+ except Exception: # noqa: BLE001
197
+ continue
198
+ counts_by_freq[f] = max(counts_by_freq.get(f, 0), n)
199
+
200
+ if counts_by_freq:
201
+ freqs = sorted(counts_by_freq)
202
+ sample_count = ",".join(str(counts_by_freq[f]) for f in freqs)
203
+ duration_seconds = ",".join(
204
+ f"{counts_by_freq[f] / f:.3f}" if f > 0 else "" for f in freqs
205
+ )
206
+ else:
207
+ # No timeseries lengths available; still report the raw frequencies so a
208
+ # loaded-but-empty blueprint is recognized as physio where appropriate.
209
+ try:
210
+ freqs = sorted({float(f) for f in freq_list})
211
+ except Exception: # noqa: BLE001
212
+ freqs = []
213
+ sample_count = ""
214
+ duration_seconds = ""
215
+
216
+ try:
217
+ n_channels = int(blueprint.ch_amount)
218
+ except Exception: # noqa: BLE001
219
+ n_channels = len(getattr(blueprint, "ch_name", []) or [])
220
+ sampling_frequencies = ",".join(_fmt_freq(f) for f in freqs)
221
+ is_physio = n_channels >= 1 and any(f > 0 for f in freqs)
222
+ return (
223
+ str(n_channels),
224
+ sampling_frequencies,
225
+ sample_count,
226
+ duration_seconds,
227
+ is_physio,
228
+ )
229
+
230
+
231
+ def _recording_label(stem: str, base: str, index: int) -> str:
232
+ """Recording label for one of several per-frequency phys2bids outputs.
233
+
234
+ phys2bids names multi-frequency outputs ``<base>_<freq>Hz``; the trailing
235
+ ``<freq>Hz`` (sanitized to alphanumerics) becomes the ``recording-`` label,
236
+ falling back to ``rec<N>`` if the suffix cannot be recovered.
237
+ """
238
+ if stem.startswith(base + "_"):
239
+ label = _NON_ALNUM.sub("", stem[len(base) + 1 :])
240
+ if label:
241
+ return label
242
+ return f"rec{index + 1}"
243
+
244
+
245
+ def _physio_basename(
246
+ participant_id: str, session_id: str, entity_name: str, recording: str | None
247
+ ) -> str:
248
+ """Assemble the BIDS basename for a physio output paired with an mri scan.
249
+
250
+ ``entity_name`` is the associated mriconvert_qc.tsv row's ``rename`` (if set)
251
+ else ``bids_name`` — e.g. ``task-rest_bold``. Its trailing
252
+ underscore-delimited suffix token is replaced with ``physio`` (e.g.
253
+ ``task-rest_bold`` -> ``task-rest_physio``; a bare suffix like ``bold``
254
+ with no other tokens yields just ``physio``), with an optional
255
+ ``recording-<label>`` entity inserted just before it for a
256
+ multi-frequency phys2bids split. Any ``echo-<N>`` entity is dropped
257
+ first: a physio recording aligns with every echo of a multi-echo scan,
258
+ not just the one row it happened to be associated with.
259
+ """
260
+ entity_name = _ECHO_ENTITY.sub("", entity_name)
261
+ prefix = entity_name.rsplit("_", 1)[0] if "_" in entity_name else ""
262
+ parts = [participant_id]
263
+ if session_id:
264
+ parts.append(session_id)
265
+ if prefix:
266
+ parts.append(prefix)
267
+ if recording:
268
+ parts.append(f"recording-{recording}")
269
+ parts.append(_SUFFIX)
270
+ return "_".join(parts)
271
+
272
+
273
+ def _destination_dir(bids_root: Path, participant_id: str, session_id: str, datatype: str) -> Path:
274
+ """``bids_root/participant_id/[session_id/]datatype`` — the same
275
+ directory the paired ``.nii.gz`` lives in."""
276
+ out = bids_root / participant_id
277
+ if session_id:
278
+ out = out / session_id
279
+ return out / datatype
280
+
281
+
282
+ def _is_valid_gzip(path: Path) -> bool:
283
+ """True if ``path`` decompresses cleanly as GZIP from start to end."""
284
+ try:
285
+ with gzip.open(path, "rb") as f:
286
+ while f.read(1024 * 1024):
287
+ pass
288
+ return True
289
+ except Exception: # noqa: BLE001 - any decompression failure = corrupt
290
+ return False
291
+
292
+
293
+ def _already_converted(bids_root: Path, row: dict[str, str]) -> bool:
294
+ """True if this association's ``_physio.tsv.gz`` already exists at its
295
+ expected destination and is not a corrupt GZIP file.
296
+
297
+ Checked only against the single-frequency basename (no ``recording-``
298
+ label): a prior multi-frequency split is not recognized here and that
299
+ association is always reconverted.
300
+ """
301
+ entity_name = row["rename"].strip() or row["bids_name"].strip()
302
+ dest_dir = _destination_dir(
303
+ bids_root, row["participant_id"], row["session_id"], row["datatype"]
304
+ )
305
+ basename = _physio_basename(row["participant_id"], row["session_id"], entity_name, None)
306
+ dest_tsv = dest_dir / f"{basename}.tsv.gz"
307
+ return dest_tsv.is_file() and _is_valid_gzip(dest_tsv)
308
+
309
+
310
+ def _find_collisions(rows: list[dict[str, str]]) -> dict[str, list[str]]:
311
+ """``{physio_basename: [filenames]}`` for every ``physio`` value
312
+ referenced by more than one row (rows must already be filtered to
313
+ non-blank ``physio``)."""
314
+ by_physio: dict[str, list[str]] = {}
315
+ for row in rows:
316
+ by_physio.setdefault(row["physio"].strip(), []).append(row["filename"])
317
+ return {k: v for k, v in by_physio.items() if len(v) > 1}
318
+
319
+
320
+ def _read_physio_parent(mriconvert_qc_json: Path) -> Path | None:
321
+ """Read ``PhysioParent.Value`` from mriconvert_qc.json.
322
+
323
+ Returns ``None`` if the file is absent/unreadable, the value is blank, or
324
+ the path is not a directory on disk.
325
+ """
326
+ if not mriconvert_qc_json.is_file():
327
+ return None
328
+ try:
329
+ with mriconvert_qc_json.open(encoding="utf-8") as f:
330
+ data = json.load(f)
331
+ except (OSError, ValueError):
332
+ return None
333
+ value = data.get("PhysioParent", {}).get("Value", "")
334
+ if not value:
335
+ return None
336
+ path = Path(value)
337
+ return path if path.is_dir() else None
338
+
339
+
340
+ def _clear_root_logging_handlers() -> None:
341
+ """Close and drop every handler on the root logger.
342
+
343
+ phys2bids configures logging via ``logging.basicConfig(handlers=[...])``
344
+ on each call, which is a no-op once the root logger already has handlers.
345
+ A pooled worker process runs many conversions in turn, so without this the
346
+ *first* conversion's handlers (bound to that call's redirected streams and
347
+ log file) would silently keep receiving every later conversion's messages.
348
+ Closing first releases the per-conversion log file phys2bids opens inside
349
+ the staging directory, so it doesn't block that directory's removal.
350
+ """
351
+ root = logging.getLogger()
352
+ for handler in root.handlers[:]:
353
+ handler.close()
354
+ root.removeHandler(handler)
355
+
356
+
357
+ def _run_phys2bids_to_staging(file_path: Path) -> tuple[str | None, str | None, str]:
358
+ """Run the phys2bids workflow for one file into a fresh staging directory.
359
+
360
+ This is the slow part of conversion — running phys2bids — isolated so it can
361
+ be parallelized across processes. phys2bids prints progress and logging
362
+ output of its own directly to stdout/stderr; when several of these run
363
+ concurrently in worker processes that output interleaves unreadably on the
364
+ shared terminal. It's captured here instead and handed back to the caller,
365
+ which prints it once this conversion's own result line is written, keeping
366
+ each association's output together and in the deterministic order results
367
+ are drained in.
368
+
369
+ The placement of phys2bids's output into the BIDS tree happens later,
370
+ serially, in the main process. Returns ``(staging_dir, error, output)``:
371
+ ``staging_dir`` is ``None`` on failure; ``output`` is the captured text
372
+ (may be empty). The staging directory is left for the caller to consume
373
+ and remove.
374
+ """
375
+ from phys2bids.phys2bids import phys2bids as run_phys2bids
376
+
377
+ staging = Path(tempfile.mkdtemp(prefix="xnatbidscli_phys2bids_"))
378
+ captured = io.StringIO()
379
+ _clear_root_logging_handlers()
380
+ try:
381
+ with contextlib.redirect_stdout(captured), contextlib.redirect_stderr(captured):
382
+ try:
383
+ run_phys2bids(
384
+ filename=file_path.name,
385
+ indir=str(file_path.parent),
386
+ outdir=str(staging),
387
+ quiet=True,
388
+ )
389
+ except Exception as exc: # noqa: BLE001 - surface phys2bids failures
390
+ shutil.rmtree(staging, ignore_errors=True)
391
+ return None, f"phys2bids failed: {exc}", captured.getvalue()
392
+ finally:
393
+ # Release the log file phys2bids opened in the staging directory
394
+ # before the caller may try to remove that directory.
395
+ _clear_root_logging_handlers()
396
+ if not list(staging.glob("*.tsv.gz")):
397
+ shutil.rmtree(staging, ignore_errors=True)
398
+ return None, "phys2bids produced no .tsv.gz output", captured.getvalue()
399
+ return str(staging), None, captured.getvalue()
400
+
401
+
402
+ def _run_worker(task: tuple[str, str]) -> dict:
403
+ """Validate and convert one physio association; the unit of work for
404
+ parallel execution.
405
+
406
+ Runs in a worker process (or the main process when serial): loads the raw
407
+ file to confirm it is physiological data and gather channel info, then —
408
+ if it is — runs phys2bids into a staging directory. Returns a picklable
409
+ dict; placement into the BIDS tree happens later in the main process.
410
+ """
411
+ filename, raw_path_str = task
412
+ path = Path(raw_path_str)
413
+ result = {
414
+ "filename": filename,
415
+ "start": _logging_now(),
416
+ "is_physio": False,
417
+ "reader_missing": False,
418
+ "err": None,
419
+ "n_ch": "",
420
+ "freqs": "",
421
+ "sample_count": "",
422
+ "duration_seconds": "",
423
+ "staging": None,
424
+ "convert_error": None,
425
+ "output": "",
426
+ }
427
+
428
+ blueprint, err, reader_missing = _load_blueprint(path)
429
+ if blueprint is not None:
430
+ (
431
+ result["n_ch"],
432
+ result["freqs"],
433
+ result["sample_count"],
434
+ result["duration_seconds"],
435
+ result["is_physio"],
436
+ ) = _blueprint_info(blueprint)
437
+ if not result["is_physio"]:
438
+ result["reader_missing"] = reader_missing
439
+ result["err"] = err
440
+ return result
441
+
442
+ result["staging"], result["convert_error"], result["output"] = _run_phys2bids_to_staging(path)
443
+ return result
444
+
445
+
446
+ def _place_at_destination(tsv_src: Path, dest_tsv: Path, bids_root: Path) -> str:
447
+ """Move a phys2bids ``.tsv.gz``/``.json`` pair from staging to
448
+ ``dest_tsv`` (which must not already exist). Returns the ``.tsv.gz``
449
+ path relative to ``bids_root`` (POSIX)."""
450
+ src_stem = tsv_src.name[: -len(".tsv.gz")]
451
+ shutil.move(str(tsv_src), str(dest_tsv))
452
+ json_src = tsv_src.with_name(src_stem + ".json")
453
+ if json_src.is_file():
454
+ shutil.move(
455
+ str(json_src), str(dest_tsv.with_name(dest_tsv.name[: -len(".tsv.gz")] + ".json"))
456
+ )
457
+ else:
458
+ print(f"WARNING: no JSON sidecar produced for {tsv_src.name}")
459
+ return dest_tsv.relative_to(bids_root).as_posix()
460
+
461
+
462
+ def _place_converted(
463
+ staging: Path,
464
+ bids_root: Path,
465
+ row: dict[str, str],
466
+ base_stem: str,
467
+ claimed: dict[str, str],
468
+ ) -> tuple[str, list[str], str | None]:
469
+ """Move a completed conversion's staged output into the mri BIDS tree.
470
+
471
+ Places each produced ``.tsv.gz``/``.json`` pair at
472
+ ``_destination_dir(...)/_physio_basename(...).{tsv.gz,json}``, adding a
473
+ ``recording-<label>`` entity for a multi-frequency phys2bids split.
474
+
475
+ Every run reconverts from scratch, so a destination left over from an
476
+ earlier run of this same association is simply replaced. ``claimed``
477
+ tracks destinations already written by another association *this run*
478
+ (keyed by the owning row's ``filename``) so two distinct associations
479
+ can never overwrite each other, even if their computed names collide.
480
+ Returns ``(status, written_relpaths, detail)``. The staging directory is
481
+ removed when done.
482
+ """
483
+ try:
484
+ produced = sorted(staging.glob("*.tsv.gz"))
485
+ if not produced:
486
+ return STATUS_CONVERT_ERROR, [], "phys2bids produced no .tsv.gz output"
487
+
488
+ multi = len(produced) > 1
489
+ dest_dir = _destination_dir(
490
+ bids_root, row["participant_id"], row["session_id"], row["datatype"]
491
+ )
492
+ dest_dir.mkdir(parents=True, exist_ok=True)
493
+ entity_name = row["rename"].strip() or row["bids_name"].strip()
494
+ assoc_key = row["filename"]
495
+
496
+ written: list[str] = []
497
+ for index, tsv_src in enumerate(produced):
498
+ stem = tsv_src.name[: -len(".tsv.gz")]
499
+ recording = _recording_label(stem, base_stem, index) if multi else None
500
+ basename = _physio_basename(
501
+ row["participant_id"], row["session_id"], entity_name, recording
502
+ )
503
+ dest_tsv = dest_dir / f"{basename}.tsv.gz"
504
+ dest_rel = dest_tsv.relative_to(bids_root).as_posix()
505
+
506
+ owner = claimed.get(dest_rel)
507
+ if owner is not None and owner != assoc_key:
508
+ return (
509
+ STATUS_CONVERT_ERROR,
510
+ [],
511
+ "destination already occupied by a different association: "
512
+ f"{dest_rel}",
513
+ )
514
+ if dest_tsv.exists():
515
+ dest_tsv.unlink()
516
+ dest_json = dest_tsv.with_name(dest_tsv.name[: -len(".tsv.gz")] + ".json")
517
+ if dest_json.is_file():
518
+ dest_json.unlink()
519
+
520
+ claimed[dest_rel] = assoc_key
521
+ written.append(_place_at_destination(tsv_src, dest_tsv, bids_root))
522
+ return STATUS_CONVERTED, written, None
523
+ finally:
524
+ shutil.rmtree(staging, ignore_errors=True)
525
+
526
+
527
+ def _carry_row(physio: str, status: str) -> dict[str, str]:
528
+ """Build a QC row with blank metrics, for an association not (re)converted
529
+ this run (COLLISION, SOURCE_MISSING)."""
530
+ return {
531
+ "physio": physio,
532
+ "status": status,
533
+ "n_channels": "",
534
+ "sampling_frequencies": "",
535
+ "sample_count": "",
536
+ "duration_seconds": "",
537
+ }
538
+
539
+
540
+ def _find_asset(name: str) -> Path | None:
541
+ """Locate a static asset (data dictionary) in the dev tree or the wheel."""
542
+ here = Path(__file__).resolve().parent
543
+ for candidate in (
544
+ here.parent / "assets" / name, # dev: src/assets/
545
+ here / "assets" / name, # wheel: xnatbidscli/assets/
546
+ ):
547
+ if candidate.is_file():
548
+ return candidate
549
+ return None
550
+
551
+
552
+ def _write_qc_dict(dest: Path, physio_parent: Path | None) -> None:
553
+ """Write physioconvert_qc.json from the static asset, injecting
554
+ ``PhysioParent`` with this run's resolved value (mirroring mriconvert's
555
+ ``PhysioParent`` key in mriconvert_qc.json). Run once after all conversions.
556
+ A missing or unreadable asset is warned about, not fatal.
557
+ """
558
+ src = _find_asset(_QC_DICT_FILENAME)
559
+ if src is None:
560
+ print(f"WARNING: {_QC_DICT_FILENAME} data dictionary not found; skipping its copy.")
561
+ return
562
+ try:
563
+ with src.open(encoding="utf-8") as f:
564
+ data = json.load(f)
565
+ except (OSError, ValueError) as exc:
566
+ print(f"WARNING: could not read {src}: {exc}; skipping its copy.")
567
+ return
568
+
569
+ data.setdefault("PhysioParent", {})["Value"] = str(physio_parent) if physio_parent else ""
570
+
571
+ with dest.open("w", encoding="utf-8") as f:
572
+ json.dump(data, f, indent=4)
573
+ f.write("\n")
574
+
575
+
576
+ def _write_tsv(path: Path, rows: list[dict[str, str]], columns: list[str]) -> None:
577
+ """Write a TSV with the given columns, one row per association, sorted
578
+ by ``physio``."""
579
+ rows = sorted(rows, key=lambda r: r["physio"])
580
+ with path.open("w", newline="", encoding="utf-8") as f:
581
+ writer = csv.DictWriter(f, fieldnames=columns, delimiter="\t")
582
+ writer.writeheader()
583
+ writer.writerows(rows)
584
+
585
+
586
+ def physioconvert_cmd(args: argparse.Namespace) -> int:
587
+ if args.nphysio < 1:
588
+ sys.exit("Error: -n/--nphysio must be >= 1.")
589
+
590
+ output_dir = Path(args.output).resolve()
591
+ bids_root = output_dir / args.project
592
+ if not bids_root.is_dir():
593
+ sys.exit(
594
+ f"Error: BIDS dataset not found at {bids_root}; run xnatbidscli "
595
+ "mriconvert first."
596
+ )
597
+
598
+ mriconvert_qc_tsv = output_dir / f"PROJECT-{args.project}_mriconvert_qc.tsv"
599
+ if not mriconvert_qc_tsv.is_file():
600
+ sys.exit(
601
+ f"Error: {mriconvert_qc_tsv} not found; run xnatbidscli mriconvert first."
602
+ )
603
+
604
+ try:
605
+ import phys2bids # noqa: F401
606
+ from phys2bids import io # noqa: F401
607
+ from phys2bids.phys2bids import phys2bids as _p2b # noqa: F401
608
+ except ImportError:
609
+ sys.exit(
610
+ "Error: phys2bids is required for physioconvert. It installs with "
611
+ "xnatbidscli on Python 3.11 only until phys2bids releases support for "
612
+ "newer numpy; reinstall xnatbidscli under Python 3.11."
613
+ )
614
+
615
+ # physioconvert only reads mriconvert_qc.tsv -- it never writes it back.
616
+ with mriconvert_qc_tsv.open(newline="", encoding="utf-8") as f:
617
+ mri_rows = list(csv.DictReader(f, delimiter="\t"))
618
+
619
+ in_scope = [r for r in mri_rows if (r.get("physio") or "").strip()]
620
+ if not in_scope:
621
+ print(
622
+ "No physio associations found in mriconvert_qc.tsv's 'physio' column; "
623
+ "nothing to do."
624
+ )
625
+ return 0
626
+
627
+ qc_path = output_dir / f"PROJECT-{args.project}_{_QC_FILENAME}"
628
+ qc_dict_path = output_dir / f"PROJECT-{args.project}_{_QC_DICT_FILENAME}"
629
+
630
+ # Logs go straight under OUTPUT_DIR/log (not nested under PROJECT), since a
631
+ # single log directory can cover runs across multiple projects.
632
+ log_path: Path | None = None
633
+ text_log_path: Path | None = None
634
+ if args.log:
635
+ while True:
636
+ ts = datetime.now().strftime("%Y%m%d_%H%M%S")
637
+ log_dir = output_dir / "log"
638
+ log_path = log_dir / f"physioconvert_{ts}_log.csv"
639
+ text_log_path = log_dir / f"physioconvert_{ts}_log.txt"
640
+ if not log_path.exists() and not text_log_path.exists():
641
+ break
642
+ time.sleep(1)
643
+ log_writer = _LogWriter(log_path)
644
+
645
+ # Mirror everything this run prints to stdout/stderr into a plain-text
646
+ # log alongside the CSV, the Python equivalent of piping through ``tee``.
647
+ orig_stdout, orig_stderr = sys.stdout, sys.stderr
648
+ text_log_file = None
649
+ if text_log_path is not None:
650
+ text_log_path.parent.mkdir(parents=True, exist_ok=True)
651
+ text_log_file = text_log_path.open("w", encoding="utf-8")
652
+ sys.stdout = _StdioTee(orig_stdout, text_log_file)
653
+ sys.stderr = _StdioTee(orig_stderr, text_log_file)
654
+
655
+ try:
656
+ counts = {
657
+ STATUS_CONVERTED: 0,
658
+ STATUS_NOT_PHYSIO: 0,
659
+ STATUS_READER_MISSING: 0,
660
+ STATUS_CONVERT_ERROR: 0,
661
+ STATUS_SOURCE_MISSING: 0,
662
+ STATUS_COLLISION: 0,
663
+ STATUS_SKIPPED: 0,
664
+ }
665
+ qc_rows: list[dict[str, str]] = []
666
+ # Destinations already written by an association this run, keyed by
667
+ # relpath -> the writing row's filename; guards against two distinct
668
+ # associations computing the same output path.
669
+ claimed: dict[str, str] = {}
670
+
671
+ # --- Collisions: the same raw physio basename referenced by more than
672
+ # one mriconvert_qc.tsv row. None of those rows are converted until resolved.
673
+ collisions = _find_collisions(in_scope)
674
+ for row in in_scope:
675
+ physio = row["physio"].strip()
676
+ if physio not in collisions:
677
+ continue
678
+ filename = row["filename"]
679
+ counts[STATUS_COLLISION] += 1
680
+ print(f"{filename}: {STATUS_COLLISION} — physio {physio!r} referenced by multiple rows")
681
+ log_writer.write(_logging_now(), filename, STATUS_COLLISION, physio, [])
682
+ for physio, filenames in collisions.items():
683
+ print(
684
+ f"WARNING: physio {physio!r} is referenced by {len(filenames)} "
685
+ f"mriconvert_qc.tsv rows ({', '.join(filenames)}); none will be "
686
+ "converted until only one row references it."
687
+ )
688
+ qc_rows.append(_carry_row(physio, STATUS_COLLISION))
689
+
690
+ remaining = [r for r in in_scope if r["physio"].strip() not in collisions]
691
+ mriconvert_qc_json = output_dir / f"PROJECT-{args.project}_mriconvert_qc.json"
692
+ physio_parent = _read_physio_parent(mriconvert_qc_json)
693
+
694
+ # --- Classify each remaining association: needs conversion, or blocked. ---
695
+ to_convert: list[dict[str, str]] = []
696
+ for row in remaining:
697
+ filename = row["filename"]
698
+ physio = row["physio"].strip()
699
+
700
+ if not row["participant_id"] or not row["datatype"]:
701
+ counts[STATUS_SOURCE_MISSING] += 1
702
+ detail = "mriconvert_qc.tsv row has blank participant_id/datatype; cannot place physio output"
703
+ print(f"{filename}: {STATUS_SOURCE_MISSING} — {detail}")
704
+ log_writer.write(_logging_now(), filename, STATUS_SOURCE_MISSING, physio, [])
705
+ qc_rows.append(_carry_row(physio, STATUS_SOURCE_MISSING))
706
+ continue
707
+
708
+ if physio_parent is None:
709
+ counts[STATUS_SOURCE_MISSING] += 1
710
+ print(f"{filename}: {STATUS_SOURCE_MISSING} — PhysioParent not set/found in mriconvert_qc.json")
711
+ log_writer.write(_logging_now(), filename, STATUS_SOURCE_MISSING, physio, [])
712
+ qc_rows.append(_carry_row(physio, STATUS_SOURCE_MISSING))
713
+ continue
714
+
715
+ raw_path = physio_parent / physio
716
+ if not raw_path.is_file():
717
+ counts[STATUS_SOURCE_MISSING] += 1
718
+ print(f"{filename}: {STATUS_SOURCE_MISSING} — {physio!r} not found under PhysioParent")
719
+ log_writer.write(_logging_now(), filename, STATUS_SOURCE_MISSING, physio, [])
720
+ qc_rows.append(_carry_row(physio, STATUS_SOURCE_MISSING))
721
+ continue
722
+
723
+ if _already_converted(bids_root, row):
724
+ counts[STATUS_SKIPPED] += 1
725
+ print(f"{filename}: {STATUS_SKIPPED} — {physio!r} already converted at its destination")
726
+ log_writer.write(_logging_now(), filename, STATUS_SKIPPED, physio, [])
727
+ qc_rows.append(_carry_row(physio, STATUS_SKIPPED))
728
+ continue
729
+
730
+ to_convert.append(row)
731
+
732
+ # ``to_convert`` is placed in sorted filename order (not completion order)
733
+ # so results (and log/QC ordering) are deterministic regardless of -n.
734
+ to_convert.sort(key=lambda r: r["filename"])
735
+ tasks_by_filename = {r["filename"]: r for r in to_convert}
736
+ tasks = [
737
+ (r["filename"], str(physio_parent / r["physio"].strip())) for r in to_convert
738
+ ]
739
+
740
+ def _finish(row: dict[str, str], result: dict) -> None:
741
+ filename = row["filename"]
742
+ physio = row["physio"].strip()
743
+ start = result["start"]
744
+
745
+ if not result["is_physio"]:
746
+ status = STATUS_READER_MISSING if result["reader_missing"] else STATUS_NOT_PHYSIO
747
+ counts[status] += 1
748
+ print(f"{filename}: {status} — {result['err'] or 'no physio channels'}")
749
+ log_writer.write(start, filename, status, physio, [])
750
+ qc_rows.append(_carry_row(physio, status))
751
+ return
752
+
753
+ if result["convert_error"] is not None:
754
+ status, written, detail = STATUS_CONVERT_ERROR, [], result["convert_error"]
755
+ else:
756
+ base_stem = Path(physio).stem
757
+ status, written, detail = _place_converted(
758
+ Path(result["staging"]), bids_root, row, base_stem, claimed,
759
+ )
760
+
761
+ counts[status] += 1
762
+ line = f"{filename}: {status}"
763
+ if detail:
764
+ line += f" — {detail}"
765
+ elif written:
766
+ line += f" — wrote {len(written)} file(s)"
767
+ print(line)
768
+ output = result["output"].strip("\n")
769
+ if output:
770
+ for output_line in output.splitlines():
771
+ print(f" {output_line}")
772
+ log_writer.write(start, filename, status, physio, written)
773
+
774
+ qc_rows.append({
775
+ "physio": physio,
776
+ "status": status,
777
+ "n_channels": result["n_ch"],
778
+ "sampling_frequencies": result["freqs"],
779
+ "sample_count": result["sample_count"],
780
+ "duration_seconds": result["duration_seconds"],
781
+ })
782
+
783
+ if args.nphysio <= 1:
784
+ for task in tasks:
785
+ _finish(tasks_by_filename[task[0]], _run_worker(task))
786
+ else:
787
+ # phys2bids runs in worker processes (real parallelism, since it is an
788
+ # in-process Python library); placement stays serial in the main
789
+ # process and is drained in sorted-filename order (out-of-order
790
+ # completions are buffered until their turn), so results are fully
791
+ # deterministic regardless of -n.
792
+ with ProcessPoolExecutor(max_workers=args.nphysio) as ex:
793
+ fut_to_index = {ex.submit(_run_worker, t): i for i, t in enumerate(tasks)}
794
+ pending: dict[int, dict] = {}
795
+ next_index = 0
796
+ for fut in as_completed(fut_to_index):
797
+ pending[fut_to_index[fut]] = fut.result()
798
+ while next_index in pending:
799
+ _finish(tasks_by_filename[tasks[next_index][0]], pending.pop(next_index))
800
+ next_index += 1
801
+
802
+ _write_tsv(qc_path, qc_rows, _QC_COLUMNS)
803
+ _write_qc_dict(qc_dict_path, physio_parent)
804
+
805
+ total = sum(counts.values())
806
+ print(f"\nProcessed {total} physio association(s):")
807
+ for status in (
808
+ STATUS_CONVERTED,
809
+ STATUS_SKIPPED,
810
+ STATUS_NOT_PHYSIO,
811
+ STATUS_READER_MISSING,
812
+ STATUS_CONVERT_ERROR,
813
+ STATUS_SOURCE_MISSING,
814
+ STATUS_COLLISION,
815
+ ):
816
+ print(f" {status}: {counts[status]}")
817
+ print(f"{_QC_FILENAME} written to {qc_path}")
818
+ if log_path is not None:
819
+ print(f"Log written to {log_path}")
820
+ if text_log_path is not None:
821
+ print(f"Text log written to {text_log_path}")
822
+ if counts[STATUS_COLLISION]:
823
+ print(
824
+ "Some physio associations were skipped because their raw file is "
825
+ "referenced by more than one mriconvert_qc.tsv row (status COLLISION). "
826
+ "Clear all but one row's physio column and re-run."
827
+ )
828
+ if counts[STATUS_SOURCE_MISSING]:
829
+ print(
830
+ "Some physio associations could not be resolved to a file (status "
831
+ "SOURCE_MISSING). Check -y/--physio (mriconvert) and the "
832
+ "physio column and re-run."
833
+ )
834
+ if counts[STATUS_READER_MISSING]:
835
+ print(
836
+ "Some files could not be read because the reader package they "
837
+ "need is not installed (e.g. 'bioread' for .acq). Install it and "
838
+ "re-run."
839
+ )
840
+
841
+ return 1 if counts[STATUS_CONVERT_ERROR] or counts[STATUS_READER_MISSING] else 0
842
+ finally:
843
+ sys.stdout, sys.stderr = orig_stdout, orig_stderr
844
+ if text_log_file is not None:
845
+ text_log_file.close()