xnatbidscli 2.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,1009 @@
1
+ import argparse
2
+ import csv
3
+ import os
4
+ import re
5
+ import shutil
6
+ import sys
7
+ import threading
8
+ import time
9
+ import zipfile
10
+ from collections.abc import Callable
11
+ from concurrent.futures import FIRST_COMPLETED, ThreadPoolExecutor, wait
12
+ from datetime import datetime
13
+ from pathlib import Path
14
+ from urllib.parse import quote, unquote
15
+
16
+ import requests
17
+ from pyxnat import Interface
18
+ from pyxnat.core import uriutil
19
+
20
+ from .archive import (
21
+ OK_STATUSES as ARCHIVE_OK_STATUSES,
22
+ archive_experiment,
23
+ delete_experiment_dir,
24
+ )
25
+ from .login import load_credentials
26
+ from .sysinfo import get_system_username
27
+
28
+ STATUS_COMPLETE = "COMPLETE"
29
+ STATUS_FAILURE = "FAILURE"
30
+ STATUS_NONEXISTENT = "NONEXISTENT"
31
+ STATUS_EMPTY = "EMPTY"
32
+
33
+ _OK_STATUSES = {STATUS_COMPLETE, STATUS_EMPTY}
34
+
35
+ _thread_iface = threading.local()
36
+
37
+
38
+ def _enable_windows_ansi() -> None:
39
+ """Best-effort: turn on ANSI escape processing in legacy Windows consoles.
40
+
41
+ Windows Terminal and modern PowerShell already interpret these
42
+ sequences; this only matters for cmd.exe/conhost, which needs the
43
+ ENABLE_VIRTUAL_TERMINAL_PROCESSING console mode flag set explicitly.
44
+ """
45
+ if sys.platform != "win32":
46
+ return
47
+ try:
48
+ import ctypes
49
+
50
+ kernel32 = ctypes.windll.kernel32
51
+ handle = kernel32.GetStdHandle(-11) # STD_OUTPUT_HANDLE
52
+ mode = ctypes.c_uint32()
53
+ if kernel32.GetConsoleMode(handle, ctypes.byref(mode)):
54
+ kernel32.SetConsoleMode(handle, mode.value | 0x0004) # VT processing
55
+ except Exception:
56
+ pass
57
+
58
+
59
+ class _ProgressBoard:
60
+ """Keeps each active experiment's download-progress line updating in place.
61
+
62
+ Backed by ANSI cursor-movement escapes so `-n`'s concurrent downloads
63
+ each get one persistently-updating line instead of a fresh line every
64
+ interval. Falls back to plain sequential prints when stdout isn't a
65
+ terminal (e.g. redirected to a file), since in-place redraws wouldn't
66
+ render there anyway.
67
+ """
68
+
69
+ def __init__(self) -> None:
70
+ self._lock = threading.Lock()
71
+ self._lines: dict[str, str] = {}
72
+ self._rendered = 0
73
+ self._enabled = sys.stdout.isatty()
74
+ if self._enabled:
75
+ _enable_windows_ansi()
76
+
77
+ def update(self, label: str, text: str) -> None:
78
+ """Set (or add) `label`'s line to `text` and repaint the board."""
79
+ with self._lock:
80
+ if not self._enabled:
81
+ print(text)
82
+ return
83
+ self._lines[label] = text
84
+ self._redraw()
85
+
86
+ def finish(self, label: str) -> None:
87
+ """Remove `label`'s line once that experiment stops downloading."""
88
+ with self._lock:
89
+ if self._lines.pop(label, None) is not None and self._enabled:
90
+ self._redraw()
91
+
92
+ def log(self, msg: str) -> None:
93
+ """Print a normal message above the progress block, then redraw it."""
94
+ with self._lock:
95
+ if not self._enabled:
96
+ print(msg)
97
+ return
98
+ if self._rendered:
99
+ sys.stdout.write(f"\x1b[{self._rendered}A\x1b[0J")
100
+ sys.stdout.write(msg + "\n")
101
+ self._rendered = 0
102
+ self._redraw()
103
+
104
+ def _redraw(self) -> None:
105
+ """Repaint the progress block in place. Caller holds `self._lock`."""
106
+ if self._rendered:
107
+ sys.stdout.write(f"\x1b[{self._rendered}A")
108
+ for text in self._lines.values():
109
+ sys.stdout.write(f"\r\x1b[K{text}\n")
110
+ extra = self._rendered - len(self._lines)
111
+ if extra > 0:
112
+ for _ in range(extra):
113
+ sys.stdout.write("\x1b[K\n")
114
+ sys.stdout.write(f"\x1b[{extra}A")
115
+ self._rendered = len(self._lines)
116
+ sys.stdout.flush()
117
+
118
+
119
+ _board = _ProgressBoard()
120
+
121
+
122
+ def _safe_print(msg: str) -> None:
123
+ _board.log(msg)
124
+
125
+
126
+ def _logging_now() -> str:
127
+ # Matches logging module's default %(asctime)s: "YYYY-MM-DD HH:MM:SS,mmm"
128
+ now = datetime.now()
129
+ return f"{now.strftime('%Y-%m-%d %H:%M:%S')},{now.microsecond // 1000:03d}"
130
+
131
+
132
+ class _LogWriter:
133
+ def __init__(self, path: Path | None):
134
+ self._path = path
135
+ self._lock = threading.Lock()
136
+ self._user = get_system_username()
137
+ if path is not None:
138
+ path.parent.mkdir(parents=True, exist_ok=True)
139
+ with path.open("w", newline="") as f:
140
+ csv.writer(f).writerow(
141
+ [
142
+ "DATESTAMP",
143
+ "USER",
144
+ "PROJECT",
145
+ "SUBJECT",
146
+ "EXPERIMENT",
147
+ "STATUS",
148
+ ]
149
+ )
150
+
151
+ def write(
152
+ self,
153
+ datestamp: str,
154
+ project: str,
155
+ subject: str,
156
+ experiment: str,
157
+ status: str,
158
+ ) -> None:
159
+ if self._path is None:
160
+ return
161
+ with self._lock, self._path.open("a", newline="") as f:
162
+ csv.writer(f).writerow(
163
+ [datestamp, self._user, project, subject, experiment, status]
164
+ )
165
+
166
+
167
+ def _get_thread_interface(server: str, user: str, password: str) -> Interface:
168
+ iface = getattr(_thread_iface, "iface", None)
169
+ if iface is None:
170
+ iface = Interface(server=server, user=user, password=password)
171
+ _thread_iface.iface = iface
172
+ return iface
173
+
174
+
175
+ def _close_thread_interface() -> None:
176
+ iface = getattr(_thread_iface, "iface", None)
177
+ if iface is not None:
178
+ try:
179
+ iface.disconnect()
180
+ except Exception:
181
+ pass
182
+ _thread_iface.iface = None
183
+
184
+
185
+ def _describe_download_error(e: Exception) -> str:
186
+ """Return a clear message for an exception raised during a zip download.
187
+
188
+ pyxnat's own zip-download code (``downloadutils.download`` and
189
+ ``resources.CObject.download``) wraps ``response.iter_content()`` in a
190
+ bare ``except Exception as e: sys.stderr.write(e)``. Since ``write()``
191
+ requires a ``str``, that line itself raises a ``TypeError`` that masks
192
+ whatever actually broke the download (almost always a
193
+ ``requests.exceptions.ChunkedEncodingError`` from the server or a proxy
194
+ dropping the connection mid-transfer). Unwrap that TypeError's context
195
+ to surface the real cause instead of the confusing "write() argument
196
+ must be str" message.
197
+ """
198
+ context = e.__context__
199
+ if isinstance(e, TypeError) and isinstance(context, requests.exceptions.RequestException):
200
+ return (
201
+ f"connection dropped during zip download ({context}); "
202
+ "this is usually a transient network/server timeout, try again"
203
+ )
204
+ if isinstance(e, requests.exceptions.RequestException):
205
+ return (
206
+ f"connection dropped during zip download ({e}); "
207
+ "this is usually a transient network/server timeout, try again"
208
+ )
209
+ return str(e)
210
+
211
+
212
+ def _human_bytes(n: float) -> str:
213
+ """Format a byte count for display, e.g. ``1536`` -> ``"1.5 KB"``."""
214
+ for unit in ("B", "KB", "MB", "GB", "TB"):
215
+ if abs(n) < 1024.0:
216
+ return f"{n:3.1f} {unit}"
217
+ n /= 1024.0
218
+ return f"{n:.1f} PB"
219
+
220
+
221
+ class _ExperimentProgress:
222
+ """Tracks bytes downloaded so far for one experiment's zip transfers.
223
+
224
+ XNAT's zip-export endpoint is fetched as two sequential whole-archive
225
+ downloads (scans, then session-level resources); each one is written
226
+ directly to its final path with incremental flushing, so at most one
227
+ in-progress zip exists under the experiment directory at a time. This
228
+ tracks the completed phase's byte count plus whatever is currently
229
+ on disk, so reported progress climbs across both phases instead of
230
+ resetting to zero when the scans zip is extracted and deleted.
231
+ """
232
+
233
+ def __init__(self) -> None:
234
+ self._lock = threading.Lock()
235
+ self._completed_bytes = 0
236
+
237
+ def add_completed(self, n: int) -> None:
238
+ with self._lock:
239
+ self._completed_bytes += n
240
+
241
+ def current_bytes(self, experiment_root: Path) -> int:
242
+ in_flight = 0
243
+ try:
244
+ for p in experiment_root.glob("*.zip"):
245
+ try:
246
+ in_flight += p.stat().st_size
247
+ except OSError:
248
+ continue
249
+ except OSError:
250
+ pass
251
+ with self._lock:
252
+ return self._completed_bytes + in_flight
253
+
254
+
255
+ def _report_progress(
256
+ label: str,
257
+ experiment_root: Path,
258
+ progress: _ExperimentProgress,
259
+ stop_event: threading.Event,
260
+ interval: float = 5.0,
261
+ ) -> None:
262
+ """Keep one download-progress line for this experiment updating in place.
263
+
264
+ Runs in its own thread, polling every `interval` seconds; one such
265
+ thread runs per experiment currently downloading, so under `-n` each
266
+ active worker keeps its own persistent line via `_board` (or, outside a
267
+ terminal, its own sequence of plain printed lines).
268
+ """
269
+ try:
270
+ while not stop_event.wait(interval):
271
+ downloaded = progress.current_bytes(experiment_root)
272
+ _board.update(label, f" [{label}] {_human_bytes(downloaded)} downloaded")
273
+ finally:
274
+ _board.finish(label)
275
+
276
+
277
+ def _zip_wrapper_prefix(names: list[str]) -> str:
278
+ """Return the shared top-level path segment across all zip members, if any.
279
+
280
+ XNAT's zip export nests every entry under a single wrapper directory
281
+ (typically named after the experiment), which would otherwise reproduce
282
+ an identically-named EXPERIMENT/EXPERIMENT folder on disk. Detecting it
283
+ by shared prefix rather than a hardcoded name keeps this robust to
284
+ whatever XNAT actually calls it.
285
+ """
286
+ segments = {n.split("/", 1)[0] for n in names if "/" in n and n.split("/", 1)[0]}
287
+ if len(segments) == 1:
288
+ return next(iter(segments)) + "/"
289
+ return ""
290
+
291
+
292
+ def _extract_zip_flattened(zip_path: Path, dest_dir: Path) -> None:
293
+ """Extract zip_path into dest_dir, stripping any shared wrapper directory."""
294
+ with zipfile.ZipFile(zip_path) as zf:
295
+ names = zf.namelist()
296
+ prefix = _zip_wrapper_prefix(names)
297
+ for name in names:
298
+ if name.endswith("/"):
299
+ continue # directory entry
300
+ rel = name[len(prefix):] if prefix and name.startswith(prefix) else name
301
+ if not rel:
302
+ continue
303
+ target = dest_dir / rel
304
+ target.parent.mkdir(parents=True, exist_ok=True)
305
+ with zf.open(name) as src, target.open("wb") as dst:
306
+ shutil.copyfileobj(src, dst)
307
+ zip_path.unlink(missing_ok=True)
308
+
309
+
310
+ _CONTENT_DISPOSITION_FILENAME_RE = re.compile(
311
+ r'filename\*?=(?:UTF-8\'\')?"?([^";]+)"?', re.IGNORECASE
312
+ )
313
+
314
+
315
+ def _filename_from_content_disposition(header: str) -> str | None:
316
+ """Pull a bare filename out of a Content-Disposition header, if present."""
317
+ match = _CONTENT_DISPOSITION_FILENAME_RE.search(header)
318
+ if not match:
319
+ return None
320
+ # unquote handles the percent-encoding an RFC 5987 filename* uses;
321
+ # Path(...).name strips any directory components a server might send.
322
+ name = Path(unquote(match.group(1).strip())).name
323
+ return name or None
324
+
325
+
326
+ def _resource_identifier(resource) -> str:
327
+ """A filesystem-safe identifier for a pyxnat Resource, from its own URI."""
328
+ return Path(uriutil.uri_last(resource._uri)).name or "unknown"
329
+
330
+
331
+ def _download_single_resource_zip(resource, dest_dir: Path) -> tuple[Path, str]:
332
+ """Download one named session-level resource as a zip archive.
333
+
334
+ Mirrors pyxnat's own ``Resource.get()`` (``.../resources/{ID}/files?format=zip``)
335
+ — the per-resource download XNAT actually supports; see ``_download_resources``
336
+ for why there's no bulk equivalent — but keeps the response's
337
+ ``Content-Disposition`` header around afterward for
338
+ ``_extract_zip_flattened_or_rescue``.
339
+ """
340
+ url = resource._uri + "/files?format=zip"
341
+ response = resource._intf.get(url, stream=True)
342
+ try:
343
+ if not response.ok:
344
+ raise RuntimeError(f"HTTP {response.status_code} {response.reason}")
345
+ content_disposition = response.headers.get("Content-Disposition", "")
346
+ zip_path = dest_dir / f"resource_{_resource_identifier(resource)}.zip"
347
+ with zip_path.open("wb") as f:
348
+ count = 0
349
+ for chunk in response.iter_content(chunk_size=1024):
350
+ if chunk:
351
+ f.write(chunk)
352
+ count += 1
353
+ if count % 10 == 0:
354
+ f.flush()
355
+ f.flush()
356
+ finally:
357
+ response.close()
358
+ return zip_path, content_disposition
359
+
360
+
361
+ def _download_resources(
362
+ exp_obj, dest_dir: Path, progress: "_ExperimentProgress | None"
363
+ ) -> None:
364
+ """Download and extract every session-level resource for an experiment.
365
+
366
+ XNAT has no bulk "download every resource at once" endpoint the way scans
367
+ do (``.../scans/ALL/files?format=zip``, where ``ALL`` is a scan-*type*
368
+ wildcard); the ``/resources`` collection has no ``.download()``/``.get()``
369
+ method in pyxnat at all — only the individual ``Resource`` element class
370
+ does, one resource at a time. This mirrors that real, working pattern via
371
+ ``_download_single_resource_zip``, and keeps going past one resource's
372
+ failure so a single bad resource doesn't lose the rest.
373
+
374
+ Raises LookupError if there are no session-level resources at all, or an
375
+ Exception summarizing every resource that failed, raised only after every
376
+ resource has been attempted (so successful ones still land on disk).
377
+ """
378
+ resources = list(exp_obj.resources())
379
+ if not resources:
380
+ raise LookupError("There are no resources to download")
381
+
382
+ errors: list[str] = []
383
+ for resource in resources:
384
+ try:
385
+ zip_path, content_disposition = _download_single_resource_zip(
386
+ resource, dest_dir
387
+ )
388
+ if progress is not None:
389
+ progress.add_completed(zip_path.stat().st_size)
390
+ _extract_zip_flattened_or_rescue(zip_path, dest_dir, content_disposition)
391
+ except Exception as e:
392
+ errors.append(
393
+ f"resource '{_resource_identifier(resource)}': "
394
+ f"{_describe_download_error(e)}"
395
+ )
396
+
397
+ if errors:
398
+ raise RuntimeError("; ".join(errors))
399
+
400
+
401
+ def _extract_zip_flattened_or_rescue(
402
+ zip_path: Path, dest_dir: Path, content_disposition: str
403
+ ) -> None:
404
+ """Extract zip_path into dest_dir, rescuing XNAT's single-file quirk.
405
+
406
+ XNAT's zip-export endpoint is meant to always return a zip archive, but
407
+ when a session's resources resolve to exactly one file it sometimes
408
+ streams that file directly instead (a server-side behavior, not
409
+ specific to any one experiment). When the downloaded bytes don't open as
410
+ a zip, save them as that one file — named from the response's
411
+ Content-Disposition header, falling back to the temp file's own name —
412
+ instead of failing the whole experiment.
413
+ """
414
+ try:
415
+ _extract_zip_flattened(zip_path, dest_dir)
416
+ except zipfile.BadZipFile:
417
+ filename = _filename_from_content_disposition(content_disposition) or zip_path.stem
418
+ zip_path.replace(dest_dir / filename)
419
+
420
+
421
+ def _process_experiment(
422
+ interface: Interface,
423
+ project: str,
424
+ subject: str,
425
+ experiment: str,
426
+ output_dir: Path,
427
+ report: Callable[[str], None],
428
+ progress: _ExperimentProgress | None = None,
429
+ local_subject: str | None = None,
430
+ local_experiment: str | None = None,
431
+ ) -> str:
432
+ """Download one experiment as whole-experiment zip archives.
433
+
434
+ Bulk requests are made against XNAT's REST zip-export endpoint via
435
+ pyxnat, rather than one HTTP request per file: one for all scans, and
436
+ one per session-level resource (see ``_download_resources`` for why
437
+ resources can't be fetched in a single bulk request the way scans can).
438
+ Each zip is flattened into the experiment's output directory (stripping
439
+ XNAT's own wrapper folder, see ``_extract_zip_flattened``), so the
440
+ on-disk layout follows XNAT's own scan/resource folder naming without an
441
+ extra EXPERIMENT/EXPERIMENT level.
442
+
443
+ `local_subject`/`local_experiment` (from `SUBJECT_BIDS_RENAME`/
444
+ `EXPERIMENT_BIDS_RENAME`) name the on-disk directory when they differ
445
+ from `subject`/`experiment`, which are always used to look up the
446
+ experiment on the XNAT server itself.
447
+ """
448
+ label = f"{project}/{subject}/{experiment}"
449
+ try:
450
+ exp_obj = (
451
+ interface.select.project(project)
452
+ .subject(subject)
453
+ .experiment(experiment)
454
+ )
455
+ if not exp_obj.exists():
456
+ return STATUS_NONEXISTENT
457
+ except Exception as e:
458
+ report(f"Error looking up {label}: {e}")
459
+ return STATUS_FAILURE
460
+
461
+ experiment_root = (
462
+ output_dir / project / (local_subject or subject) / (local_experiment or experiment)
463
+ )
464
+ experiment_root.mkdir(parents=True, exist_ok=True)
465
+
466
+ got_scans = False
467
+ got_resources = False
468
+ failed = False
469
+
470
+ try:
471
+ zip_path = Path(exp_obj.scans().download(str(experiment_root), extract=False))
472
+ if progress is not None:
473
+ progress.add_completed(zip_path.stat().st_size)
474
+ _extract_zip_flattened(zip_path, experiment_root)
475
+ got_scans = True
476
+ except LookupError:
477
+ pass # no scans on this experiment
478
+ except Exception as e:
479
+ report(f" Error downloading scans for {label}: {_describe_download_error(e)}")
480
+ failed = True
481
+
482
+ try:
483
+ _download_resources(exp_obj, experiment_root, progress)
484
+ got_resources = True
485
+ except LookupError:
486
+ pass # no session-level resources on this experiment
487
+ except Exception as e:
488
+ report(f" Error downloading resources for {label}: {e}")
489
+ failed = True
490
+
491
+ if failed:
492
+ return STATUS_FAILURE
493
+ if not got_scans and not got_resources:
494
+ return STATUS_EMPTY
495
+ return STATUS_COMPLETE
496
+
497
+
498
+ def _format_bids_rename(value: str, prefix: str) -> str | None:
499
+ """Normalize a *_BIDS_RENAME cell, prepending `prefix` if not already present.
500
+
501
+ Returns None if the value (after stripping an already-present prefix) is
502
+ not purely alphanumeric, signaling an invalid rename value.
503
+ """
504
+ remainder = value[len(prefix):] if value.startswith(prefix) else value
505
+ if not remainder.isalnum():
506
+ return None
507
+ return value if value.startswith(prefix) else prefix + value
508
+
509
+
510
+ def _read_csv_rows(
511
+ path: Path,
512
+ ) -> list[tuple[str, str, str, str | None, str | None]]:
513
+ if not path.exists():
514
+ sys.exit(f"Error: input CSV not found: {path}")
515
+ rows: list[tuple[str, str, str, str | None, str | None]] = []
516
+ with path.open(newline="") as f:
517
+ reader = csv.DictReader(f)
518
+ required = {"PROJECT", "SUBJECT_LABEL", "EXPERIMENT_LABEL"}
519
+ if not reader.fieldnames or not required.issubset(reader.fieldnames):
520
+ sys.exit(
521
+ f"Error: input CSV {path} must have columns "
522
+ "PROJECT, SUBJECT_LABEL, EXPERIMENT_LABEL."
523
+ )
524
+ for i, row in enumerate(reader, start=2):
525
+ p = (row.get("PROJECT") or "").strip()
526
+ s = (row.get("SUBJECT_LABEL") or "").strip()
527
+ e = (row.get("EXPERIMENT_LABEL") or "").strip()
528
+ if not (p and s and e):
529
+ sys.exit(
530
+ f"Error: row {i} of {path} is missing a required value."
531
+ )
532
+ subject_rename: str | None = None
533
+ raw_subject_rename = (row.get("SUBJECT_BIDS_RENAME") or "").strip()
534
+ if raw_subject_rename:
535
+ subject_rename = _format_bids_rename(raw_subject_rename, "sub-")
536
+ if subject_rename is None:
537
+ sys.exit(
538
+ f"Error: row {i} of {path} has an invalid "
539
+ f"SUBJECT_BIDS_RENAME value '{raw_subject_rename}': "
540
+ "must be alphanumeric only (after an optional "
541
+ "'sub-' prefix)."
542
+ )
543
+
544
+ experiment_rename: str | None = None
545
+ raw_experiment_rename = (row.get("EXPERIMENT_BIDS_RENAME") or "").strip()
546
+ if raw_experiment_rename:
547
+ experiment_rename = _format_bids_rename(raw_experiment_rename, "ses-")
548
+ if experiment_rename is None:
549
+ sys.exit(
550
+ f"Error: row {i} of {path} has an invalid "
551
+ f"EXPERIMENT_BIDS_RENAME value '{raw_experiment_rename}': "
552
+ "must be alphanumeric only (after an optional "
553
+ "'ses-' prefix)."
554
+ )
555
+
556
+ rows.append((p, s, e, subject_rename, experiment_rename))
557
+ return rows
558
+
559
+
560
+ def _archive_and_maybe_delete(
561
+ output_dir: Path,
562
+ project: str,
563
+ subject: str,
564
+ experiment: str,
565
+ do_archive: bool,
566
+ do_delete: bool,
567
+ report: Callable[[str], None],
568
+ ) -> None:
569
+ if not do_archive:
570
+ return
571
+ label = f"{project}/{subject}/{experiment}"
572
+ a_status, a_detail = archive_experiment(
573
+ output_dir, project, subject, experiment
574
+ )
575
+ line = f" archive {label}: {a_status}"
576
+ if a_detail:
577
+ line += f" — {a_detail}"
578
+ report(line)
579
+ if do_delete and a_status in ARCHIVE_OK_STATUSES:
580
+ delete_experiment_dir(output_dir, project, subject, experiment)
581
+
582
+
583
+ def _run_single(
584
+ server: str,
585
+ user: str,
586
+ password: str,
587
+ project: str,
588
+ subject: str,
589
+ experiment: str,
590
+ output_dir: Path,
591
+ log_writer: _LogWriter,
592
+ do_archive: bool,
593
+ do_delete: bool,
594
+ local_subject: str | None = None,
595
+ local_experiment: str | None = None,
596
+ ) -> str:
597
+ local_s = local_subject or subject
598
+ local_e = local_experiment or experiment
599
+ iface = Interface(server=server, user=user, password=password)
600
+ try:
601
+ start = _logging_now()
602
+ status = _process_experiment(
603
+ iface, project, subject, experiment, output_dir, _safe_print,
604
+ local_subject=local_s, local_experiment=local_e,
605
+ )
606
+ finally:
607
+ try:
608
+ iface.disconnect()
609
+ except Exception:
610
+ pass
611
+ log_writer.write(start, project, local_s, local_e, status)
612
+ _archive_and_maybe_delete(
613
+ output_dir,
614
+ project,
615
+ local_s,
616
+ local_e,
617
+ do_archive,
618
+ do_delete,
619
+ _safe_print,
620
+ )
621
+ return status
622
+
623
+
624
+ def _run_csv(
625
+ server: str,
626
+ user: str,
627
+ password: str,
628
+ rows: list[tuple[str, str, str, str | None, str | None]],
629
+ output_dir: Path,
630
+ n_parallel_experiments: int,
631
+ log_writer: _LogWriter,
632
+ do_archive: bool,
633
+ do_delete: bool,
634
+ ) -> dict[str, int]:
635
+ counts = {
636
+ STATUS_COMPLETE: 0,
637
+ STATUS_FAILURE: 0,
638
+ STATUS_NONEXISTENT: 0,
639
+ STATUS_EMPTY: 0,
640
+ }
641
+
642
+ def _worker(row: tuple[str, str, str, str | None, str | None]) -> str:
643
+ p, s, e, subject_rename, experiment_rename = row
644
+ local_s = subject_rename or s
645
+ local_e = experiment_rename or e
646
+ iface = _get_thread_interface(server, user, password)
647
+ label = f"{p}/{s}/{e}"
648
+ experiment_root = output_dir / p / local_s / local_e
649
+ progress = _ExperimentProgress()
650
+ stop_event = threading.Event()
651
+ monitor = threading.Thread(
652
+ target=_report_progress,
653
+ args=(label, experiment_root, progress, stop_event),
654
+ daemon=True,
655
+ )
656
+ monitor.start()
657
+ try:
658
+ start = _logging_now()
659
+ status = _process_experiment(
660
+ iface, p, s, e, output_dir, _safe_print, progress,
661
+ local_s, local_e,
662
+ )
663
+ finally:
664
+ stop_event.set()
665
+ monitor.join()
666
+ log_writer.write(start, p, local_s, local_e, status)
667
+ _archive_and_maybe_delete(
668
+ output_dir, p, local_s, local_e, do_archive, do_delete, _safe_print
669
+ )
670
+ return status
671
+
672
+ if n_parallel_experiments <= 1:
673
+ try:
674
+ for triplet in rows:
675
+ counts[_worker(triplet)] += 1
676
+ except KeyboardInterrupt:
677
+ _board.log("\nInterrupted: stopping before starting the next experiment.")
678
+ sys.stdout.flush()
679
+ sys.exit(130)
680
+ finally:
681
+ _close_thread_interface()
682
+ else:
683
+ ex = ThreadPoolExecutor(max_workers=n_parallel_experiments)
684
+ pending = {ex.submit(_worker, t) for t in rows}
685
+ try:
686
+ while pending:
687
+ # A short timeout (rather than an unbounded wait) hands control
688
+ # back to the interpreter every 0.5s, which is what lets a
689
+ # pending Ctrl+C actually get raised here instead of sitting
690
+ # queued until the whole batch finishes.
691
+ done, pending = wait(pending, timeout=0.5, return_when=FIRST_COMPLETED)
692
+ for fut in done:
693
+ counts[fut.result()] += 1
694
+ except KeyboardInterrupt:
695
+ still_running = sum(1 for fut in pending if fut.running())
696
+ queued = len(pending) - still_running
697
+ for fut in pending:
698
+ fut.cancel() # No-op for already-running futures.
699
+ _board.log(
700
+ f"\nInterrupted: {queued} queued download(s) cancelled; "
701
+ f"{still_running} already in progress were abandoned "
702
+ "mid-transfer (their output may be incomplete)."
703
+ )
704
+ sys.stdout.flush()
705
+ # ThreadPoolExecutor's worker threads are non-daemon, and CPython
706
+ # joins non-daemon threads on interpreter shutdown regardless of
707
+ # sys.exit()/exceptions — which is exactly what made Ctrl+C
708
+ # appear to do nothing while downloads already in flight kept
709
+ # running. os._exit() skips that join entirely.
710
+ os._exit(130)
711
+ ex.shutdown(wait=True)
712
+ # Worker threads' Interface objects are GC'd when the pool shuts down.
713
+
714
+ return counts
715
+
716
+
717
+ def _resolve_accession(
718
+ interface: Interface, accession: str
719
+ ) -> tuple[str, str, str, str | None]:
720
+ """Resolve an accession to its project/subject/experiment, server-wide.
721
+
722
+ Tries, in order: subject ID, experiment ID, study UID (XNAT's stored
723
+ session ``UID``), then experiment and subject labels together. IDs and
724
+ UIDs are unique across the server; labels are only unique within their
725
+ parent, so a label matching more than one subject/experiment exits with
726
+ the list of matches. pyxnat has no server-wide subject lookup, so this
727
+ reaches XNAT's root-level ``/subjects`` and ``/experiments`` listings
728
+ via ``interface._get_json``. Each listing is also filtered here with an
729
+ exact, case-insensitive match, in case the server ignores or loosens
730
+ the query-string filter.
731
+
732
+ Parameters
733
+ ----------
734
+ interface : Interface
735
+ Connected pyxnat interface.
736
+ accession : str
737
+ A subject ID or label, experiment ID or label, or StudyInstanceUID.
738
+
739
+ Returns
740
+ -------
741
+ tuple[str, str, str, str | None]
742
+ ``("subject", project, subject_id, None)`` or
743
+ ``("experiment", project, subject_id, experiment_id)``.
744
+ """
745
+ interface._get_entry_point()
746
+ encoded = quote(accession, safe="")
747
+ wanted = accession.strip().lower()
748
+
749
+ def lookup(kind: str, field: str, columns: str) -> list[dict]:
750
+ rows = interface._get_json(
751
+ f"{interface._entry}/{kind}?{field}={encoded}"
752
+ f"&columns={columns}&format=json"
753
+ )
754
+ return [r for r in rows if (r.get(field) or "").strip().lower() == wanted]
755
+
756
+ def as_experiment(r: dict) -> tuple[str, str, str, str]:
757
+ return ("experiment", r["project"], r["subject_ID"], r["ID"])
758
+
759
+ exp_cols = "ID,label,project,subject_ID"
760
+ try:
761
+ rows = lookup("subjects", "ID", "ID,project")
762
+ if rows:
763
+ return ("subject", rows[0]["project"], rows[0]["ID"], None)
764
+ rows = lookup("experiments", "ID", exp_cols)
765
+ if rows:
766
+ return as_experiment(rows[0])
767
+ rows = lookup("experiments", "UID", exp_cols + ",UID")
768
+ if len(rows) == 1:
769
+ return as_experiment(rows[0])
770
+ uid_rows = rows
771
+ exp_rows = lookup("experiments", "label", exp_cols)
772
+ subj_rows = lookup("subjects", "label", "ID,label,project")
773
+ except Exception as e:
774
+ sys.exit(f"Error: could not resolve accession '{accession}': {e}")
775
+
776
+ matches = [
777
+ (f"experiment {r['ID']} (project {r['project']}, subject {r['subject_ID']})",
778
+ as_experiment(r))
779
+ for r in uid_rows + exp_rows
780
+ ] + [
781
+ (f"subject {r['ID']} (project {r['project']})",
782
+ ("subject", r["project"], r["ID"], None))
783
+ for r in subj_rows
784
+ ]
785
+ if len(matches) == 1:
786
+ return matches[0][1]
787
+ if not matches:
788
+ sys.exit(
789
+ f"Error: accession '{accession}' not found as a subject ID or "
790
+ "label, experiment ID or label, or study UID on the configured "
791
+ "server."
792
+ )
793
+ listing = "\n".join(f" {what}" for what, _ in matches)
794
+ sys.exit(
795
+ f"Error: accession '{accession}' matches more than one subject or "
796
+ f"experiment:\n{listing}\nRerun with one of the XNAT IDs above."
797
+ )
798
+
799
+
800
+ def _download_accession(
801
+ server: str,
802
+ user: str,
803
+ password: str,
804
+ accession: str,
805
+ output_dir: Path,
806
+ log_writer: _LogWriter,
807
+ do_archive: bool,
808
+ do_delete: bool,
809
+ n_parallel_experiments: int,
810
+ local_subject: str | None,
811
+ local_experiment: str | None,
812
+ log_path: Path | None,
813
+ ) -> int:
814
+ """Handle `download --accession`: resolve it, then delegate like -1/--csv.
815
+
816
+ A subject accession downloads every experiment belonging to that
817
+ subject (like `--csv` batch mode); an experiment accession downloads
818
+ just that one experiment (like `-1`).
819
+ """
820
+ interface = Interface(server=server, user=user, password=password)
821
+ try:
822
+ kind, project, subject_id, experiment_id = _resolve_accession(
823
+ interface, accession
824
+ )
825
+ finally:
826
+ try:
827
+ interface.disconnect()
828
+ except Exception:
829
+ pass
830
+
831
+ if kind == "subject":
832
+ if local_experiment is not None:
833
+ sys.exit(
834
+ f"Error: --rename-experiment cannot be used with subject "
835
+ f"accession '{accession}', since a subject may have more "
836
+ "than one experiment."
837
+ )
838
+ interface = Interface(server=server, user=user, password=password)
839
+ try:
840
+ subj_obj = interface.select.project(project).subject(subject_id)
841
+ experiment_labels = [e.label() for e in subj_obj.experiments()]
842
+ except Exception as e:
843
+ sys.exit(
844
+ f"Error: could not list experiments for subject accession "
845
+ f"'{accession}': {e}"
846
+ )
847
+ finally:
848
+ try:
849
+ interface.disconnect()
850
+ except Exception:
851
+ pass
852
+ if not experiment_labels:
853
+ sys.exit(
854
+ f"Error: subject accession '{accession}' has no experiments "
855
+ "to download."
856
+ )
857
+ rows = [
858
+ (project, subject_id, label, local_subject, None)
859
+ for label in experiment_labels
860
+ ]
861
+ counts = _run_csv(
862
+ server, user, password, rows, output_dir, n_parallel_experiments,
863
+ log_writer, do_archive, do_delete,
864
+ )
865
+ total = sum(counts.values())
866
+ print(f"\nProcessed {total} experiment(s) for subject accession {accession}:")
867
+ for status in (
868
+ STATUS_COMPLETE,
869
+ STATUS_FAILURE,
870
+ STATUS_NONEXISTENT,
871
+ STATUS_EMPTY,
872
+ ):
873
+ print(f" {status}: {counts[status]}")
874
+ if log_path is not None:
875
+ print(f"Log written to {log_path}")
876
+ bad = counts[STATUS_FAILURE] + counts[STATUS_NONEXISTENT]
877
+ return 0 if bad == 0 else 1
878
+
879
+ status = _run_single(
880
+ server, user, password, project, subject_id, experiment_id,
881
+ output_dir, log_writer, do_archive, do_delete,
882
+ local_subject, local_experiment,
883
+ )
884
+ print(
885
+ f"Status for accession {accession} "
886
+ f"({project}/{subject_id}/{experiment_id}): {status}"
887
+ )
888
+ if log_path is not None:
889
+ print(f"Log written to {log_path}")
890
+ return 0 if status in _OK_STATUSES else 1
891
+
892
+
893
+ def download_cmd(args: argparse.Namespace) -> int:
894
+ if args.ndownload < 1:
895
+ sys.exit("Error: -n/--ndownload must be >= 1.")
896
+ if args.triplet is not None and args.ndownload != 1:
897
+ sys.exit(
898
+ "Error: -n/--ndownload only applies to --csv/--input or a "
899
+ "subject --accession."
900
+ )
901
+ if args.delete and not args.archive:
902
+ sys.exit("Error: -d/--delete requires -a/--archive.")
903
+ if (
904
+ args.triplet is None
905
+ and args.accession is None
906
+ and (args.rename_subject or args.rename_experiment)
907
+ ):
908
+ sys.exit(
909
+ "Error: --rename-subject/--rename-experiment only apply to "
910
+ "-1 or --accession downloads."
911
+ )
912
+
913
+ local_subject: str | None = None
914
+ if args.rename_subject:
915
+ local_subject = _format_bids_rename(args.rename_subject, "sub-")
916
+ if local_subject is None:
917
+ sys.exit(
918
+ f"Error: invalid --rename-subject value '{args.rename_subject}': "
919
+ "must be alphanumeric only (after an optional 'sub-' prefix)."
920
+ )
921
+
922
+ local_experiment: str | None = None
923
+ if args.rename_experiment:
924
+ local_experiment = _format_bids_rename(args.rename_experiment, "ses-")
925
+ if local_experiment is None:
926
+ sys.exit(
927
+ f"Error: invalid --rename-experiment value "
928
+ f"'{args.rename_experiment}': must be alphanumeric only "
929
+ "(after an optional 'ses-' prefix)."
930
+ )
931
+
932
+ server, user, password = load_credentials()
933
+ output_dir = Path(args.output)
934
+ output_dir.mkdir(parents=True, exist_ok=True)
935
+
936
+ log_path: Path | None = None
937
+ if args.log:
938
+ while True:
939
+ ts = datetime.now().strftime("%Y%m%d_%H%M%S")
940
+ log_path = output_dir / "log" / f"download_{ts}_log.csv"
941
+ if not log_path.exists():
942
+ break
943
+ time.sleep(1)
944
+ log_writer = _LogWriter(log_path)
945
+
946
+ if args.triplet is not None:
947
+ project, subject, experiment = args.triplet
948
+ status = _run_single(
949
+ server,
950
+ user,
951
+ password,
952
+ project,
953
+ subject,
954
+ experiment,
955
+ output_dir,
956
+ log_writer,
957
+ args.archive,
958
+ args.delete,
959
+ local_subject,
960
+ local_experiment,
961
+ )
962
+ print(
963
+ f"Status for {project}/{subject}/{experiment}: {status}"
964
+ )
965
+ if log_path is not None:
966
+ print(f"Log written to {log_path}")
967
+ return 0 if status in _OK_STATUSES else 1
968
+
969
+ if args.accession is not None:
970
+ return _download_accession(
971
+ server,
972
+ user,
973
+ password,
974
+ args.accession,
975
+ output_dir,
976
+ log_writer,
977
+ args.archive,
978
+ args.delete,
979
+ args.ndownload,
980
+ local_subject,
981
+ local_experiment,
982
+ log_path,
983
+ )
984
+
985
+ rows = _read_csv_rows(Path(args.input))
986
+ counts = _run_csv(
987
+ server,
988
+ user,
989
+ password,
990
+ rows,
991
+ output_dir,
992
+ args.ndownload,
993
+ log_writer,
994
+ args.archive,
995
+ args.delete,
996
+ )
997
+ total = sum(counts.values())
998
+ print(f"\nProcessed {total} experiment(s):")
999
+ for status in (
1000
+ STATUS_COMPLETE,
1001
+ STATUS_FAILURE,
1002
+ STATUS_NONEXISTENT,
1003
+ STATUS_EMPTY,
1004
+ ):
1005
+ print(f" {status}: {counts[status]}")
1006
+ if log_path is not None:
1007
+ print(f"Log written to {log_path}")
1008
+ bad = counts[STATUS_FAILURE] + counts[STATUS_NONEXISTENT]
1009
+ return 0 if bad == 0 else 1