jamofetch 3.7.7__tar.gz → 3.7.9__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- jamofetch-3.7.9/CHANGELOG.md +53 -0
- {jamofetch-3.7.7 → jamofetch-3.7.9}/PKG-INFO +4 -4
- {jamofetch-3.7.7 → jamofetch-3.7.9}/README.md +3 -3
- {jamofetch-3.7.7 → jamofetch-3.7.9}/pyproject.toml +1 -1
- jamofetch-3.7.9/src/jamofetch/jamofetch.py +427 -0
- jamofetch-3.7.9/tests/test_jamofetch.py +184 -0
- jamofetch-3.7.7/CHANGELOG.md +0 -7
- jamofetch-3.7.7/src/jamofetch/jamofetch.py +0 -271
- jamofetch-3.7.7/tests/test_jamofetch.py +0 -2
- {jamofetch-3.7.7 → jamofetch-3.7.9}/.gitignore +0 -0
- {jamofetch-3.7.7 → jamofetch-3.7.9}/.gitlab-ci.yml +0 -0
- {jamofetch-3.7.7 → jamofetch-3.7.9}/.readthedocs.yml +0 -0
- {jamofetch-3.7.7 → jamofetch-3.7.9}/CONDUCT.md +0 -0
- {jamofetch-3.7.7 → jamofetch-3.7.9}/CONTRIBUTING.md +0 -0
- {jamofetch-3.7.7 → jamofetch-3.7.9}/docs/Makefile +0 -0
- {jamofetch-3.7.7 → jamofetch-3.7.9}/docs/changelog.md +0 -0
- {jamofetch-3.7.7 → jamofetch-3.7.9}/docs/conduct.md +0 -0
- {jamofetch-3.7.7 → jamofetch-3.7.9}/docs/conf.py +0 -0
- {jamofetch-3.7.7 → jamofetch-3.7.9}/docs/contributing.md +0 -0
- {jamofetch-3.7.7 → jamofetch-3.7.9}/docs/demo_script.py +0 -0
- {jamofetch-3.7.7 → jamofetch-3.7.9}/docs/index.md +0 -0
- {jamofetch-3.7.7 → jamofetch-3.7.9}/docs/make.bat +0 -0
- {jamofetch-3.7.7 → jamofetch-3.7.9}/docs/requirements.txt +0 -0
- {jamofetch-3.7.7 → jamofetch-3.7.9}/prepare-gitlab-publish.sh +0 -0
- {jamofetch-3.7.7 → jamofetch-3.7.9}/push-version-tag.sh +0 -0
- {jamofetch-3.7.7 → jamofetch-3.7.9}/src/jamofetch/__init__.py +0 -0
|
@@ -0,0 +1,53 @@
|
|
|
1
|
+
# Changelog
|
|
2
|
+
|
|
3
|
+
<!--next-version-placeholder-->
|
|
4
|
+
|
|
5
|
+
## v3.7.9 (17/09/2026)
|
|
6
|
+
|
|
7
|
+
### Fixed
|
|
8
|
+
|
|
9
|
+
- **A library sequenced more than once no longer resolves to an arbitrary file.**
|
|
10
|
+
`_find_fastq_path()` sorted the link directory by `ST_CTIME` and kept the last
|
|
11
|
+
match. Every link comes from one `jamo link` call within the same second and
|
|
12
|
+
`ST_CTIME` is whole seconds, so the candidates all tie and Python's stable sort
|
|
13
|
+
returns whichever `os.listdir()` listed last - directory hash order on Linux,
|
|
14
|
+
so not alphabetical, not newest, and not necessarily the same on two hosts.
|
|
15
|
+
It now returns the single match, or raises and names every candidate.
|
|
16
|
+
|
|
17
|
+
Observed on library LBBCZRJ: four matching records across three pools - cell
|
|
18
|
+
37397 (run 3207, 1,296,169 reads), cell 37892 (run 3232, 1,965,608 reads,
|
|
19
|
+
registered twice under different `AUTO-` folders with identical md5s) and cell
|
|
20
|
+
37964 (run 3230, 948,348 reads). Any of the three could be analysed, silently,
|
|
21
|
+
with a 2x spread in read count.
|
|
22
|
+
|
|
23
|
+
Links pointing at the same target are not ambiguous and are still accepted.
|
|
24
|
+
|
|
25
|
+
### Added
|
|
26
|
+
|
|
27
|
+
- `get_cmd()` and `JamoFetcher.fetch_lib_seq()` take `smrt_cell_id` and
|
|
28
|
+
`physical_run_id`, which narrow the query to one sequencing of the library.
|
|
29
|
+
`smrt_cell_id` is the one to prefer: it identifies a library's membership in a
|
|
30
|
+
single pool, and equals the PacBio pipeline service's `libraries[].id`. For
|
|
31
|
+
LBBCZRJ it reduces four matches to one (or, for cell 37892, to two records
|
|
32
|
+
that are byte-identical and share a `file_name`).
|
|
33
|
+
- CLI: `--smrt-cell-id` (single `--library` only) and `--physical-run-id`
|
|
34
|
+
(applies to every `--library`).
|
|
35
|
+
|
|
36
|
+
### Changed
|
|
37
|
+
|
|
38
|
+
- The CLI no longer swallows per-library fetch errors. `_fetch_seq()` caught
|
|
39
|
+
every exception with a bare `pass`, which hid the new ambiguity error whose
|
|
40
|
+
whole purpose is to be seen.
|
|
41
|
+
- `JamoFetcher.fetch_lib_seq()` logs every candidate link when more than one
|
|
42
|
+
matches, before selection.
|
|
43
|
+
|
|
44
|
+
### Compatibility
|
|
45
|
+
|
|
46
|
+
- Calls that pass no id produce the byte-identical command 3.7.8 produced;
|
|
47
|
+
asserted in `tests/test_jamofetch.py`. Callers that relied on a library with
|
|
48
|
+
several matches silently resolving to one of them now get a `RuntimeError`
|
|
49
|
+
instead - that is the point of the release.
|
|
50
|
+
|
|
51
|
+
## v0.1.0 (07/07/2023)
|
|
52
|
+
|
|
53
|
+
- First release of `jamofetch`!
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: jamofetch
|
|
3
|
-
Version: 3.7.
|
|
3
|
+
Version: 3.7.9
|
|
4
4
|
Summary: A thin wrapper to retrieve sequence from JAMO at NERSC and on Dori.
|
|
5
5
|
Author: Duncan Scott
|
|
6
6
|
Requires-Python: >=3.9
|
|
@@ -89,15 +89,15 @@ options:
|
|
|
89
89
|
(venv) [dnscott@ln005 jamofetch]$ jamofetch -d data -l NPUNN -l NOOHG -l HOGH -w --max -1
|
|
90
90
|
fetching sequence:
|
|
91
91
|
|
|
92
|
-
apptainer --silent
|
|
92
|
+
apptainer --silent exec docker://doejgi/jamo-dori jamo link -s dori custom '{"metadata.library_name":"NPUNN","metadata.fastq_type":"sdm_normal","group":"sdm"}'
|
|
93
93
|
NPUNN /global/dna/dm_archive/sdm/pacbio/00/27/47/pbio-2747.27352.bc1001_BAK8A_OA--bc1001_BAK8A_OA.ccs.fastq.gz BACKUP_COMPLETE 6391936239a7711d789a9380
|
|
94
94
|
NPUNN /clusterfs/jgi/groups/dsi/homes/dnscott/git/jamofetch/data/NPUNN.pbio-2747.27352.bc1001_BAK8A_OA--bc1001_BAK8A_OA.ccs.fastq.gz
|
|
95
95
|
|
|
96
|
-
apptainer --silent
|
|
96
|
+
apptainer --silent exec docker://doejgi/jamo-dori jamo link -s dori custom '{"metadata.library_name":"NOOHG","metadata.fastq_type":"sdm_normal","group":"sdm"}'
|
|
97
97
|
NOOHG /global/dna/dm_archive/sdm/pacbio/00/26/91/pbio-2691.26653.bc1001_BAK8A_OA--bc1001_BAK8A_OA.ccs.fastq.gz RESTORED 6347dbb35bc59487d7e768d6
|
|
98
98
|
NOOHG /clusterfs/jgi/groups/dsi/homes/dnscott/git/jamofetch/data/NOOHG.pbio-2691.26653.bc1001_BAK8A_OA--bc1001_BAK8A_OA.ccs.fastq.gz
|
|
99
99
|
|
|
100
|
-
apptainer --silent
|
|
100
|
+
apptainer --silent exec docker://doejgi/jamo-dori jamo link -s dori custom '{"metadata.library_name":"HOGH","metadata.fastq_type":"sdm_normal","group":"sdm"}'
|
|
101
101
|
HOGH /global/dna/dm_archive/sdm/illumina/00/63/97/6397.2.44053.GGCTAC.fastq.gz RESTORED 51d52a82067c014cd6ef4f6f
|
|
102
102
|
HOGH /clusterfs/jgi/groups/dsi/homes/dnscott/git/jamofetch/data/HOGH.6397.2.44053.GGCTAC.fastq.gz
|
|
103
103
|
|
|
@@ -78,15 +78,15 @@ options:
|
|
|
78
78
|
(venv) [dnscott@ln005 jamofetch]$ jamofetch -d data -l NPUNN -l NOOHG -l HOGH -w --max -1
|
|
79
79
|
fetching sequence:
|
|
80
80
|
|
|
81
|
-
apptainer --silent
|
|
81
|
+
apptainer --silent exec docker://doejgi/jamo-dori jamo link -s dori custom '{"metadata.library_name":"NPUNN","metadata.fastq_type":"sdm_normal","group":"sdm"}'
|
|
82
82
|
NPUNN /global/dna/dm_archive/sdm/pacbio/00/27/47/pbio-2747.27352.bc1001_BAK8A_OA--bc1001_BAK8A_OA.ccs.fastq.gz BACKUP_COMPLETE 6391936239a7711d789a9380
|
|
83
83
|
NPUNN /clusterfs/jgi/groups/dsi/homes/dnscott/git/jamofetch/data/NPUNN.pbio-2747.27352.bc1001_BAK8A_OA--bc1001_BAK8A_OA.ccs.fastq.gz
|
|
84
84
|
|
|
85
|
-
apptainer --silent
|
|
85
|
+
apptainer --silent exec docker://doejgi/jamo-dori jamo link -s dori custom '{"metadata.library_name":"NOOHG","metadata.fastq_type":"sdm_normal","group":"sdm"}'
|
|
86
86
|
NOOHG /global/dna/dm_archive/sdm/pacbio/00/26/91/pbio-2691.26653.bc1001_BAK8A_OA--bc1001_BAK8A_OA.ccs.fastq.gz RESTORED 6347dbb35bc59487d7e768d6
|
|
87
87
|
NOOHG /clusterfs/jgi/groups/dsi/homes/dnscott/git/jamofetch/data/NOOHG.pbio-2691.26653.bc1001_BAK8A_OA--bc1001_BAK8A_OA.ccs.fastq.gz
|
|
88
88
|
|
|
89
|
-
apptainer --silent
|
|
89
|
+
apptainer --silent exec docker://doejgi/jamo-dori jamo link -s dori custom '{"metadata.library_name":"HOGH","metadata.fastq_type":"sdm_normal","group":"sdm"}'
|
|
90
90
|
HOGH /global/dna/dm_archive/sdm/illumina/00/63/97/6397.2.44053.GGCTAC.fastq.gz RESTORED 51d52a82067c014cd6ef4f6f
|
|
91
91
|
HOGH /clusterfs/jgi/groups/dsi/homes/dnscott/git/jamofetch/data/HOGH.6397.2.44053.GGCTAC.fastq.gz
|
|
92
92
|
|
|
@@ -0,0 +1,427 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
|
|
3
|
+
import json
|
|
4
|
+
import logging
|
|
5
|
+
import os
|
|
6
|
+
import pathlib
|
|
7
|
+
import re
|
|
8
|
+
import subprocess
|
|
9
|
+
import sys
|
|
10
|
+
import time
|
|
11
|
+
from argparse import ArgumentParser
|
|
12
|
+
|
|
13
|
+
# Base commands; get_cmd() appends the query.
|
|
14
|
+
#
|
|
15
|
+
# Dori uses `apptainer exec`, not `run`: the jamo-dori image's runscript strips the
|
|
16
|
+
# double quotes out of the JSON query, and jamo then fails with a JSONDecodeError.
|
|
17
|
+
JAMO_CMD_NERSC = 'module load jamo; jamo link'
|
|
18
|
+
JAMO_CMD_DORI = 'apptainer --silent exec docker://doejgi/jamo-dori jamo link -s dori'
|
|
19
|
+
|
|
20
|
+
TWO_HOURS = 7200 # seconds
|
|
21
|
+
ONE_MINUTE = 60 # seconds
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def get_cmd(clean_lib_name, smrt_cell_id=None, physical_run_id=None) -> str:
|
|
25
|
+
"""
|
|
26
|
+
Build the site-appropriate `jamo link` command for a library's SDM fastq.
|
|
27
|
+
|
|
28
|
+
Uses a custom query rather than `jamo link library <name>`. That form applies
|
|
29
|
+
jamo's default `raw_normal` filter, {'metadata.fastq_type': 'sdm_normal',
|
|
30
|
+
'user': 'sdm'}, and files from SDM's newer pipeline are owned by user
|
|
31
|
+
'sdm_pipeline' (group still 'sdm'), so it matched nothing for them. Filtering
|
|
32
|
+
on group instead of user matches both.
|
|
33
|
+
|
|
34
|
+
Key order matters: jamo names each symlink '<first string-valued query key's
|
|
35
|
+
value>.<file_name>', so metadata.library_name must stay first or
|
|
36
|
+
_find_fastq_path() will not find the link. The disambiguators below are
|
|
37
|
+
integers, so they cannot take that position, but they are appended last
|
|
38
|
+
anyway.
|
|
39
|
+
|
|
40
|
+
A library sequenced on more than one run matches one fastq per run, and
|
|
41
|
+
without a disambiguator NOTHING here chooses between them - the caller ends
|
|
42
|
+
up with several symlinks and _find_fastq_path() refuses to guess. Pass
|
|
43
|
+
smrt_cell_id (preferred) or physical_run_id to narrow the query to the
|
|
44
|
+
sequencing this analysis is actually about.
|
|
45
|
+
|
|
46
|
+
Observed case, library LBBCZRJ: four matching records across three pools -
|
|
47
|
+
cell 37397 (run 3207, 1,296,169 reads), cell 37892 (run 3232, 1,965,608
|
|
48
|
+
reads, registered twice under different AUTO- folders with identical md5s)
|
|
49
|
+
and cell 37964 (run 3230, 948,348 reads). Adding the cell id returns exactly
|
|
50
|
+
one record in the first and third cases, and in the second the two records
|
|
51
|
+
are byte-identical and share a file_name, so they collapse to one symlink.
|
|
52
|
+
|
|
53
|
+
smrt_cell_id is the finer key and is the one to prefer: it identifies a
|
|
54
|
+
library's membership in one specific pool. It is JAMO's
|
|
55
|
+
metadata.sdm_smrt_cell_id, and equals the PacBio pipeline service's
|
|
56
|
+
libraries[].id for the same library.
|
|
57
|
+
|
|
58
|
+
Args:
|
|
59
|
+
clean_lib_name: the library name, e.g. 'LBBCZRJ'.
|
|
60
|
+
smrt_cell_id: optional metadata.sdm_smrt_cell_id to narrow to.
|
|
61
|
+
physical_run_id: optional metadata.pacbio_physical_run_id to narrow to.
|
|
62
|
+
Coarser than smrt_cell_id; a run holds many cells.
|
|
63
|
+
|
|
64
|
+
Returns:
|
|
65
|
+
The full shell command, with the JSON query single-quoted.
|
|
66
|
+
"""
|
|
67
|
+
# Validated because the name is interpolated into a shell command.
|
|
68
|
+
lib_name = _clean_library_name(clean_lib_name)
|
|
69
|
+
query = {
|
|
70
|
+
"metadata.library_name": lib_name,
|
|
71
|
+
"metadata.fastq_type": "sdm_normal",
|
|
72
|
+
"group": "sdm",
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
cell = _check_id(smrt_cell_id, 'smrt_cell_id')
|
|
76
|
+
if cell is not None:
|
|
77
|
+
query["metadata.sdm_smrt_cell_id"] = cell
|
|
78
|
+
|
|
79
|
+
run = _check_id(physical_run_id, 'physical_run_id')
|
|
80
|
+
if run is not None:
|
|
81
|
+
query["metadata.pacbio_physical_run_id"] = run
|
|
82
|
+
|
|
83
|
+
query = json.dumps(query, separators=(',', ':'))
|
|
84
|
+
base = JAMO_CMD_DORI if os.getenv('SLURM_PARTITION') == 'dori' else JAMO_CMD_NERSC
|
|
85
|
+
return f"{base} custom '{query}'"
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
def _check_id(value, name):
|
|
89
|
+
"""
|
|
90
|
+
Normalise an optional integer query value, or raise.
|
|
91
|
+
|
|
92
|
+
Accepts an int or a string of digits (ids arrive from JSON APIs and from
|
|
93
|
+
argparse as either) and returns an int, or None when nothing was given.
|
|
94
|
+
Anything else raises: these values are interpolated into a shell command, so
|
|
95
|
+
a stray string must never reach it.
|
|
96
|
+
"""
|
|
97
|
+
if value is None:
|
|
98
|
+
return None
|
|
99
|
+
if isinstance(value, bool):
|
|
100
|
+
raise ValueError(f"invalid {name}: expecting a positive integer, got {value!r}")
|
|
101
|
+
if isinstance(value, str):
|
|
102
|
+
stripped = value.strip()
|
|
103
|
+
if not stripped.isdigit():
|
|
104
|
+
raise ValueError(f"invalid {name}: expecting a positive integer, got {value!r}")
|
|
105
|
+
value = int(stripped)
|
|
106
|
+
if not isinstance(value, int):
|
|
107
|
+
raise ValueError(f"invalid {name}: expecting a positive integer, got {value!r}")
|
|
108
|
+
if value <= 0:
|
|
109
|
+
raise ValueError(f"invalid {name}: expecting a positive integer, got {value!r}")
|
|
110
|
+
return value
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
def _clean_library_name(library_name):
|
|
114
|
+
clean_name = f"{library_name}".strip()
|
|
115
|
+
if not re.match(r'^[A-Z]+$', clean_name):
|
|
116
|
+
raise ValueError(f"invalid library name {library_name}")
|
|
117
|
+
return clean_name
|
|
118
|
+
|
|
119
|
+
|
|
120
|
+
def _check_int_not_negative(value_to_check, value_name, allow_minus_1=False) -> int:
|
|
121
|
+
"""
|
|
122
|
+
Throw an error if value_to_check is not a positive integer. Provide
|
|
123
|
+
value_name in error message.
|
|
124
|
+
"""
|
|
125
|
+
if not value_to_check:
|
|
126
|
+
return 0
|
|
127
|
+
if not isinstance(value_to_check, int):
|
|
128
|
+
raise ValueError(f"invalid value for {value_name}: expecting integer")
|
|
129
|
+
if allow_minus_1 and value_to_check == -1:
|
|
130
|
+
return -1
|
|
131
|
+
if value_to_check <= 0:
|
|
132
|
+
raise ValueError(f"invalid value for {value_name}: expecting positive integer")
|
|
133
|
+
return value_to_check
|
|
134
|
+
|
|
135
|
+
|
|
136
|
+
def _wait_for_seq(seq_path, wait_interval_secs, wait_max_secs):
|
|
137
|
+
"""
|
|
138
|
+
Check if seq_path is a link. If seq_path is a broken link, wait until it is valid.
|
|
139
|
+
Check link each wait_interval_secs seconds. Exit when link is valid. Throw
|
|
140
|
+
an error if wait_max_secs is exceeded. Set wait_max_secs to None to wait
|
|
141
|
+
indefinitely.
|
|
142
|
+
"""
|
|
143
|
+
if seq_path is None:
|
|
144
|
+
raise ValueError("null seq_path")
|
|
145
|
+
|
|
146
|
+
if not os.path.islink(seq_path):
|
|
147
|
+
raise ValueError("seq_path is not a link")
|
|
148
|
+
|
|
149
|
+
wait_interval = _check_int_not_negative(wait_interval_secs, 'wait_interval_secs')
|
|
150
|
+
wait_max = _check_int_not_negative(wait_max_secs, 'wait_max_secs', allow_minus_1=True)
|
|
151
|
+
|
|
152
|
+
total_wait = 0
|
|
153
|
+
if wait_max and wait_interval:
|
|
154
|
+
logging.debug(f"waiting for link {seq_path}; wait_interval: {wait_interval}; wait_max: {wait_max}")
|
|
155
|
+
while (not os.path.exists(seq_path)) and (wait_max == -1 or total_wait < wait_max):
|
|
156
|
+
time.sleep(wait_interval_secs)
|
|
157
|
+
if wait_interval > 0:
|
|
158
|
+
total_wait += wait_interval
|
|
159
|
+
|
|
160
|
+
if not os.path.exists(seq_path):
|
|
161
|
+
raise RuntimeError(f"sequence for link {seq_path} not provisioned within max wait seconds: {wait_max}")
|
|
162
|
+
|
|
163
|
+
|
|
164
|
+
def _matching_links(lib_name, seq_dir):
|
|
165
|
+
"""
|
|
166
|
+
Return every symlink in seq_dir that jamo created for this library, as a
|
|
167
|
+
list of (link_path, target) sorted by link name.
|
|
168
|
+
|
|
169
|
+
jamo names each link '<library>.<file_name>', so the prefix is what
|
|
170
|
+
identifies ownership. Targets are read with os.readlink rather than
|
|
171
|
+
os.path.realpath: a link whose target has not been copied to this host yet
|
|
172
|
+
is still a real candidate, and realpath would obscure that.
|
|
173
|
+
"""
|
|
174
|
+
links = []
|
|
175
|
+
for file_name in sorted(os.listdir(os.path.realpath(seq_dir))):
|
|
176
|
+
file_path = os.path.join(seq_dir, file_name)
|
|
177
|
+
if not os.path.islink(file_path):
|
|
178
|
+
continue
|
|
179
|
+
if file_name.startswith(f"{lib_name}.") and file_name.endswith('.fastq.gz'):
|
|
180
|
+
links.append((file_path, os.readlink(file_path)))
|
|
181
|
+
return links
|
|
182
|
+
|
|
183
|
+
|
|
184
|
+
def _find_fastq_path(lib_name, seq_dir):
|
|
185
|
+
"""
|
|
186
|
+
Return the one sequence symlink jamo created for this library.
|
|
187
|
+
|
|
188
|
+
This used to sort the directory by ctime and keep the last match, which was
|
|
189
|
+
a coin flip whenever a library matched more than one fastq: every link comes
|
|
190
|
+
from a single `jamo link` call within the same second, ST_CTIME is whole
|
|
191
|
+
seconds, so all the candidates tie and Python's stable sort just hands back
|
|
192
|
+
whichever os.listdir() happened to return last. That is directory hash order
|
|
193
|
+
on Linux - not alphabetical, not newest, and not the same on two hosts. A
|
|
194
|
+
library sequenced three times could therefore be analysed against any of the
|
|
195
|
+
three, silently, with a 2x spread in read count and nothing in the log to
|
|
196
|
+
say which.
|
|
197
|
+
|
|
198
|
+
So it no longer guesses. One candidate is returned; several distinct targets
|
|
199
|
+
raise, naming every one of them. Narrow the query instead - pass
|
|
200
|
+
smrt_cell_id or physical_run_id to get_cmd()/fetch_lib_seq() - or clear out
|
|
201
|
+
stale links from an earlier fetch.
|
|
202
|
+
|
|
203
|
+
Several links pointing at the SAME target are not ambiguous and are accepted:
|
|
204
|
+
SDM registers the same file under more than one path (observed for LBBCZRJ,
|
|
205
|
+
two AUTO- folders, identical md5).
|
|
206
|
+
"""
|
|
207
|
+
links = _matching_links(lib_name, seq_dir)
|
|
208
|
+
|
|
209
|
+
if not links:
|
|
210
|
+
raise RuntimeError(f"failed to find sequence file for library {lib_name}")
|
|
211
|
+
|
|
212
|
+
distinct_targets = {target for _, target in links}
|
|
213
|
+
if len(distinct_targets) > 1:
|
|
214
|
+
listing = "\n".join(
|
|
215
|
+
f" {os.path.basename(link)} -> {target}" for link, target in links
|
|
216
|
+
)
|
|
217
|
+
raise RuntimeError(
|
|
218
|
+
f"library {lib_name} matches {len(distinct_targets)} different "
|
|
219
|
+
f"sequence files in {seq_dir}; refusing to pick one:\n{listing}\n"
|
|
220
|
+
" Narrow the query with smrt_cell_id (preferred) or "
|
|
221
|
+
"physical_run_id, or remove links left over from an earlier fetch."
|
|
222
|
+
)
|
|
223
|
+
|
|
224
|
+
logging.debug(f"returning library seq file: {links[0][0]}")
|
|
225
|
+
return links[0][0]
|
|
226
|
+
|
|
227
|
+
|
|
228
|
+
class LibSeq:
|
|
229
|
+
|
|
230
|
+
def __init__(self, lib_name, seq_path):
|
|
231
|
+
self._lib_name = _clean_library_name(lib_name)
|
|
232
|
+
self._seq_path = seq_path
|
|
233
|
+
|
|
234
|
+
def get_lib_name(self) -> str:
|
|
235
|
+
return self._lib_name
|
|
236
|
+
|
|
237
|
+
def get_seq_path(self):
|
|
238
|
+
return self._seq_path
|
|
239
|
+
|
|
240
|
+
def seq_exists(self) -> bool:
|
|
241
|
+
return os.path.exists(self._seq_path)
|
|
242
|
+
|
|
243
|
+
def get_real_path(self):
|
|
244
|
+
if self._seq_path is None:
|
|
245
|
+
return None
|
|
246
|
+
return os.path.realpath(self._seq_path)
|
|
247
|
+
|
|
248
|
+
def get_real_path_wait(self):
|
|
249
|
+
if self.seq_exists():
|
|
250
|
+
return self.get_real_path()
|
|
251
|
+
else:
|
|
252
|
+
raise RuntimeError(f"{self}: sequence file not available")
|
|
253
|
+
|
|
254
|
+
def __str__(self):
|
|
255
|
+
f"{self.__class__.__name__}(lib={self._lib_name};path={self._seq_path})"
|
|
256
|
+
|
|
257
|
+
|
|
258
|
+
class JamoLibSeq(LibSeq):
|
|
259
|
+
|
|
260
|
+
def __init__(self, lib_name, seq_path, wait_interval_secs=10, wait_max_secs=-1):
|
|
261
|
+
LibSeq.__init__(self, lib_name, seq_path)
|
|
262
|
+
self._wait_interval_secs = _check_int_not_negative(wait_interval_secs, 'wait_interval_secs')
|
|
263
|
+
self._wait_max_secs = _check_int_not_negative(wait_max_secs, 'wait_max_secs', allow_minus_1=True)
|
|
264
|
+
self._sequence_ready = False
|
|
265
|
+
|
|
266
|
+
def get_real_path_wait(self):
|
|
267
|
+
if not self._sequence_ready:
|
|
268
|
+
_wait_for_seq(self._seq_path, self._wait_interval_secs, self._wait_max_secs)
|
|
269
|
+
self._sequence_ready = True
|
|
270
|
+
return os.path.realpath(self._seq_path)
|
|
271
|
+
|
|
272
|
+
|
|
273
|
+
class JamoFetcher():
|
|
274
|
+
def __init__(self, link_dir='.', wait_interval_secs=10, wait_max_secs=-1):
|
|
275
|
+
self._wait_interval_secs = _check_int_not_negative(wait_interval_secs, 'wait_interval_secs')
|
|
276
|
+
self._wait_max_secs = _check_int_not_negative(wait_max_secs, 'wait_max_secs', allow_minus_1=True)
|
|
277
|
+
self._link_dir = os.path.realpath(link_dir)
|
|
278
|
+
|
|
279
|
+
def fetch_lib_seq(self, lib_name, out_file=sys.stderr, smrt_cell_id=None,
|
|
280
|
+
physical_run_id=None) -> JamoLibSeq:
|
|
281
|
+
"""
|
|
282
|
+
Execute JAMO command to link sequence for library
|
|
283
|
+
and return a LibSeq object containing the path to the sequence.
|
|
284
|
+
|
|
285
|
+
smrt_cell_id (preferred) or physical_run_id narrow the query to one
|
|
286
|
+
sequencing of this library. Without one, a library sequenced more than
|
|
287
|
+
once links several files and _find_fastq_path() raises rather than pick
|
|
288
|
+
between them - see get_cmd().
|
|
289
|
+
|
|
290
|
+
They are per-call rather than per-fetcher because one JamoFetcher is
|
|
291
|
+
commonly shared across a pool's libraries, and each library has its own
|
|
292
|
+
cell id.
|
|
293
|
+
"""
|
|
294
|
+
pathlib.Path(self._link_dir).mkdir(parents=True, exist_ok=True, mode=0o775)
|
|
295
|
+
os.chdir(self._link_dir)
|
|
296
|
+
clean_lib_name = _clean_library_name(lib_name)
|
|
297
|
+
|
|
298
|
+
cmd = get_cmd(clean_lib_name, smrt_cell_id=smrt_cell_id,
|
|
299
|
+
physical_run_id=physical_run_id)
|
|
300
|
+
print(f"\n{cmd}", file=out_file)
|
|
301
|
+
output = subprocess.check_output(cmd, shell=True)
|
|
302
|
+
for line in output.splitlines():
|
|
303
|
+
print(line.decode("utf-8"), file=out_file)
|
|
304
|
+
|
|
305
|
+
# Report what was linked before choosing, so an ambiguous fetch is
|
|
306
|
+
# visible in the log even though the next line raises on it.
|
|
307
|
+
links = _matching_links(clean_lib_name, self._link_dir)
|
|
308
|
+
if len(links) > 1:
|
|
309
|
+
print(f"{clean_lib_name}: {len(links)} candidate links in "
|
|
310
|
+
f"{self._link_dir}", file=out_file)
|
|
311
|
+
for link, target in links:
|
|
312
|
+
print(f" {os.path.basename(link)} -> {target}", file=out_file)
|
|
313
|
+
|
|
314
|
+
seq_path = _find_fastq_path(clean_lib_name, self._link_dir)
|
|
315
|
+
print(f"{clean_lib_name} {seq_path}", file=out_file)
|
|
316
|
+
return JamoLibSeq(clean_lib_name, seq_path, self._wait_interval_secs, self._wait_max_secs)
|
|
317
|
+
|
|
318
|
+
|
|
319
|
+
def _fetch_seq(fetcher: JamoFetcher, libs, smrt_cell_id=None, physical_run_id=None):
|
|
320
|
+
lib_seq_dict = {}
|
|
321
|
+
for lib in libs:
|
|
322
|
+
try:
|
|
323
|
+
lib_seq: JamoLibSeq = fetcher.fetch_lib_seq(
|
|
324
|
+
lib, out_file=sys.stderr, smrt_cell_id=smrt_cell_id,
|
|
325
|
+
physical_run_id=physical_run_id)
|
|
326
|
+
lib_seq_dict[lib_seq.get_lib_name()] = lib_seq
|
|
327
|
+
except Exception as e:
|
|
328
|
+
# Was a bare `pass`, which hid every failure - including the
|
|
329
|
+
# ambiguous-match error, whose whole purpose is to be seen.
|
|
330
|
+
print(f"{lib}: {e}", file=sys.stderr)
|
|
331
|
+
return lib_seq_dict
|
|
332
|
+
|
|
333
|
+
|
|
334
|
+
def main(args):
|
|
335
|
+
logging.debug(f"args: {args}")
|
|
336
|
+
|
|
337
|
+
if not args.library:
|
|
338
|
+
return
|
|
339
|
+
|
|
340
|
+
wait_interval_secs = _check_int_not_negative(args.interval, "interval")
|
|
341
|
+
wait_max_secs = _check_int_not_negative(args.max, "max", allow_minus_1=True)
|
|
342
|
+
|
|
343
|
+
# A cell id identifies one library's membership in one pool, so it cannot be
|
|
344
|
+
# shared across several -l libraries. A run id can: a run holds many cells.
|
|
345
|
+
if args.smrt_cell_id is not None and len(args.library) > 1:
|
|
346
|
+
raise ValueError(
|
|
347
|
+
"--smrt-cell-id applies to a single library; it was given with "
|
|
348
|
+
f"{len(args.library)} -l options. Use --physical-run-id, or fetch "
|
|
349
|
+
"one library at a time.")
|
|
350
|
+
|
|
351
|
+
link_dir = os.path.realpath(args.directory if args.directory else '.')
|
|
352
|
+
fetcher: JamoFetcher = JamoFetcher(link_dir=link_dir, wait_interval_secs=wait_interval_secs,
|
|
353
|
+
wait_max_secs=wait_max_secs)
|
|
354
|
+
|
|
355
|
+
print("fetching sequence:")
|
|
356
|
+
lib_seq_dict = _fetch_seq(fetcher, args.library,
|
|
357
|
+
smrt_cell_id=args.smrt_cell_id,
|
|
358
|
+
physical_run_id=args.physical_run_id)
|
|
359
|
+
|
|
360
|
+
if not lib_seq_dict:
|
|
361
|
+
return # no sequence was fetched
|
|
362
|
+
|
|
363
|
+
print("\nsequence links:")
|
|
364
|
+
for lib_name, lib_seq in sorted(lib_seq_dict.items()):
|
|
365
|
+
print(f"{lib_seq.get_lib_name()} symlink: {lib_seq.get_seq_path()}")
|
|
366
|
+
print(f"{lib_seq.get_lib_name()} realpath: {lib_seq.get_real_path()}")
|
|
367
|
+
|
|
368
|
+
if args.wait:
|
|
369
|
+
total_wait = 0
|
|
370
|
+
seq_ready = set()
|
|
371
|
+
print("\nwaiting for JAMO to provision sequence . . . .")
|
|
372
|
+
while (len(seq_ready) < len(lib_seq_dict)) and (wait_max_secs == -1 or total_wait <= wait_max_secs):
|
|
373
|
+
for lib_name, lib_seq in sorted(lib_seq_dict.items()):
|
|
374
|
+
if not lib_name in seq_ready and lib_seq.seq_exists():
|
|
375
|
+
print(f"{lib_name} sequence ready")
|
|
376
|
+
seq_ready.add(lib_name)
|
|
377
|
+
if not wait_interval_secs or not wait_max_secs:
|
|
378
|
+
break # do not wait
|
|
379
|
+
if len(seq_ready) < len(lib_seq_dict):
|
|
380
|
+
time.sleep(wait_interval_secs)
|
|
381
|
+
if wait_max_secs > 0:
|
|
382
|
+
total_wait += wait_interval_secs
|
|
383
|
+
if len(seq_ready) < len(lib_seq_dict):
|
|
384
|
+
print("\nexiting, not all sequence ready")
|
|
385
|
+
if wait_max_secs != -1 and total_wait >= wait_max_secs:
|
|
386
|
+
print(f"max wait {wait_max_secs} exceeded")
|
|
387
|
+
|
|
388
|
+
|
|
389
|
+
def cli():
|
|
390
|
+
try:
|
|
391
|
+
parser: ArgumentParser = ArgumentParser()
|
|
392
|
+
parser.add_argument('-l', '--library', action='append',
|
|
393
|
+
help="library name(s) for which to retrieve sequence")
|
|
394
|
+
parser.add_argument('-d', '--directory', required=False, default='.',
|
|
395
|
+
help="directory where to link sequence, defaults to current directory. " +
|
|
396
|
+
"Directory will be created if it doesn't exit.")
|
|
397
|
+
parser.add_argument('--smrt-cell-id', required=False, type=int, default=None,
|
|
398
|
+
help="metadata.sdm_smrt_cell_id to fetch, for a library sequenced " +
|
|
399
|
+
"more than once. Identifies one library in one pool, so it " +
|
|
400
|
+
"applies to a single --library.")
|
|
401
|
+
parser.add_argument('--physical-run-id', required=False, type=int, default=None,
|
|
402
|
+
help="metadata.pacbio_physical_run_id to fetch, for a library " +
|
|
403
|
+
"sequenced more than once. Coarser than --smrt-cell-id; " +
|
|
404
|
+
"applies to every --library given.")
|
|
405
|
+
parser.add_argument('-i', '--interval', required=False, type=int, default=10,
|
|
406
|
+
help="wait interval in seconds to check if sequence has been fetched, " +
|
|
407
|
+
"ignored if wait flag not set")
|
|
408
|
+
parser.add_argument('-m', '--max', required=False, type=int, default=TWO_HOURS,
|
|
409
|
+
help="maximum time to wait for sequence in seconds, " +
|
|
410
|
+
"ignored if wait flag not set. Specify -1 to wait indefinetely.")
|
|
411
|
+
parser.add_argument('-w', '--wait', action='store_true',
|
|
412
|
+
help='wait for jamo to link sequence, then print "sequence ready"')
|
|
413
|
+
parser.add_argument('--logging', required=False, default='WARN',
|
|
414
|
+
help="logging level (specify DEBUG for verbose logging)")
|
|
415
|
+
|
|
416
|
+
ARGS = parser.parse_args()
|
|
417
|
+
# logging.basicConfig(format='%(asctime)s %(message)s', datefmt='%m/%d/%Y %I:%M:%S %p', level=log_level)
|
|
418
|
+
logging.basicConfig(format='%(message)s', level=ARGS.logging.upper())
|
|
419
|
+
main(ARGS)
|
|
420
|
+
except KeyboardInterrupt:
|
|
421
|
+
print('Interrupted', file=sys.stderr)
|
|
422
|
+
# https://unix.stackexchange.com/questions/251996/why-does-bash-set-exit-status-to-non-zero-on-ctrl-c-or-ctrl-z
|
|
423
|
+
sys.exit(130)
|
|
424
|
+
|
|
425
|
+
|
|
426
|
+
if __name__ == "__main__":
|
|
427
|
+
cli()
|
|
@@ -0,0 +1,184 @@
|
|
|
1
|
+
import json
|
|
2
|
+
import os
|
|
3
|
+
import shlex
|
|
4
|
+
|
|
5
|
+
import pytest
|
|
6
|
+
|
|
7
|
+
from jamofetch import jamofetch
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
def _query(cmd):
|
|
11
|
+
# The query is the single-quoted final argument; shlex undoes the shell quoting.
|
|
12
|
+
return json.loads(shlex.split(cmd)[-1])
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def test_nersc_cmd(monkeypatch):
|
|
16
|
+
monkeypatch.delenv('SLURM_PARTITION', raising=False)
|
|
17
|
+
cmd = jamofetch.get_cmd('LBBDHFZ')
|
|
18
|
+
assert cmd.startswith('module load jamo; jamo link custom ')
|
|
19
|
+
assert _query(cmd) == {
|
|
20
|
+
"metadata.library_name": "LBBDHFZ",
|
|
21
|
+
"metadata.fastq_type": "sdm_normal",
|
|
22
|
+
"group": "sdm",
|
|
23
|
+
}
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def test_dori_cmd_uses_exec(monkeypatch):
|
|
27
|
+
monkeypatch.setenv('SLURM_PARTITION', 'dori')
|
|
28
|
+
cmd = jamofetch.get_cmd('LBBDHFZ')
|
|
29
|
+
assert cmd.startswith('apptainer --silent exec docker://doejgi/jamo-dori jamo link -s dori custom ')
|
|
30
|
+
assert _query(cmd)["metadata.library_name"] == "LBBDHFZ"
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def test_library_name_is_first_query_key(monkeypatch):
|
|
34
|
+
# jamo names the symlink after the first string-valued key; _find_fastq_path
|
|
35
|
+
# relies on that being the library name.
|
|
36
|
+
monkeypatch.delenv('SLURM_PARTITION', raising=False)
|
|
37
|
+
assert next(iter(_query(jamofetch.get_cmd('LBBDHFZ')))) == "metadata.library_name"
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def test_query_does_not_filter_on_user(monkeypatch):
|
|
41
|
+
monkeypatch.delenv('SLURM_PARTITION', raising=False)
|
|
42
|
+
assert "user" not in _query(jamofetch.get_cmd('LBBDHFZ'))
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
@pytest.mark.parametrize('bad', ["LBB'DHFZ", 'lbbdhfz', 'LBB DHFZ', 'LBB;rm'])
|
|
46
|
+
def test_invalid_library_name_rejected(bad):
|
|
47
|
+
with pytest.raises(ValueError):
|
|
48
|
+
jamofetch.get_cmd(bad)
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
# --- disambiguating a library sequenced more than once -----------------------
|
|
52
|
+
#
|
|
53
|
+
# Real case this exists for, library LBBCZRJ: four records matching the plain
|
|
54
|
+
# query, across three pools -- cell 37397 (run 3207), cell 37892 (run 3232,
|
|
55
|
+
# registered twice with identical md5s) and cell 37964 (run 3230). Read counts
|
|
56
|
+
# 1,296,169 / 1,965,608 / 948,348. Before 3.7.9 the one analysed was whichever
|
|
57
|
+
# os.listdir() returned last.
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def test_smrt_cell_id_narrows_query(monkeypatch):
|
|
61
|
+
monkeypatch.delenv('SLURM_PARTITION', raising=False)
|
|
62
|
+
query = _query(jamofetch.get_cmd('LBBCZRJ', smrt_cell_id=37892))
|
|
63
|
+
assert query == {
|
|
64
|
+
"metadata.library_name": "LBBCZRJ",
|
|
65
|
+
"metadata.fastq_type": "sdm_normal",
|
|
66
|
+
"group": "sdm",
|
|
67
|
+
"metadata.sdm_smrt_cell_id": 37892,
|
|
68
|
+
}
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def test_physical_run_id_narrows_query(monkeypatch):
|
|
72
|
+
monkeypatch.delenv('SLURM_PARTITION', raising=False)
|
|
73
|
+
query = _query(jamofetch.get_cmd('LBBCZRJ', physical_run_id=3232))
|
|
74
|
+
assert query["metadata.pacbio_physical_run_id"] == 3232
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def test_both_ids_may_be_given(monkeypatch):
|
|
78
|
+
monkeypatch.delenv('SLURM_PARTITION', raising=False)
|
|
79
|
+
query = _query(jamofetch.get_cmd('LBBCZRJ', smrt_cell_id=37892, physical_run_id=3232))
|
|
80
|
+
assert query["metadata.sdm_smrt_cell_id"] == 37892
|
|
81
|
+
assert query["metadata.pacbio_physical_run_id"] == 3232
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
def test_library_name_still_first_key_with_ids(monkeypatch):
|
|
85
|
+
# jamo names the symlink after the first string-valued key. The ids are
|
|
86
|
+
# ints so they cannot take that slot, but the ordering is load-bearing
|
|
87
|
+
# enough to assert directly.
|
|
88
|
+
monkeypatch.delenv('SLURM_PARTITION', raising=False)
|
|
89
|
+
query = _query(jamofetch.get_cmd('LBBCZRJ', smrt_cell_id=37892))
|
|
90
|
+
assert next(iter(query)) == "metadata.library_name"
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
def test_no_ids_produces_the_3_7_8_command(monkeypatch):
|
|
94
|
+
# Backward compatibility: callers that pass no id must get the byte-identical
|
|
95
|
+
# command 3.7.8 produced. synbioqc-pbj pins these strings in its own tests.
|
|
96
|
+
monkeypatch.delenv('SLURM_PARTITION', raising=False)
|
|
97
|
+
assert jamofetch.get_cmd('LBBDHFZ') == (
|
|
98
|
+
"module load jamo; jamo link custom "
|
|
99
|
+
'\'{"metadata.library_name":"LBBDHFZ","metadata.fastq_type":"sdm_normal","group":"sdm"}\''
|
|
100
|
+
)
|
|
101
|
+
|
|
102
|
+
|
|
103
|
+
@pytest.mark.parametrize('bad', ['37892; rm -rf /', 'abc', '', -1, 0, 3.5, True, [37892]])
|
|
104
|
+
def test_invalid_ids_rejected(bad):
|
|
105
|
+
with pytest.raises(ValueError):
|
|
106
|
+
jamofetch.get_cmd('LBBCZRJ', smrt_cell_id=bad)
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
def test_numeric_string_id_accepted(monkeypatch):
|
|
110
|
+
# Ids arrive from JSON APIs and argparse as either int or str.
|
|
111
|
+
monkeypatch.delenv('SLURM_PARTITION', raising=False)
|
|
112
|
+
assert _query(jamofetch.get_cmd('LBBCZRJ', smrt_cell_id='37892'))[
|
|
113
|
+
"metadata.sdm_smrt_cell_id"] == 37892
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
# --- _find_fastq_path no longer guesses --------------------------------------
|
|
117
|
+
|
|
118
|
+
LBBCZRJ_TARGETS = {
|
|
119
|
+
"pbio-3207.37397.bc2080_OA--bc2080_OA.hifi_reads.bc2080_OA.ccs.fastq.gz":
|
|
120
|
+
"/global/dna/dm_archive/sdm/pacbio/00/32/07",
|
|
121
|
+
"pbr6184.3232.37663.37892.bc2080_OA--bc2080_OA.pacbio.processed_well.37663.37892.fastq.gz":
|
|
122
|
+
"/global/dna/dm_archive/sdm/analyses-242/AUTO-2421306",
|
|
123
|
+
"pbr6186.3230.37667.37964.bc2080_OA--bc2080_OA.pacbio.processed_well.37667.37964.fastq.gz":
|
|
124
|
+
"/global/dna/dm_archive/sdm/analyses-242/AUTO-2421406",
|
|
125
|
+
}
|
|
126
|
+
|
|
127
|
+
|
|
128
|
+
def _link(seq_dir, lib, file_name, folder):
|
|
129
|
+
link = os.path.join(seq_dir, f"{lib}.{file_name}")
|
|
130
|
+
os.symlink(os.path.join(folder, file_name), link)
|
|
131
|
+
return link
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
def test_single_match_is_returned(tmp_path):
|
|
135
|
+
name, folder = next(iter(LBBCZRJ_TARGETS.items()))
|
|
136
|
+
_link(str(tmp_path), 'LBBCZRJ', name, folder)
|
|
137
|
+
assert jamofetch._find_fastq_path('LBBCZRJ', str(tmp_path)).endswith(name)
|
|
138
|
+
|
|
139
|
+
|
|
140
|
+
def test_ambiguous_match_raises_and_names_every_candidate(tmp_path):
|
|
141
|
+
for name, folder in LBBCZRJ_TARGETS.items():
|
|
142
|
+
_link(str(tmp_path), 'LBBCZRJ', name, folder)
|
|
143
|
+
|
|
144
|
+
with pytest.raises(RuntimeError) as excinfo:
|
|
145
|
+
jamofetch._find_fastq_path('LBBCZRJ', str(tmp_path))
|
|
146
|
+
|
|
147
|
+
message = str(excinfo.value)
|
|
148
|
+
assert "3 different" in message
|
|
149
|
+
for name, folder in LBBCZRJ_TARGETS.items():
|
|
150
|
+
assert os.path.join(folder, name) in message
|
|
151
|
+
assert "smrt_cell_id" in message
|
|
152
|
+
|
|
153
|
+
|
|
154
|
+
def test_duplicate_registrations_of_one_file_are_not_ambiguous(tmp_path):
|
|
155
|
+
# SDM registered the same file under two AUTO- folders with identical md5s.
|
|
156
|
+
# Same file_name means one link name, so only one link can exist -- but if
|
|
157
|
+
# two links ever point at the same target, that is not a real choice.
|
|
158
|
+
name = "pbr6184.3232.37663.37892.bc2080_OA--bc2080_OA.pacbio.processed_well.37663.37892.fastq.gz"
|
|
159
|
+
target = "/global/dna/dm_archive/sdm/analyses-242/AUTO-2421306/" + name
|
|
160
|
+
os.symlink(target, os.path.join(str(tmp_path), f"LBBCZRJ.{name}"))
|
|
161
|
+
os.symlink(target, os.path.join(str(tmp_path), f"LBBCZRJ.copy.{name}"))
|
|
162
|
+
assert jamofetch._find_fastq_path('LBBCZRJ', str(tmp_path)) is not None
|
|
163
|
+
|
|
164
|
+
|
|
165
|
+
def test_other_libraries_links_are_ignored(tmp_path):
|
|
166
|
+
name, folder = next(iter(LBBCZRJ_TARGETS.items()))
|
|
167
|
+
_link(str(tmp_path), 'LBBCZRJ', name, folder)
|
|
168
|
+
for other_name, other_folder in list(LBBCZRJ_TARGETS.items())[1:]:
|
|
169
|
+
_link(str(tmp_path), 'LBBDHFZ', other_name, other_folder)
|
|
170
|
+
assert jamofetch._find_fastq_path('LBBCZRJ', str(tmp_path)).endswith(name)
|
|
171
|
+
|
|
172
|
+
|
|
173
|
+
def test_no_match_raises(tmp_path):
|
|
174
|
+
with pytest.raises(RuntimeError, match="failed to find sequence file"):
|
|
175
|
+
jamofetch._find_fastq_path('LBBCZRJ', str(tmp_path))
|
|
176
|
+
|
|
177
|
+
|
|
178
|
+
def test_broken_link_is_still_a_candidate(tmp_path):
|
|
179
|
+
# On dori the target is copied into a local cache and the link is broken
|
|
180
|
+
# until that lands; it is still the file this library resolved to.
|
|
181
|
+
name, folder = next(iter(LBBCZRJ_TARGETS.items()))
|
|
182
|
+
link = _link(str(tmp_path), 'LBBCZRJ', name, folder)
|
|
183
|
+
assert not os.path.exists(link)
|
|
184
|
+
assert jamofetch._find_fastq_path('LBBCZRJ', str(tmp_path)) == link
|
jamofetch-3.7.7/CHANGELOG.md
DELETED
|
@@ -1,271 +0,0 @@
|
|
|
1
|
-
#!/usr/bin/env python3
|
|
2
|
-
|
|
3
|
-
import logging
|
|
4
|
-
import os
|
|
5
|
-
import pathlib
|
|
6
|
-
import re
|
|
7
|
-
import subprocess
|
|
8
|
-
import sys
|
|
9
|
-
import time
|
|
10
|
-
from argparse import ArgumentParser
|
|
11
|
-
from stat import ST_CTIME
|
|
12
|
-
|
|
13
|
-
JAMO_CMD_NERSC = 'module load jamo; jamo link library'
|
|
14
|
-
JAMO_CMD_DORI = 'apptainer --silent run docker://doejgi/jamo-dori jamo link -s dori library'
|
|
15
|
-
|
|
16
|
-
TWO_HOURS = 7200 # seconds
|
|
17
|
-
ONE_MINUTE = 60 # seconds
|
|
18
|
-
|
|
19
|
-
|
|
20
|
-
def get_cmd(clean_lib_name) -> str:
|
|
21
|
-
if os.getenv('SLURM_PARTITION') == 'dori':
|
|
22
|
-
return f"{JAMO_CMD_DORI} {clean_lib_name}"
|
|
23
|
-
return f"{JAMO_CMD_NERSC} {clean_lib_name}"
|
|
24
|
-
|
|
25
|
-
|
|
26
|
-
def _clean_library_name(library_name):
|
|
27
|
-
clean_name = f"{library_name}".strip()
|
|
28
|
-
if not re.match(r'^[A-Z]+$', clean_name):
|
|
29
|
-
raise ValueError(f"invalid library name {library_name}")
|
|
30
|
-
return clean_name
|
|
31
|
-
|
|
32
|
-
|
|
33
|
-
def _check_int_not_negative(value_to_check, value_name, allow_minus_1=False) -> int:
|
|
34
|
-
"""
|
|
35
|
-
Throw an error if value_to_check is not a positive integer. Provide
|
|
36
|
-
value_name in error message.
|
|
37
|
-
"""
|
|
38
|
-
if not value_to_check:
|
|
39
|
-
return 0
|
|
40
|
-
if not isinstance(value_to_check, int):
|
|
41
|
-
raise ValueError(f"invalid value for {value_name}: expecting integer")
|
|
42
|
-
if allow_minus_1 and value_to_check == -1:
|
|
43
|
-
return -1
|
|
44
|
-
if value_to_check <= 0:
|
|
45
|
-
raise ValueError(f"invalid value for {value_name}: expecting positive integer")
|
|
46
|
-
return value_to_check
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
def _wait_for_seq(seq_path, wait_interval_secs, wait_max_secs):
|
|
50
|
-
"""
|
|
51
|
-
Check if seq_path is a link. If seq_path is a broken link, wait until it is valid.
|
|
52
|
-
Check link each wait_interval_secs seconds. Exit when link is valid. Throw
|
|
53
|
-
an error if wait_max_secs is exceeded. Set wait_max_secs to None to wait
|
|
54
|
-
indefinitely.
|
|
55
|
-
"""
|
|
56
|
-
if seq_path is None:
|
|
57
|
-
raise ValueError("null seq_path")
|
|
58
|
-
|
|
59
|
-
if not os.path.islink(seq_path):
|
|
60
|
-
raise ValueError("seq_path is not a link")
|
|
61
|
-
|
|
62
|
-
wait_interval = _check_int_not_negative(wait_interval_secs, 'wait_interval_secs')
|
|
63
|
-
wait_max = _check_int_not_negative(wait_max_secs, 'wait_max_secs', allow_minus_1=True)
|
|
64
|
-
|
|
65
|
-
total_wait = 0
|
|
66
|
-
if wait_max and wait_interval:
|
|
67
|
-
logging.debug(f"waiting for link {seq_path}; wait_interval: {wait_interval}; wait_max: {wait_max}")
|
|
68
|
-
while (not os.path.exists(seq_path)) and (wait_max == -1 or total_wait < wait_max):
|
|
69
|
-
time.sleep(wait_interval_secs)
|
|
70
|
-
if wait_interval > 0:
|
|
71
|
-
total_wait += wait_interval
|
|
72
|
-
|
|
73
|
-
if not os.path.exists(seq_path):
|
|
74
|
-
raise RuntimeError(f"sequence for link {seq_path} not provisioned within max wait seconds: {wait_max}")
|
|
75
|
-
|
|
76
|
-
|
|
77
|
-
def _find_fastq_path(lib_name, seq_dir):
|
|
78
|
-
"""
|
|
79
|
-
Find symlink created for library in sequence directory
|
|
80
|
-
by matching pattern similar to:
|
|
81
|
-
HHCHO.52485.1.359430.GTGCTTA-GTAAGCA.fastq.gz
|
|
82
|
-
"""
|
|
83
|
-
# sort contents in directory by modification time (we want the last modified)
|
|
84
|
-
# adapted from:
|
|
85
|
-
# https://www.tutorialspoint.com/How-do-you-get-a-directory-listing-sorted-by-creation-date-in-Python
|
|
86
|
-
file_paths = [os.path.join(seq_dir, file_name) for file_name in os.listdir(os.path.realpath(seq_dir))]
|
|
87
|
-
logging.debug(f"file paths: {file_paths}")
|
|
88
|
-
# get file stats
|
|
89
|
-
path_stats = [(path, os.lstat(path)) for path in file_paths]
|
|
90
|
-
path_stats.sort(key=lambda x: x[1][ST_CTIME])
|
|
91
|
-
for path_stat in path_stats:
|
|
92
|
-
logging.debug(f"{path_stat[0]}:")
|
|
93
|
-
for stat in sorted(filter(lambda a: a.startswith('st_'), dir(path_stat[1]))):
|
|
94
|
-
logging.debug(f" {stat}: {getattr(path_stat[1], stat)}")
|
|
95
|
-
|
|
96
|
-
matched_link = None
|
|
97
|
-
for path_stat in path_stats:
|
|
98
|
-
filepath = path_stat[0]
|
|
99
|
-
filetime = path_stat[1][ST_CTIME]
|
|
100
|
-
filename = os.path.basename(filepath)
|
|
101
|
-
if not os.path.islink(filepath):
|
|
102
|
-
continue
|
|
103
|
-
logging.debug(f"link candidate: {filename}; creation time: {filetime}")
|
|
104
|
-
if str(filename).startswith(f"{lib_name}.") and str(filename).endswith('.fastq.gz'):
|
|
105
|
-
matched_link = filepath
|
|
106
|
-
logging.debug(f"link match: {filename}")
|
|
107
|
-
|
|
108
|
-
if matched_link is not None:
|
|
109
|
-
logging.debug(f"returning library seq file: {matched_link}")
|
|
110
|
-
return matched_link
|
|
111
|
-
|
|
112
|
-
raise RuntimeError(f"failed to find sequence file for library {lib_name}")
|
|
113
|
-
|
|
114
|
-
|
|
115
|
-
class LibSeq:
|
|
116
|
-
|
|
117
|
-
def __init__(self, lib_name, seq_path):
|
|
118
|
-
self._lib_name = _clean_library_name(lib_name)
|
|
119
|
-
self._seq_path = seq_path
|
|
120
|
-
|
|
121
|
-
def get_lib_name(self) -> str:
|
|
122
|
-
return self._lib_name
|
|
123
|
-
|
|
124
|
-
def get_seq_path(self):
|
|
125
|
-
return self._seq_path
|
|
126
|
-
|
|
127
|
-
def seq_exists(self) -> bool:
|
|
128
|
-
return os.path.exists(self._seq_path)
|
|
129
|
-
|
|
130
|
-
def get_real_path(self):
|
|
131
|
-
if self._seq_path is None:
|
|
132
|
-
return None
|
|
133
|
-
return os.path.realpath(self._seq_path)
|
|
134
|
-
|
|
135
|
-
def get_real_path_wait(self):
|
|
136
|
-
if self.seq_exists():
|
|
137
|
-
return self.get_real_path()
|
|
138
|
-
else:
|
|
139
|
-
raise RuntimeError(f"{self}: sequence file not available")
|
|
140
|
-
|
|
141
|
-
def __str__(self):
|
|
142
|
-
f"{self.__class__.__name__}(lib={self._lib_name};path={self._seq_path})"
|
|
143
|
-
|
|
144
|
-
|
|
145
|
-
class JamoLibSeq(LibSeq):
|
|
146
|
-
|
|
147
|
-
def __init__(self, lib_name, seq_path, wait_interval_secs=10, wait_max_secs=-1):
|
|
148
|
-
LibSeq.__init__(self, lib_name, seq_path)
|
|
149
|
-
self._wait_interval_secs = _check_int_not_negative(wait_interval_secs, 'wait_interval_secs')
|
|
150
|
-
self._wait_max_secs = _check_int_not_negative(wait_max_secs, 'wait_max_secs', allow_minus_1=True)
|
|
151
|
-
self._sequence_ready = False
|
|
152
|
-
|
|
153
|
-
def get_real_path_wait(self):
|
|
154
|
-
if not self._sequence_ready:
|
|
155
|
-
_wait_for_seq(self._seq_path, self._wait_interval_secs, self._wait_max_secs)
|
|
156
|
-
self._sequence_ready = True
|
|
157
|
-
return os.path.realpath(self._seq_path)
|
|
158
|
-
|
|
159
|
-
|
|
160
|
-
class JamoFetcher():
|
|
161
|
-
def __init__(self, link_dir='.', wait_interval_secs=10, wait_max_secs=-1):
|
|
162
|
-
self._wait_interval_secs = _check_int_not_negative(wait_interval_secs, 'wait_interval_secs')
|
|
163
|
-
self._wait_max_secs = _check_int_not_negative(wait_max_secs, 'wait_max_secs', allow_minus_1=True)
|
|
164
|
-
self._link_dir = os.path.realpath(link_dir)
|
|
165
|
-
|
|
166
|
-
def fetch_lib_seq(self, lib_name, out_file=sys.stderr) -> JamoLibSeq:
|
|
167
|
-
"""
|
|
168
|
-
Execute JAMO command to link sequence for library
|
|
169
|
-
and return a LibSeq object containing the path to the sequence.
|
|
170
|
-
"""
|
|
171
|
-
pathlib.Path(self._link_dir).mkdir(parents=True, exist_ok=True, mode=0o775)
|
|
172
|
-
os.chdir(self._link_dir)
|
|
173
|
-
clean_lib_name = _clean_library_name(lib_name)
|
|
174
|
-
|
|
175
|
-
cmd = get_cmd(clean_lib_name)
|
|
176
|
-
print(f"\n{cmd}", file=out_file)
|
|
177
|
-
output = subprocess.check_output(cmd, shell=True)
|
|
178
|
-
for line in output.splitlines():
|
|
179
|
-
print(line.decode("utf-8"), file=out_file)
|
|
180
|
-
seq_path = _find_fastq_path(clean_lib_name, self._link_dir)
|
|
181
|
-
print(f"{clean_lib_name} {seq_path}", file=out_file)
|
|
182
|
-
return JamoLibSeq(clean_lib_name, seq_path, self._wait_interval_secs, self._wait_max_secs)
|
|
183
|
-
|
|
184
|
-
|
|
185
|
-
def _fetch_seq(fetcher: JamoFetcher, libs):
|
|
186
|
-
lib_seq_dict = {}
|
|
187
|
-
for lib in libs:
|
|
188
|
-
try:
|
|
189
|
-
lib_seq: JamoLibSeq = fetcher.fetch_lib_seq(lib, out_file=sys.stderr)
|
|
190
|
-
lib_seq_dict[lib_seq.get_lib_name()] = lib_seq
|
|
191
|
-
except Exception as e:
|
|
192
|
-
pass
|
|
193
|
-
return lib_seq_dict
|
|
194
|
-
|
|
195
|
-
|
|
196
|
-
def main(args):
|
|
197
|
-
logging.debug(f"args: {args}")
|
|
198
|
-
|
|
199
|
-
if not args.library:
|
|
200
|
-
return
|
|
201
|
-
|
|
202
|
-
wait_interval_secs = _check_int_not_negative(args.interval, "interval")
|
|
203
|
-
wait_max_secs = _check_int_not_negative(args.max, "max", allow_minus_1=True)
|
|
204
|
-
|
|
205
|
-
link_dir = os.path.realpath(args.directory if args.directory else '.')
|
|
206
|
-
fetcher: JamoFetcher = JamoFetcher(link_dir=link_dir, wait_interval_secs=wait_interval_secs,
|
|
207
|
-
wait_max_secs=wait_max_secs)
|
|
208
|
-
|
|
209
|
-
print("fetching sequence:")
|
|
210
|
-
lib_seq_dict = _fetch_seq(fetcher, args.library)
|
|
211
|
-
|
|
212
|
-
if not lib_seq_dict:
|
|
213
|
-
return # no sequence was fetched
|
|
214
|
-
|
|
215
|
-
print("\nsequence links:")
|
|
216
|
-
for lib_name, lib_seq in sorted(lib_seq_dict.items()):
|
|
217
|
-
print(f"{lib_seq.get_lib_name()} symlink: {lib_seq.get_seq_path()}")
|
|
218
|
-
print(f"{lib_seq.get_lib_name()} realpath: {lib_seq.get_real_path()}")
|
|
219
|
-
|
|
220
|
-
if args.wait:
|
|
221
|
-
total_wait = 0
|
|
222
|
-
seq_ready = set()
|
|
223
|
-
print("\nwaiting for JAMO to provision sequence . . . .")
|
|
224
|
-
while (len(seq_ready) < len(lib_seq_dict)) and (wait_max_secs == -1 or total_wait <= wait_max_secs):
|
|
225
|
-
for lib_name, lib_seq in sorted(lib_seq_dict.items()):
|
|
226
|
-
if not lib_name in seq_ready and lib_seq.seq_exists():
|
|
227
|
-
print(f"{lib_name} sequence ready")
|
|
228
|
-
seq_ready.add(lib_name)
|
|
229
|
-
if not wait_interval_secs or not wait_max_secs:
|
|
230
|
-
break # do not wait
|
|
231
|
-
if len(seq_ready) < len(lib_seq_dict):
|
|
232
|
-
time.sleep(wait_interval_secs)
|
|
233
|
-
if wait_max_secs > 0:
|
|
234
|
-
total_wait += wait_interval_secs
|
|
235
|
-
if len(seq_ready) < len(lib_seq_dict):
|
|
236
|
-
print("\nexiting, not all sequence ready")
|
|
237
|
-
if wait_max_secs != -1 and total_wait >= wait_max_secs:
|
|
238
|
-
print(f"max wait {wait_max_secs} exceeded")
|
|
239
|
-
|
|
240
|
-
|
|
241
|
-
def cli():
|
|
242
|
-
try:
|
|
243
|
-
parser: ArgumentParser = ArgumentParser()
|
|
244
|
-
parser.add_argument('-l', '--library', action='append',
|
|
245
|
-
help="library name(s) for which to retrieve sequence")
|
|
246
|
-
parser.add_argument('-d', '--directory', required=False, default='.',
|
|
247
|
-
help="directory where to link sequence, defaults to current directory. " +
|
|
248
|
-
"Directory will be created if it doesn't exit.")
|
|
249
|
-
parser.add_argument('-i', '--interval', required=False, type=int, default=10,
|
|
250
|
-
help="wait interval in seconds to check if sequence has been fetched, " +
|
|
251
|
-
"ignored if wait flag not set")
|
|
252
|
-
parser.add_argument('-m', '--max', required=False, type=int, default=TWO_HOURS,
|
|
253
|
-
help="maximum time to wait for sequence in seconds, " +
|
|
254
|
-
"ignored if wait flag not set. Specify -1 to wait indefinetely.")
|
|
255
|
-
parser.add_argument('-w', '--wait', action='store_true',
|
|
256
|
-
help='wait for jamo to link sequence, then print "sequence ready"')
|
|
257
|
-
parser.add_argument('--logging', required=False, default='WARN',
|
|
258
|
-
help="logging level (specify DEBUG for verbose logging)")
|
|
259
|
-
|
|
260
|
-
ARGS = parser.parse_args()
|
|
261
|
-
# logging.basicConfig(format='%(asctime)s %(message)s', datefmt='%m/%d/%Y %I:%M:%S %p', level=log_level)
|
|
262
|
-
logging.basicConfig(format='%(message)s', level=ARGS.logging.upper())
|
|
263
|
-
main(ARGS)
|
|
264
|
-
except KeyboardInterrupt:
|
|
265
|
-
print('Interrupted', file=sys.stderr)
|
|
266
|
-
# https://unix.stackexchange.com/questions/251996/why-does-bash-set-exit-status-to-non-zero-on-ctrl-c-or-ctrl-z
|
|
267
|
-
sys.exit(130)
|
|
268
|
-
|
|
269
|
-
|
|
270
|
-
if __name__ == "__main__":
|
|
271
|
-
cli()
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|