jamofetch 3.7.7__tar.gz → 3.7.9__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (26) hide show
  1. jamofetch-3.7.9/CHANGELOG.md +53 -0
  2. {jamofetch-3.7.7 → jamofetch-3.7.9}/PKG-INFO +4 -4
  3. {jamofetch-3.7.7 → jamofetch-3.7.9}/README.md +3 -3
  4. {jamofetch-3.7.7 → jamofetch-3.7.9}/pyproject.toml +1 -1
  5. jamofetch-3.7.9/src/jamofetch/jamofetch.py +427 -0
  6. jamofetch-3.7.9/tests/test_jamofetch.py +184 -0
  7. jamofetch-3.7.7/CHANGELOG.md +0 -7
  8. jamofetch-3.7.7/src/jamofetch/jamofetch.py +0 -271
  9. jamofetch-3.7.7/tests/test_jamofetch.py +0 -2
  10. {jamofetch-3.7.7 → jamofetch-3.7.9}/.gitignore +0 -0
  11. {jamofetch-3.7.7 → jamofetch-3.7.9}/.gitlab-ci.yml +0 -0
  12. {jamofetch-3.7.7 → jamofetch-3.7.9}/.readthedocs.yml +0 -0
  13. {jamofetch-3.7.7 → jamofetch-3.7.9}/CONDUCT.md +0 -0
  14. {jamofetch-3.7.7 → jamofetch-3.7.9}/CONTRIBUTING.md +0 -0
  15. {jamofetch-3.7.7 → jamofetch-3.7.9}/docs/Makefile +0 -0
  16. {jamofetch-3.7.7 → jamofetch-3.7.9}/docs/changelog.md +0 -0
  17. {jamofetch-3.7.7 → jamofetch-3.7.9}/docs/conduct.md +0 -0
  18. {jamofetch-3.7.7 → jamofetch-3.7.9}/docs/conf.py +0 -0
  19. {jamofetch-3.7.7 → jamofetch-3.7.9}/docs/contributing.md +0 -0
  20. {jamofetch-3.7.7 → jamofetch-3.7.9}/docs/demo_script.py +0 -0
  21. {jamofetch-3.7.7 → jamofetch-3.7.9}/docs/index.md +0 -0
  22. {jamofetch-3.7.7 → jamofetch-3.7.9}/docs/make.bat +0 -0
  23. {jamofetch-3.7.7 → jamofetch-3.7.9}/docs/requirements.txt +0 -0
  24. {jamofetch-3.7.7 → jamofetch-3.7.9}/prepare-gitlab-publish.sh +0 -0
  25. {jamofetch-3.7.7 → jamofetch-3.7.9}/push-version-tag.sh +0 -0
  26. {jamofetch-3.7.7 → jamofetch-3.7.9}/src/jamofetch/__init__.py +0 -0
@@ -0,0 +1,53 @@
1
+ # Changelog
2
+
3
+ <!--next-version-placeholder-->
4
+
5
+ ## v3.7.9 (17/09/2026)
6
+
7
+ ### Fixed
8
+
9
+ - **A library sequenced more than once no longer resolves to an arbitrary file.**
10
+ `_find_fastq_path()` sorted the link directory by `ST_CTIME` and kept the last
11
+ match. Every link comes from one `jamo link` call within the same second and
12
+ `ST_CTIME` is whole seconds, so the candidates all tie and Python's stable sort
13
+ returns whichever `os.listdir()` listed last - directory hash order on Linux,
14
+ so not alphabetical, not newest, and not necessarily the same on two hosts.
15
+ It now returns the single match, or raises and names every candidate.
16
+
17
+ Observed on library LBBCZRJ: four matching records across three pools - cell
18
+ 37397 (run 3207, 1,296,169 reads), cell 37892 (run 3232, 1,965,608 reads,
19
+ registered twice under different `AUTO-` folders with identical md5s) and cell
20
+ 37964 (run 3230, 948,348 reads). Any of the three could be analysed, silently,
21
+ with a 2x spread in read count.
22
+
23
+ Links pointing at the same target are not ambiguous and are still accepted.
24
+
25
+ ### Added
26
+
27
+ - `get_cmd()` and `JamoFetcher.fetch_lib_seq()` take `smrt_cell_id` and
28
+ `physical_run_id`, which narrow the query to one sequencing of the library.
29
+ `smrt_cell_id` is the one to prefer: it identifies a library's membership in a
30
+ single pool, and equals the PacBio pipeline service's `libraries[].id`. For
31
+ LBBCZRJ it reduces four matches to one (or, for cell 37892, to two records
32
+ that are byte-identical and share a `file_name`).
33
+ - CLI: `--smrt-cell-id` (single `--library` only) and `--physical-run-id`
34
+ (applies to every `--library`).
35
+
36
+ ### Changed
37
+
38
+ - The CLI no longer swallows per-library fetch errors. `_fetch_seq()` caught
39
+ every exception with a bare `pass`, which hid the new ambiguity error whose
40
+ whole purpose is to be seen.
41
+ - `JamoFetcher.fetch_lib_seq()` logs every candidate link when more than one
42
+ matches, before selection.
43
+
44
+ ### Compatibility
45
+
46
+ - Calls that pass no id produce the byte-identical command 3.7.8 produced;
47
+ asserted in `tests/test_jamofetch.py`. Callers that relied on a library with
48
+ several matches silently resolving to one of them now get a `RuntimeError`
49
+ instead - that is the point of the release.
50
+
51
+ ## v0.1.0 (07/07/2023)
52
+
53
+ - First release of `jamofetch`!
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: jamofetch
3
- Version: 3.7.7
3
+ Version: 3.7.9
4
4
  Summary: A thin wrapper to retrieve sequence from JAMO at NERSC and on Dori.
5
5
  Author: Duncan Scott
6
6
  Requires-Python: >=3.9
@@ -89,15 +89,15 @@ options:
89
89
  (venv) [dnscott@ln005 jamofetch]$ jamofetch -d data -l NPUNN -l NOOHG -l HOGH -w --max -1
90
90
  fetching sequence:
91
91
 
92
- apptainer --silent run docker://doejgi/jamo-dori jamo link -s dori library NPUNN
92
+ apptainer --silent exec docker://doejgi/jamo-dori jamo link -s dori custom '{"metadata.library_name":"NPUNN","metadata.fastq_type":"sdm_normal","group":"sdm"}'
93
93
  NPUNN /global/dna/dm_archive/sdm/pacbio/00/27/47/pbio-2747.27352.bc1001_BAK8A_OA--bc1001_BAK8A_OA.ccs.fastq.gz BACKUP_COMPLETE 6391936239a7711d789a9380
94
94
  NPUNN /clusterfs/jgi/groups/dsi/homes/dnscott/git/jamofetch/data/NPUNN.pbio-2747.27352.bc1001_BAK8A_OA--bc1001_BAK8A_OA.ccs.fastq.gz
95
95
 
96
- apptainer --silent run docker://doejgi/jamo-dori jamo link -s dori library NOOHG
96
+ apptainer --silent exec docker://doejgi/jamo-dori jamo link -s dori custom '{"metadata.library_name":"NOOHG","metadata.fastq_type":"sdm_normal","group":"sdm"}'
97
97
  NOOHG /global/dna/dm_archive/sdm/pacbio/00/26/91/pbio-2691.26653.bc1001_BAK8A_OA--bc1001_BAK8A_OA.ccs.fastq.gz RESTORED 6347dbb35bc59487d7e768d6
98
98
  NOOHG /clusterfs/jgi/groups/dsi/homes/dnscott/git/jamofetch/data/NOOHG.pbio-2691.26653.bc1001_BAK8A_OA--bc1001_BAK8A_OA.ccs.fastq.gz
99
99
 
100
- apptainer --silent run docker://doejgi/jamo-dori jamo link -s dori library HOGH
100
+ apptainer --silent exec docker://doejgi/jamo-dori jamo link -s dori custom '{"metadata.library_name":"HOGH","metadata.fastq_type":"sdm_normal","group":"sdm"}'
101
101
  HOGH /global/dna/dm_archive/sdm/illumina/00/63/97/6397.2.44053.GGCTAC.fastq.gz RESTORED 51d52a82067c014cd6ef4f6f
102
102
  HOGH /clusterfs/jgi/groups/dsi/homes/dnscott/git/jamofetch/data/HOGH.6397.2.44053.GGCTAC.fastq.gz
103
103
 
@@ -78,15 +78,15 @@ options:
78
78
  (venv) [dnscott@ln005 jamofetch]$ jamofetch -d data -l NPUNN -l NOOHG -l HOGH -w --max -1
79
79
  fetching sequence:
80
80
 
81
- apptainer --silent run docker://doejgi/jamo-dori jamo link -s dori library NPUNN
81
+ apptainer --silent exec docker://doejgi/jamo-dori jamo link -s dori custom '{"metadata.library_name":"NPUNN","metadata.fastq_type":"sdm_normal","group":"sdm"}'
82
82
  NPUNN /global/dna/dm_archive/sdm/pacbio/00/27/47/pbio-2747.27352.bc1001_BAK8A_OA--bc1001_BAK8A_OA.ccs.fastq.gz BACKUP_COMPLETE 6391936239a7711d789a9380
83
83
  NPUNN /clusterfs/jgi/groups/dsi/homes/dnscott/git/jamofetch/data/NPUNN.pbio-2747.27352.bc1001_BAK8A_OA--bc1001_BAK8A_OA.ccs.fastq.gz
84
84
 
85
- apptainer --silent run docker://doejgi/jamo-dori jamo link -s dori library NOOHG
85
+ apptainer --silent exec docker://doejgi/jamo-dori jamo link -s dori custom '{"metadata.library_name":"NOOHG","metadata.fastq_type":"sdm_normal","group":"sdm"}'
86
86
  NOOHG /global/dna/dm_archive/sdm/pacbio/00/26/91/pbio-2691.26653.bc1001_BAK8A_OA--bc1001_BAK8A_OA.ccs.fastq.gz RESTORED 6347dbb35bc59487d7e768d6
87
87
  NOOHG /clusterfs/jgi/groups/dsi/homes/dnscott/git/jamofetch/data/NOOHG.pbio-2691.26653.bc1001_BAK8A_OA--bc1001_BAK8A_OA.ccs.fastq.gz
88
88
 
89
- apptainer --silent run docker://doejgi/jamo-dori jamo link -s dori library HOGH
89
+ apptainer --silent exec docker://doejgi/jamo-dori jamo link -s dori custom '{"metadata.library_name":"HOGH","metadata.fastq_type":"sdm_normal","group":"sdm"}'
90
90
  HOGH /global/dna/dm_archive/sdm/illumina/00/63/97/6397.2.44053.GGCTAC.fastq.gz RESTORED 51d52a82067c014cd6ef4f6f
91
91
  HOGH /clusterfs/jgi/groups/dsi/homes/dnscott/git/jamofetch/data/HOGH.6397.2.44053.GGCTAC.fastq.gz
92
92
 
@@ -4,7 +4,7 @@ build-backend = "flit_core.buildapi"
4
4
 
5
5
  [project]
6
6
  name = "jamofetch"
7
- version = "3.7.7"
7
+ version = "3.7.9"
8
8
  description = "A thin wrapper to retrieve sequence from JAMO at NERSC and on Dori."
9
9
  authors = [{ name = "Duncan Scott" }]
10
10
  readme = "README.md"
@@ -0,0 +1,427 @@
1
+ #!/usr/bin/env python3
2
+
3
+ import json
4
+ import logging
5
+ import os
6
+ import pathlib
7
+ import re
8
+ import subprocess
9
+ import sys
10
+ import time
11
+ from argparse import ArgumentParser
12
+
13
+ # Base commands; get_cmd() appends the query.
14
+ #
15
+ # Dori uses `apptainer exec`, not `run`: the jamo-dori image's runscript strips the
16
+ # double quotes out of the JSON query, and jamo then fails with a JSONDecodeError.
17
+ JAMO_CMD_NERSC = 'module load jamo; jamo link'
18
+ JAMO_CMD_DORI = 'apptainer --silent exec docker://doejgi/jamo-dori jamo link -s dori'
19
+
20
+ TWO_HOURS = 7200 # seconds
21
+ ONE_MINUTE = 60 # seconds
22
+
23
+
24
+ def get_cmd(clean_lib_name, smrt_cell_id=None, physical_run_id=None) -> str:
25
+ """
26
+ Build the site-appropriate `jamo link` command for a library's SDM fastq.
27
+
28
+ Uses a custom query rather than `jamo link library <name>`. That form applies
29
+ jamo's default `raw_normal` filter, {'metadata.fastq_type': 'sdm_normal',
30
+ 'user': 'sdm'}, and files from SDM's newer pipeline are owned by user
31
+ 'sdm_pipeline' (group still 'sdm'), so it matched nothing for them. Filtering
32
+ on group instead of user matches both.
33
+
34
+ Key order matters: jamo names each symlink '<first string-valued query key's
35
+ value>.<file_name>', so metadata.library_name must stay first or
36
+ _find_fastq_path() will not find the link. The disambiguators below are
37
+ integers, so they cannot take that position, but they are appended last
38
+ anyway.
39
+
40
+ A library sequenced on more than one run matches one fastq per run, and
41
+ without a disambiguator NOTHING here chooses between them - the caller ends
42
+ up with several symlinks and _find_fastq_path() refuses to guess. Pass
43
+ smrt_cell_id (preferred) or physical_run_id to narrow the query to the
44
+ sequencing this analysis is actually about.
45
+
46
+ Observed case, library LBBCZRJ: four matching records across three pools -
47
+ cell 37397 (run 3207, 1,296,169 reads), cell 37892 (run 3232, 1,965,608
48
+ reads, registered twice under different AUTO- folders with identical md5s)
49
+ and cell 37964 (run 3230, 948,348 reads). Adding the cell id returns exactly
50
+ one record in the first and third cases, and in the second the two records
51
+ are byte-identical and share a file_name, so they collapse to one symlink.
52
+
53
+ smrt_cell_id is the finer key and is the one to prefer: it identifies a
54
+ library's membership in one specific pool. It is JAMO's
55
+ metadata.sdm_smrt_cell_id, and equals the PacBio pipeline service's
56
+ libraries[].id for the same library.
57
+
58
+ Args:
59
+ clean_lib_name: the library name, e.g. 'LBBCZRJ'.
60
+ smrt_cell_id: optional metadata.sdm_smrt_cell_id to narrow to.
61
+ physical_run_id: optional metadata.pacbio_physical_run_id to narrow to.
62
+ Coarser than smrt_cell_id; a run holds many cells.
63
+
64
+ Returns:
65
+ The full shell command, with the JSON query single-quoted.
66
+ """
67
+ # Validated because the name is interpolated into a shell command.
68
+ lib_name = _clean_library_name(clean_lib_name)
69
+ query = {
70
+ "metadata.library_name": lib_name,
71
+ "metadata.fastq_type": "sdm_normal",
72
+ "group": "sdm",
73
+ }
74
+
75
+ cell = _check_id(smrt_cell_id, 'smrt_cell_id')
76
+ if cell is not None:
77
+ query["metadata.sdm_smrt_cell_id"] = cell
78
+
79
+ run = _check_id(physical_run_id, 'physical_run_id')
80
+ if run is not None:
81
+ query["metadata.pacbio_physical_run_id"] = run
82
+
83
+ query = json.dumps(query, separators=(',', ':'))
84
+ base = JAMO_CMD_DORI if os.getenv('SLURM_PARTITION') == 'dori' else JAMO_CMD_NERSC
85
+ return f"{base} custom '{query}'"
86
+
87
+
88
+ def _check_id(value, name):
89
+ """
90
+ Normalise an optional integer query value, or raise.
91
+
92
+ Accepts an int or a string of digits (ids arrive from JSON APIs and from
93
+ argparse as either) and returns an int, or None when nothing was given.
94
+ Anything else raises: these values are interpolated into a shell command, so
95
+ a stray string must never reach it.
96
+ """
97
+ if value is None:
98
+ return None
99
+ if isinstance(value, bool):
100
+ raise ValueError(f"invalid {name}: expecting a positive integer, got {value!r}")
101
+ if isinstance(value, str):
102
+ stripped = value.strip()
103
+ if not stripped.isdigit():
104
+ raise ValueError(f"invalid {name}: expecting a positive integer, got {value!r}")
105
+ value = int(stripped)
106
+ if not isinstance(value, int):
107
+ raise ValueError(f"invalid {name}: expecting a positive integer, got {value!r}")
108
+ if value <= 0:
109
+ raise ValueError(f"invalid {name}: expecting a positive integer, got {value!r}")
110
+ return value
111
+
112
+
113
+ def _clean_library_name(library_name):
114
+ clean_name = f"{library_name}".strip()
115
+ if not re.match(r'^[A-Z]+$', clean_name):
116
+ raise ValueError(f"invalid library name {library_name}")
117
+ return clean_name
118
+
119
+
120
+ def _check_int_not_negative(value_to_check, value_name, allow_minus_1=False) -> int:
121
+ """
122
+ Throw an error if value_to_check is not a positive integer. Provide
123
+ value_name in error message.
124
+ """
125
+ if not value_to_check:
126
+ return 0
127
+ if not isinstance(value_to_check, int):
128
+ raise ValueError(f"invalid value for {value_name}: expecting integer")
129
+ if allow_minus_1 and value_to_check == -1:
130
+ return -1
131
+ if value_to_check <= 0:
132
+ raise ValueError(f"invalid value for {value_name}: expecting positive integer")
133
+ return value_to_check
134
+
135
+
136
+ def _wait_for_seq(seq_path, wait_interval_secs, wait_max_secs):
137
+ """
138
+ Check if seq_path is a link. If seq_path is a broken link, wait until it is valid.
139
+ Check link each wait_interval_secs seconds. Exit when link is valid. Throw
140
+ an error if wait_max_secs is exceeded. Set wait_max_secs to None to wait
141
+ indefinitely.
142
+ """
143
+ if seq_path is None:
144
+ raise ValueError("null seq_path")
145
+
146
+ if not os.path.islink(seq_path):
147
+ raise ValueError("seq_path is not a link")
148
+
149
+ wait_interval = _check_int_not_negative(wait_interval_secs, 'wait_interval_secs')
150
+ wait_max = _check_int_not_negative(wait_max_secs, 'wait_max_secs', allow_minus_1=True)
151
+
152
+ total_wait = 0
153
+ if wait_max and wait_interval:
154
+ logging.debug(f"waiting for link {seq_path}; wait_interval: {wait_interval}; wait_max: {wait_max}")
155
+ while (not os.path.exists(seq_path)) and (wait_max == -1 or total_wait < wait_max):
156
+ time.sleep(wait_interval_secs)
157
+ if wait_interval > 0:
158
+ total_wait += wait_interval
159
+
160
+ if not os.path.exists(seq_path):
161
+ raise RuntimeError(f"sequence for link {seq_path} not provisioned within max wait seconds: {wait_max}")
162
+
163
+
164
+ def _matching_links(lib_name, seq_dir):
165
+ """
166
+ Return every symlink in seq_dir that jamo created for this library, as a
167
+ list of (link_path, target) sorted by link name.
168
+
169
+ jamo names each link '<library>.<file_name>', so the prefix is what
170
+ identifies ownership. Targets are read with os.readlink rather than
171
+ os.path.realpath: a link whose target has not been copied to this host yet
172
+ is still a real candidate, and realpath would obscure that.
173
+ """
174
+ links = []
175
+ for file_name in sorted(os.listdir(os.path.realpath(seq_dir))):
176
+ file_path = os.path.join(seq_dir, file_name)
177
+ if not os.path.islink(file_path):
178
+ continue
179
+ if file_name.startswith(f"{lib_name}.") and file_name.endswith('.fastq.gz'):
180
+ links.append((file_path, os.readlink(file_path)))
181
+ return links
182
+
183
+
184
+ def _find_fastq_path(lib_name, seq_dir):
185
+ """
186
+ Return the one sequence symlink jamo created for this library.
187
+
188
+ This used to sort the directory by ctime and keep the last match, which was
189
+ a coin flip whenever a library matched more than one fastq: every link comes
190
+ from a single `jamo link` call within the same second, ST_CTIME is whole
191
+ seconds, so all the candidates tie and Python's stable sort just hands back
192
+ whichever os.listdir() happened to return last. That is directory hash order
193
+ on Linux - not alphabetical, not newest, and not the same on two hosts. A
194
+ library sequenced three times could therefore be analysed against any of the
195
+ three, silently, with a 2x spread in read count and nothing in the log to
196
+ say which.
197
+
198
+ So it no longer guesses. One candidate is returned; several distinct targets
199
+ raise, naming every one of them. Narrow the query instead - pass
200
+ smrt_cell_id or physical_run_id to get_cmd()/fetch_lib_seq() - or clear out
201
+ stale links from an earlier fetch.
202
+
203
+ Several links pointing at the SAME target are not ambiguous and are accepted:
204
+ SDM registers the same file under more than one path (observed for LBBCZRJ,
205
+ two AUTO- folders, identical md5).
206
+ """
207
+ links = _matching_links(lib_name, seq_dir)
208
+
209
+ if not links:
210
+ raise RuntimeError(f"failed to find sequence file for library {lib_name}")
211
+
212
+ distinct_targets = {target for _, target in links}
213
+ if len(distinct_targets) > 1:
214
+ listing = "\n".join(
215
+ f" {os.path.basename(link)} -> {target}" for link, target in links
216
+ )
217
+ raise RuntimeError(
218
+ f"library {lib_name} matches {len(distinct_targets)} different "
219
+ f"sequence files in {seq_dir}; refusing to pick one:\n{listing}\n"
220
+ " Narrow the query with smrt_cell_id (preferred) or "
221
+ "physical_run_id, or remove links left over from an earlier fetch."
222
+ )
223
+
224
+ logging.debug(f"returning library seq file: {links[0][0]}")
225
+ return links[0][0]
226
+
227
+
228
+ class LibSeq:
229
+
230
+ def __init__(self, lib_name, seq_path):
231
+ self._lib_name = _clean_library_name(lib_name)
232
+ self._seq_path = seq_path
233
+
234
+ def get_lib_name(self) -> str:
235
+ return self._lib_name
236
+
237
+ def get_seq_path(self):
238
+ return self._seq_path
239
+
240
+ def seq_exists(self) -> bool:
241
+ return os.path.exists(self._seq_path)
242
+
243
+ def get_real_path(self):
244
+ if self._seq_path is None:
245
+ return None
246
+ return os.path.realpath(self._seq_path)
247
+
248
+ def get_real_path_wait(self):
249
+ if self.seq_exists():
250
+ return self.get_real_path()
251
+ else:
252
+ raise RuntimeError(f"{self}: sequence file not available")
253
+
254
+ def __str__(self):
255
+ f"{self.__class__.__name__}(lib={self._lib_name};path={self._seq_path})"
256
+
257
+
258
+ class JamoLibSeq(LibSeq):
259
+
260
+ def __init__(self, lib_name, seq_path, wait_interval_secs=10, wait_max_secs=-1):
261
+ LibSeq.__init__(self, lib_name, seq_path)
262
+ self._wait_interval_secs = _check_int_not_negative(wait_interval_secs, 'wait_interval_secs')
263
+ self._wait_max_secs = _check_int_not_negative(wait_max_secs, 'wait_max_secs', allow_minus_1=True)
264
+ self._sequence_ready = False
265
+
266
+ def get_real_path_wait(self):
267
+ if not self._sequence_ready:
268
+ _wait_for_seq(self._seq_path, self._wait_interval_secs, self._wait_max_secs)
269
+ self._sequence_ready = True
270
+ return os.path.realpath(self._seq_path)
271
+
272
+
273
+ class JamoFetcher():
274
+ def __init__(self, link_dir='.', wait_interval_secs=10, wait_max_secs=-1):
275
+ self._wait_interval_secs = _check_int_not_negative(wait_interval_secs, 'wait_interval_secs')
276
+ self._wait_max_secs = _check_int_not_negative(wait_max_secs, 'wait_max_secs', allow_minus_1=True)
277
+ self._link_dir = os.path.realpath(link_dir)
278
+
279
+ def fetch_lib_seq(self, lib_name, out_file=sys.stderr, smrt_cell_id=None,
280
+ physical_run_id=None) -> JamoLibSeq:
281
+ """
282
+ Execute JAMO command to link sequence for library
283
+ and return a LibSeq object containing the path to the sequence.
284
+
285
+ smrt_cell_id (preferred) or physical_run_id narrow the query to one
286
+ sequencing of this library. Without one, a library sequenced more than
287
+ once links several files and _find_fastq_path() raises rather than pick
288
+ between them - see get_cmd().
289
+
290
+ They are per-call rather than per-fetcher because one JamoFetcher is
291
+ commonly shared across a pool's libraries, and each library has its own
292
+ cell id.
293
+ """
294
+ pathlib.Path(self._link_dir).mkdir(parents=True, exist_ok=True, mode=0o775)
295
+ os.chdir(self._link_dir)
296
+ clean_lib_name = _clean_library_name(lib_name)
297
+
298
+ cmd = get_cmd(clean_lib_name, smrt_cell_id=smrt_cell_id,
299
+ physical_run_id=physical_run_id)
300
+ print(f"\n{cmd}", file=out_file)
301
+ output = subprocess.check_output(cmd, shell=True)
302
+ for line in output.splitlines():
303
+ print(line.decode("utf-8"), file=out_file)
304
+
305
+ # Report what was linked before choosing, so an ambiguous fetch is
306
+ # visible in the log even though the next line raises on it.
307
+ links = _matching_links(clean_lib_name, self._link_dir)
308
+ if len(links) > 1:
309
+ print(f"{clean_lib_name}: {len(links)} candidate links in "
310
+ f"{self._link_dir}", file=out_file)
311
+ for link, target in links:
312
+ print(f" {os.path.basename(link)} -> {target}", file=out_file)
313
+
314
+ seq_path = _find_fastq_path(clean_lib_name, self._link_dir)
315
+ print(f"{clean_lib_name} {seq_path}", file=out_file)
316
+ return JamoLibSeq(clean_lib_name, seq_path, self._wait_interval_secs, self._wait_max_secs)
317
+
318
+
319
+ def _fetch_seq(fetcher: JamoFetcher, libs, smrt_cell_id=None, physical_run_id=None):
320
+ lib_seq_dict = {}
321
+ for lib in libs:
322
+ try:
323
+ lib_seq: JamoLibSeq = fetcher.fetch_lib_seq(
324
+ lib, out_file=sys.stderr, smrt_cell_id=smrt_cell_id,
325
+ physical_run_id=physical_run_id)
326
+ lib_seq_dict[lib_seq.get_lib_name()] = lib_seq
327
+ except Exception as e:
328
+ # Was a bare `pass`, which hid every failure - including the
329
+ # ambiguous-match error, whose whole purpose is to be seen.
330
+ print(f"{lib}: {e}", file=sys.stderr)
331
+ return lib_seq_dict
332
+
333
+
334
+ def main(args):
335
+ logging.debug(f"args: {args}")
336
+
337
+ if not args.library:
338
+ return
339
+
340
+ wait_interval_secs = _check_int_not_negative(args.interval, "interval")
341
+ wait_max_secs = _check_int_not_negative(args.max, "max", allow_minus_1=True)
342
+
343
+ # A cell id identifies one library's membership in one pool, so it cannot be
344
+ # shared across several -l libraries. A run id can: a run holds many cells.
345
+ if args.smrt_cell_id is not None and len(args.library) > 1:
346
+ raise ValueError(
347
+ "--smrt-cell-id applies to a single library; it was given with "
348
+ f"{len(args.library)} -l options. Use --physical-run-id, or fetch "
349
+ "one library at a time.")
350
+
351
+ link_dir = os.path.realpath(args.directory if args.directory else '.')
352
+ fetcher: JamoFetcher = JamoFetcher(link_dir=link_dir, wait_interval_secs=wait_interval_secs,
353
+ wait_max_secs=wait_max_secs)
354
+
355
+ print("fetching sequence:")
356
+ lib_seq_dict = _fetch_seq(fetcher, args.library,
357
+ smrt_cell_id=args.smrt_cell_id,
358
+ physical_run_id=args.physical_run_id)
359
+
360
+ if not lib_seq_dict:
361
+ return # no sequence was fetched
362
+
363
+ print("\nsequence links:")
364
+ for lib_name, lib_seq in sorted(lib_seq_dict.items()):
365
+ print(f"{lib_seq.get_lib_name()} symlink: {lib_seq.get_seq_path()}")
366
+ print(f"{lib_seq.get_lib_name()} realpath: {lib_seq.get_real_path()}")
367
+
368
+ if args.wait:
369
+ total_wait = 0
370
+ seq_ready = set()
371
+ print("\nwaiting for JAMO to provision sequence . . . .")
372
+ while (len(seq_ready) < len(lib_seq_dict)) and (wait_max_secs == -1 or total_wait <= wait_max_secs):
373
+ for lib_name, lib_seq in sorted(lib_seq_dict.items()):
374
+ if not lib_name in seq_ready and lib_seq.seq_exists():
375
+ print(f"{lib_name} sequence ready")
376
+ seq_ready.add(lib_name)
377
+ if not wait_interval_secs or not wait_max_secs:
378
+ break # do not wait
379
+ if len(seq_ready) < len(lib_seq_dict):
380
+ time.sleep(wait_interval_secs)
381
+ if wait_max_secs > 0:
382
+ total_wait += wait_interval_secs
383
+ if len(seq_ready) < len(lib_seq_dict):
384
+ print("\nexiting, not all sequence ready")
385
+ if wait_max_secs != -1 and total_wait >= wait_max_secs:
386
+ print(f"max wait {wait_max_secs} exceeded")
387
+
388
+
389
+ def cli():
390
+ try:
391
+ parser: ArgumentParser = ArgumentParser()
392
+ parser.add_argument('-l', '--library', action='append',
393
+ help="library name(s) for which to retrieve sequence")
394
+ parser.add_argument('-d', '--directory', required=False, default='.',
395
+ help="directory where to link sequence, defaults to current directory. " +
396
+ "Directory will be created if it doesn't exit.")
397
+ parser.add_argument('--smrt-cell-id', required=False, type=int, default=None,
398
+ help="metadata.sdm_smrt_cell_id to fetch, for a library sequenced " +
399
+ "more than once. Identifies one library in one pool, so it " +
400
+ "applies to a single --library.")
401
+ parser.add_argument('--physical-run-id', required=False, type=int, default=None,
402
+ help="metadata.pacbio_physical_run_id to fetch, for a library " +
403
+ "sequenced more than once. Coarser than --smrt-cell-id; " +
404
+ "applies to every --library given.")
405
+ parser.add_argument('-i', '--interval', required=False, type=int, default=10,
406
+ help="wait interval in seconds to check if sequence has been fetched, " +
407
+ "ignored if wait flag not set")
408
+ parser.add_argument('-m', '--max', required=False, type=int, default=TWO_HOURS,
409
+ help="maximum time to wait for sequence in seconds, " +
410
+ "ignored if wait flag not set. Specify -1 to wait indefinetely.")
411
+ parser.add_argument('-w', '--wait', action='store_true',
412
+ help='wait for jamo to link sequence, then print "sequence ready"')
413
+ parser.add_argument('--logging', required=False, default='WARN',
414
+ help="logging level (specify DEBUG for verbose logging)")
415
+
416
+ ARGS = parser.parse_args()
417
+ # logging.basicConfig(format='%(asctime)s %(message)s', datefmt='%m/%d/%Y %I:%M:%S %p', level=log_level)
418
+ logging.basicConfig(format='%(message)s', level=ARGS.logging.upper())
419
+ main(ARGS)
420
+ except KeyboardInterrupt:
421
+ print('Interrupted', file=sys.stderr)
422
+ # https://unix.stackexchange.com/questions/251996/why-does-bash-set-exit-status-to-non-zero-on-ctrl-c-or-ctrl-z
423
+ sys.exit(130)
424
+
425
+
426
+ if __name__ == "__main__":
427
+ cli()
@@ -0,0 +1,184 @@
1
+ import json
2
+ import os
3
+ import shlex
4
+
5
+ import pytest
6
+
7
+ from jamofetch import jamofetch
8
+
9
+
10
+ def _query(cmd):
11
+ # The query is the single-quoted final argument; shlex undoes the shell quoting.
12
+ return json.loads(shlex.split(cmd)[-1])
13
+
14
+
15
+ def test_nersc_cmd(monkeypatch):
16
+ monkeypatch.delenv('SLURM_PARTITION', raising=False)
17
+ cmd = jamofetch.get_cmd('LBBDHFZ')
18
+ assert cmd.startswith('module load jamo; jamo link custom ')
19
+ assert _query(cmd) == {
20
+ "metadata.library_name": "LBBDHFZ",
21
+ "metadata.fastq_type": "sdm_normal",
22
+ "group": "sdm",
23
+ }
24
+
25
+
26
+ def test_dori_cmd_uses_exec(monkeypatch):
27
+ monkeypatch.setenv('SLURM_PARTITION', 'dori')
28
+ cmd = jamofetch.get_cmd('LBBDHFZ')
29
+ assert cmd.startswith('apptainer --silent exec docker://doejgi/jamo-dori jamo link -s dori custom ')
30
+ assert _query(cmd)["metadata.library_name"] == "LBBDHFZ"
31
+
32
+
33
+ def test_library_name_is_first_query_key(monkeypatch):
34
+ # jamo names the symlink after the first string-valued key; _find_fastq_path
35
+ # relies on that being the library name.
36
+ monkeypatch.delenv('SLURM_PARTITION', raising=False)
37
+ assert next(iter(_query(jamofetch.get_cmd('LBBDHFZ')))) == "metadata.library_name"
38
+
39
+
40
+ def test_query_does_not_filter_on_user(monkeypatch):
41
+ monkeypatch.delenv('SLURM_PARTITION', raising=False)
42
+ assert "user" not in _query(jamofetch.get_cmd('LBBDHFZ'))
43
+
44
+
45
+ @pytest.mark.parametrize('bad', ["LBB'DHFZ", 'lbbdhfz', 'LBB DHFZ', 'LBB;rm'])
46
+ def test_invalid_library_name_rejected(bad):
47
+ with pytest.raises(ValueError):
48
+ jamofetch.get_cmd(bad)
49
+
50
+
51
+ # --- disambiguating a library sequenced more than once -----------------------
52
+ #
53
+ # Real case this exists for, library LBBCZRJ: four records matching the plain
54
+ # query, across three pools -- cell 37397 (run 3207), cell 37892 (run 3232,
55
+ # registered twice with identical md5s) and cell 37964 (run 3230). Read counts
56
+ # 1,296,169 / 1,965,608 / 948,348. Before 3.7.9 the one analysed was whichever
57
+ # os.listdir() returned last.
58
+
59
+
60
+ def test_smrt_cell_id_narrows_query(monkeypatch):
61
+ monkeypatch.delenv('SLURM_PARTITION', raising=False)
62
+ query = _query(jamofetch.get_cmd('LBBCZRJ', smrt_cell_id=37892))
63
+ assert query == {
64
+ "metadata.library_name": "LBBCZRJ",
65
+ "metadata.fastq_type": "sdm_normal",
66
+ "group": "sdm",
67
+ "metadata.sdm_smrt_cell_id": 37892,
68
+ }
69
+
70
+
71
+ def test_physical_run_id_narrows_query(monkeypatch):
72
+ monkeypatch.delenv('SLURM_PARTITION', raising=False)
73
+ query = _query(jamofetch.get_cmd('LBBCZRJ', physical_run_id=3232))
74
+ assert query["metadata.pacbio_physical_run_id"] == 3232
75
+
76
+
77
+ def test_both_ids_may_be_given(monkeypatch):
78
+ monkeypatch.delenv('SLURM_PARTITION', raising=False)
79
+ query = _query(jamofetch.get_cmd('LBBCZRJ', smrt_cell_id=37892, physical_run_id=3232))
80
+ assert query["metadata.sdm_smrt_cell_id"] == 37892
81
+ assert query["metadata.pacbio_physical_run_id"] == 3232
82
+
83
+
84
+ def test_library_name_still_first_key_with_ids(monkeypatch):
85
+ # jamo names the symlink after the first string-valued key. The ids are
86
+ # ints so they cannot take that slot, but the ordering is load-bearing
87
+ # enough to assert directly.
88
+ monkeypatch.delenv('SLURM_PARTITION', raising=False)
89
+ query = _query(jamofetch.get_cmd('LBBCZRJ', smrt_cell_id=37892))
90
+ assert next(iter(query)) == "metadata.library_name"
91
+
92
+
93
+ def test_no_ids_produces_the_3_7_8_command(monkeypatch):
94
+ # Backward compatibility: callers that pass no id must get the byte-identical
95
+ # command 3.7.8 produced. synbioqc-pbj pins these strings in its own tests.
96
+ monkeypatch.delenv('SLURM_PARTITION', raising=False)
97
+ assert jamofetch.get_cmd('LBBDHFZ') == (
98
+ "module load jamo; jamo link custom "
99
+ '\'{"metadata.library_name":"LBBDHFZ","metadata.fastq_type":"sdm_normal","group":"sdm"}\''
100
+ )
101
+
102
+
103
+ @pytest.mark.parametrize('bad', ['37892; rm -rf /', 'abc', '', -1, 0, 3.5, True, [37892]])
104
+ def test_invalid_ids_rejected(bad):
105
+ with pytest.raises(ValueError):
106
+ jamofetch.get_cmd('LBBCZRJ', smrt_cell_id=bad)
107
+
108
+
109
+ def test_numeric_string_id_accepted(monkeypatch):
110
+ # Ids arrive from JSON APIs and argparse as either int or str.
111
+ monkeypatch.delenv('SLURM_PARTITION', raising=False)
112
+ assert _query(jamofetch.get_cmd('LBBCZRJ', smrt_cell_id='37892'))[
113
+ "metadata.sdm_smrt_cell_id"] == 37892
114
+
115
+
116
+ # --- _find_fastq_path no longer guesses --------------------------------------
117
+
118
+ LBBCZRJ_TARGETS = {
119
+ "pbio-3207.37397.bc2080_OA--bc2080_OA.hifi_reads.bc2080_OA.ccs.fastq.gz":
120
+ "/global/dna/dm_archive/sdm/pacbio/00/32/07",
121
+ "pbr6184.3232.37663.37892.bc2080_OA--bc2080_OA.pacbio.processed_well.37663.37892.fastq.gz":
122
+ "/global/dna/dm_archive/sdm/analyses-242/AUTO-2421306",
123
+ "pbr6186.3230.37667.37964.bc2080_OA--bc2080_OA.pacbio.processed_well.37667.37964.fastq.gz":
124
+ "/global/dna/dm_archive/sdm/analyses-242/AUTO-2421406",
125
+ }
126
+
127
+
128
+ def _link(seq_dir, lib, file_name, folder):
129
+ link = os.path.join(seq_dir, f"{lib}.{file_name}")
130
+ os.symlink(os.path.join(folder, file_name), link)
131
+ return link
132
+
133
+
134
+ def test_single_match_is_returned(tmp_path):
135
+ name, folder = next(iter(LBBCZRJ_TARGETS.items()))
136
+ _link(str(tmp_path), 'LBBCZRJ', name, folder)
137
+ assert jamofetch._find_fastq_path('LBBCZRJ', str(tmp_path)).endswith(name)
138
+
139
+
140
+ def test_ambiguous_match_raises_and_names_every_candidate(tmp_path):
141
+ for name, folder in LBBCZRJ_TARGETS.items():
142
+ _link(str(tmp_path), 'LBBCZRJ', name, folder)
143
+
144
+ with pytest.raises(RuntimeError) as excinfo:
145
+ jamofetch._find_fastq_path('LBBCZRJ', str(tmp_path))
146
+
147
+ message = str(excinfo.value)
148
+ assert "3 different" in message
149
+ for name, folder in LBBCZRJ_TARGETS.items():
150
+ assert os.path.join(folder, name) in message
151
+ assert "smrt_cell_id" in message
152
+
153
+
154
+ def test_duplicate_registrations_of_one_file_are_not_ambiguous(tmp_path):
155
+ # SDM registered the same file under two AUTO- folders with identical md5s.
156
+ # Same file_name means one link name, so only one link can exist -- but if
157
+ # two links ever point at the same target, that is not a real choice.
158
+ name = "pbr6184.3232.37663.37892.bc2080_OA--bc2080_OA.pacbio.processed_well.37663.37892.fastq.gz"
159
+ target = "/global/dna/dm_archive/sdm/analyses-242/AUTO-2421306/" + name
160
+ os.symlink(target, os.path.join(str(tmp_path), f"LBBCZRJ.{name}"))
161
+ os.symlink(target, os.path.join(str(tmp_path), f"LBBCZRJ.copy.{name}"))
162
+ assert jamofetch._find_fastq_path('LBBCZRJ', str(tmp_path)) is not None
163
+
164
+
165
+ def test_other_libraries_links_are_ignored(tmp_path):
166
+ name, folder = next(iter(LBBCZRJ_TARGETS.items()))
167
+ _link(str(tmp_path), 'LBBCZRJ', name, folder)
168
+ for other_name, other_folder in list(LBBCZRJ_TARGETS.items())[1:]:
169
+ _link(str(tmp_path), 'LBBDHFZ', other_name, other_folder)
170
+ assert jamofetch._find_fastq_path('LBBCZRJ', str(tmp_path)).endswith(name)
171
+
172
+
173
+ def test_no_match_raises(tmp_path):
174
+ with pytest.raises(RuntimeError, match="failed to find sequence file"):
175
+ jamofetch._find_fastq_path('LBBCZRJ', str(tmp_path))
176
+
177
+
178
+ def test_broken_link_is_still_a_candidate(tmp_path):
179
+ # On dori the target is copied into a local cache and the link is broken
180
+ # until that lands; it is still the file this library resolved to.
181
+ name, folder = next(iter(LBBCZRJ_TARGETS.items()))
182
+ link = _link(str(tmp_path), 'LBBCZRJ', name, folder)
183
+ assert not os.path.exists(link)
184
+ assert jamofetch._find_fastq_path('LBBCZRJ', str(tmp_path)) == link
@@ -1,7 +0,0 @@
1
- # Changelog
2
-
3
- <!--next-version-placeholder-->
4
-
5
- ## v0.1.0 (07/07/2023)
6
-
7
- - First release of `jamofetch`!
@@ -1,271 +0,0 @@
1
- #!/usr/bin/env python3
2
-
3
- import logging
4
- import os
5
- import pathlib
6
- import re
7
- import subprocess
8
- import sys
9
- import time
10
- from argparse import ArgumentParser
11
- from stat import ST_CTIME
12
-
13
- JAMO_CMD_NERSC = 'module load jamo; jamo link library'
14
- JAMO_CMD_DORI = 'apptainer --silent run docker://doejgi/jamo-dori jamo link -s dori library'
15
-
16
- TWO_HOURS = 7200 # seconds
17
- ONE_MINUTE = 60 # seconds
18
-
19
-
20
- def get_cmd(clean_lib_name) -> str:
21
- if os.getenv('SLURM_PARTITION') == 'dori':
22
- return f"{JAMO_CMD_DORI} {clean_lib_name}"
23
- return f"{JAMO_CMD_NERSC} {clean_lib_name}"
24
-
25
-
26
- def _clean_library_name(library_name):
27
- clean_name = f"{library_name}".strip()
28
- if not re.match(r'^[A-Z]+$', clean_name):
29
- raise ValueError(f"invalid library name {library_name}")
30
- return clean_name
31
-
32
-
33
- def _check_int_not_negative(value_to_check, value_name, allow_minus_1=False) -> int:
34
- """
35
- Throw an error if value_to_check is not a positive integer. Provide
36
- value_name in error message.
37
- """
38
- if not value_to_check:
39
- return 0
40
- if not isinstance(value_to_check, int):
41
- raise ValueError(f"invalid value for {value_name}: expecting integer")
42
- if allow_minus_1 and value_to_check == -1:
43
- return -1
44
- if value_to_check <= 0:
45
- raise ValueError(f"invalid value for {value_name}: expecting positive integer")
46
- return value_to_check
47
-
48
-
49
- def _wait_for_seq(seq_path, wait_interval_secs, wait_max_secs):
50
- """
51
- Check if seq_path is a link. If seq_path is a broken link, wait until it is valid.
52
- Check link each wait_interval_secs seconds. Exit when link is valid. Throw
53
- an error if wait_max_secs is exceeded. Set wait_max_secs to None to wait
54
- indefinitely.
55
- """
56
- if seq_path is None:
57
- raise ValueError("null seq_path")
58
-
59
- if not os.path.islink(seq_path):
60
- raise ValueError("seq_path is not a link")
61
-
62
- wait_interval = _check_int_not_negative(wait_interval_secs, 'wait_interval_secs')
63
- wait_max = _check_int_not_negative(wait_max_secs, 'wait_max_secs', allow_minus_1=True)
64
-
65
- total_wait = 0
66
- if wait_max and wait_interval:
67
- logging.debug(f"waiting for link {seq_path}; wait_interval: {wait_interval}; wait_max: {wait_max}")
68
- while (not os.path.exists(seq_path)) and (wait_max == -1 or total_wait < wait_max):
69
- time.sleep(wait_interval_secs)
70
- if wait_interval > 0:
71
- total_wait += wait_interval
72
-
73
- if not os.path.exists(seq_path):
74
- raise RuntimeError(f"sequence for link {seq_path} not provisioned within max wait seconds: {wait_max}")
75
-
76
-
77
- def _find_fastq_path(lib_name, seq_dir):
78
- """
79
- Find symlink created for library in sequence directory
80
- by matching pattern similar to:
81
- HHCHO.52485.1.359430.GTGCTTA-GTAAGCA.fastq.gz
82
- """
83
- # sort contents in directory by modification time (we want the last modified)
84
- # adapted from:
85
- # https://www.tutorialspoint.com/How-do-you-get-a-directory-listing-sorted-by-creation-date-in-Python
86
- file_paths = [os.path.join(seq_dir, file_name) for file_name in os.listdir(os.path.realpath(seq_dir))]
87
- logging.debug(f"file paths: {file_paths}")
88
- # get file stats
89
- path_stats = [(path, os.lstat(path)) for path in file_paths]
90
- path_stats.sort(key=lambda x: x[1][ST_CTIME])
91
- for path_stat in path_stats:
92
- logging.debug(f"{path_stat[0]}:")
93
- for stat in sorted(filter(lambda a: a.startswith('st_'), dir(path_stat[1]))):
94
- logging.debug(f" {stat}: {getattr(path_stat[1], stat)}")
95
-
96
- matched_link = None
97
- for path_stat in path_stats:
98
- filepath = path_stat[0]
99
- filetime = path_stat[1][ST_CTIME]
100
- filename = os.path.basename(filepath)
101
- if not os.path.islink(filepath):
102
- continue
103
- logging.debug(f"link candidate: {filename}; creation time: {filetime}")
104
- if str(filename).startswith(f"{lib_name}.") and str(filename).endswith('.fastq.gz'):
105
- matched_link = filepath
106
- logging.debug(f"link match: {filename}")
107
-
108
- if matched_link is not None:
109
- logging.debug(f"returning library seq file: {matched_link}")
110
- return matched_link
111
-
112
- raise RuntimeError(f"failed to find sequence file for library {lib_name}")
113
-
114
-
115
- class LibSeq:
116
-
117
- def __init__(self, lib_name, seq_path):
118
- self._lib_name = _clean_library_name(lib_name)
119
- self._seq_path = seq_path
120
-
121
- def get_lib_name(self) -> str:
122
- return self._lib_name
123
-
124
- def get_seq_path(self):
125
- return self._seq_path
126
-
127
- def seq_exists(self) -> bool:
128
- return os.path.exists(self._seq_path)
129
-
130
- def get_real_path(self):
131
- if self._seq_path is None:
132
- return None
133
- return os.path.realpath(self._seq_path)
134
-
135
- def get_real_path_wait(self):
136
- if self.seq_exists():
137
- return self.get_real_path()
138
- else:
139
- raise RuntimeError(f"{self}: sequence file not available")
140
-
141
- def __str__(self):
142
- f"{self.__class__.__name__}(lib={self._lib_name};path={self._seq_path})"
143
-
144
-
145
- class JamoLibSeq(LibSeq):
146
-
147
- def __init__(self, lib_name, seq_path, wait_interval_secs=10, wait_max_secs=-1):
148
- LibSeq.__init__(self, lib_name, seq_path)
149
- self._wait_interval_secs = _check_int_not_negative(wait_interval_secs, 'wait_interval_secs')
150
- self._wait_max_secs = _check_int_not_negative(wait_max_secs, 'wait_max_secs', allow_minus_1=True)
151
- self._sequence_ready = False
152
-
153
- def get_real_path_wait(self):
154
- if not self._sequence_ready:
155
- _wait_for_seq(self._seq_path, self._wait_interval_secs, self._wait_max_secs)
156
- self._sequence_ready = True
157
- return os.path.realpath(self._seq_path)
158
-
159
-
160
- class JamoFetcher():
161
- def __init__(self, link_dir='.', wait_interval_secs=10, wait_max_secs=-1):
162
- self._wait_interval_secs = _check_int_not_negative(wait_interval_secs, 'wait_interval_secs')
163
- self._wait_max_secs = _check_int_not_negative(wait_max_secs, 'wait_max_secs', allow_minus_1=True)
164
- self._link_dir = os.path.realpath(link_dir)
165
-
166
- def fetch_lib_seq(self, lib_name, out_file=sys.stderr) -> JamoLibSeq:
167
- """
168
- Execute JAMO command to link sequence for library
169
- and return a LibSeq object containing the path to the sequence.
170
- """
171
- pathlib.Path(self._link_dir).mkdir(parents=True, exist_ok=True, mode=0o775)
172
- os.chdir(self._link_dir)
173
- clean_lib_name = _clean_library_name(lib_name)
174
-
175
- cmd = get_cmd(clean_lib_name)
176
- print(f"\n{cmd}", file=out_file)
177
- output = subprocess.check_output(cmd, shell=True)
178
- for line in output.splitlines():
179
- print(line.decode("utf-8"), file=out_file)
180
- seq_path = _find_fastq_path(clean_lib_name, self._link_dir)
181
- print(f"{clean_lib_name} {seq_path}", file=out_file)
182
- return JamoLibSeq(clean_lib_name, seq_path, self._wait_interval_secs, self._wait_max_secs)
183
-
184
-
185
- def _fetch_seq(fetcher: JamoFetcher, libs):
186
- lib_seq_dict = {}
187
- for lib in libs:
188
- try:
189
- lib_seq: JamoLibSeq = fetcher.fetch_lib_seq(lib, out_file=sys.stderr)
190
- lib_seq_dict[lib_seq.get_lib_name()] = lib_seq
191
- except Exception as e:
192
- pass
193
- return lib_seq_dict
194
-
195
-
196
- def main(args):
197
- logging.debug(f"args: {args}")
198
-
199
- if not args.library:
200
- return
201
-
202
- wait_interval_secs = _check_int_not_negative(args.interval, "interval")
203
- wait_max_secs = _check_int_not_negative(args.max, "max", allow_minus_1=True)
204
-
205
- link_dir = os.path.realpath(args.directory if args.directory else '.')
206
- fetcher: JamoFetcher = JamoFetcher(link_dir=link_dir, wait_interval_secs=wait_interval_secs,
207
- wait_max_secs=wait_max_secs)
208
-
209
- print("fetching sequence:")
210
- lib_seq_dict = _fetch_seq(fetcher, args.library)
211
-
212
- if not lib_seq_dict:
213
- return # no sequence was fetched
214
-
215
- print("\nsequence links:")
216
- for lib_name, lib_seq in sorted(lib_seq_dict.items()):
217
- print(f"{lib_seq.get_lib_name()} symlink: {lib_seq.get_seq_path()}")
218
- print(f"{lib_seq.get_lib_name()} realpath: {lib_seq.get_real_path()}")
219
-
220
- if args.wait:
221
- total_wait = 0
222
- seq_ready = set()
223
- print("\nwaiting for JAMO to provision sequence . . . .")
224
- while (len(seq_ready) < len(lib_seq_dict)) and (wait_max_secs == -1 or total_wait <= wait_max_secs):
225
- for lib_name, lib_seq in sorted(lib_seq_dict.items()):
226
- if not lib_name in seq_ready and lib_seq.seq_exists():
227
- print(f"{lib_name} sequence ready")
228
- seq_ready.add(lib_name)
229
- if not wait_interval_secs or not wait_max_secs:
230
- break # do not wait
231
- if len(seq_ready) < len(lib_seq_dict):
232
- time.sleep(wait_interval_secs)
233
- if wait_max_secs > 0:
234
- total_wait += wait_interval_secs
235
- if len(seq_ready) < len(lib_seq_dict):
236
- print("\nexiting, not all sequence ready")
237
- if wait_max_secs != -1 and total_wait >= wait_max_secs:
238
- print(f"max wait {wait_max_secs} exceeded")
239
-
240
-
241
- def cli():
242
- try:
243
- parser: ArgumentParser = ArgumentParser()
244
- parser.add_argument('-l', '--library', action='append',
245
- help="library name(s) for which to retrieve sequence")
246
- parser.add_argument('-d', '--directory', required=False, default='.',
247
- help="directory where to link sequence, defaults to current directory. " +
248
- "Directory will be created if it doesn't exit.")
249
- parser.add_argument('-i', '--interval', required=False, type=int, default=10,
250
- help="wait interval in seconds to check if sequence has been fetched, " +
251
- "ignored if wait flag not set")
252
- parser.add_argument('-m', '--max', required=False, type=int, default=TWO_HOURS,
253
- help="maximum time to wait for sequence in seconds, " +
254
- "ignored if wait flag not set. Specify -1 to wait indefinetely.")
255
- parser.add_argument('-w', '--wait', action='store_true',
256
- help='wait for jamo to link sequence, then print "sequence ready"')
257
- parser.add_argument('--logging', required=False, default='WARN',
258
- help="logging level (specify DEBUG for verbose logging)")
259
-
260
- ARGS = parser.parse_args()
261
- # logging.basicConfig(format='%(asctime)s %(message)s', datefmt='%m/%d/%Y %I:%M:%S %p', level=log_level)
262
- logging.basicConfig(format='%(message)s', level=ARGS.logging.upper())
263
- main(ARGS)
264
- except KeyboardInterrupt:
265
- print('Interrupted', file=sys.stderr)
266
- # https://unix.stackexchange.com/questions/251996/why-does-bash-set-exit-status-to-non-zero-on-ctrl-c-or-ctrl-z
267
- sys.exit(130)
268
-
269
-
270
- if __name__ == "__main__":
271
- cli()
@@ -1,2 +0,0 @@
1
- from jamofetch import jamofetch
2
-
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes