pysz 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- pysz-0.1.0/LICENSE +21 -0
- pysz-0.1.0/PKG-INFO +13 -0
- pysz-0.1.0/pyproject.toml +50 -0
- pysz-0.1.0/setup.py +37 -0
- pysz-0.1.0/src/pysz/.pytest_cache/.gitignore +2 -0
- pysz-0.1.0/src/pysz/.pytest_cache/CACHEDIR.TAG +4 -0
- pysz-0.1.0/src/pysz/.pytest_cache/README.md +8 -0
- pysz-0.1.0/src/pysz/.pytest_cache/v/cache/lastfailed +3 -0
- pysz-0.1.0/src/pysz/.pytest_cache/v/cache/nodeids +3 -0
- pysz-0.1.0/src/pysz/.pytest_cache/v/cache/stepwise +1 -0
- pysz-0.1.0/src/pysz/__init__.py +1 -0
- pysz-0.1.0/src/pysz/api.py +325 -0
- pysz-0.1.0/src/pysz/compression.py +108 -0
- pysz-0.1.0/src/pysz/utils.py +35 -0
pysz-0.1.0/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2023 Jinyang Zhang
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
pysz-0.1.0/PKG-INFO
ADDED
|
@@ -0,0 +1,13 @@
|
|
|
1
|
+
Metadata-Version: 2.1
|
|
2
|
+
Name: pysz
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Package for reading and writing svb-zstd compressed data
|
|
5
|
+
Author: Jinyang Zhang
|
|
6
|
+
Author-email: zhangjinyang@biols.ac.cn
|
|
7
|
+
Requires-Python: >=3.10,<4.0
|
|
8
|
+
Classifier: Programming Language :: Python :: 3
|
|
9
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
10
|
+
Requires-Dist: numpy (>=1.24.2,<2.0.0)
|
|
11
|
+
Requires-Dist: pandas (>=1.5.3,<2.0.0)
|
|
12
|
+
Requires-Dist: pystreamvbyte
|
|
13
|
+
Requires-Dist: zstandard (==0.19.0)
|
|
@@ -0,0 +1,50 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
build-backend = "setuptools.build_meta"
|
|
3
|
+
requires = [
|
|
4
|
+
"setuptools",
|
|
5
|
+
"wheel",
|
|
6
|
+
]
|
|
7
|
+
|
|
8
|
+
[project]
|
|
9
|
+
name = "pysz"
|
|
10
|
+
version = "0.1.0"
|
|
11
|
+
authors = [
|
|
12
|
+
{name = "Jinyang Zhang", email = "zhangjinyang@biols.ac.cn"},
|
|
13
|
+
]
|
|
14
|
+
description = "Package for reading and writing svb-zstd compressed data"
|
|
15
|
+
readme = {file = "README.md", content-type = "text/markdown"}
|
|
16
|
+
requires-python = "==3.10"
|
|
17
|
+
keywords = ["svb", "zstd"]
|
|
18
|
+
license = {file = "LICENSE"}
|
|
19
|
+
classifiers = [
|
|
20
|
+
"Development Status :: 4 - Beta",
|
|
21
|
+
"License :: OSI Approved :: MIT License",
|
|
22
|
+
"Programming Language :: Python :: 3"
|
|
23
|
+
]
|
|
24
|
+
|
|
25
|
+
dependencies = [
|
|
26
|
+
"numpy==1.24.2",
|
|
27
|
+
"pandas==1.5.3",
|
|
28
|
+
"zstandard==0.19.0",
|
|
29
|
+
"pystreamvbyte @ git+https://github.com/Kevinzjy/pystreamvbyte@50eaf184b734fd7d64d18fd9b0847fb210e05b75#egg=pystreamvbyte",
|
|
30
|
+
]
|
|
31
|
+
|
|
32
|
+
[tool.poetry]
|
|
33
|
+
name = "pysz"
|
|
34
|
+
version = "0.1.0"
|
|
35
|
+
description = "Package for reading and writing svb-zstd compressed data"
|
|
36
|
+
authors = ["Jinyang Zhang <zhangjinyang@biols.ac.cn>"]
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
[tool.poetry.dependencies]
|
|
40
|
+
python = "^3.10"
|
|
41
|
+
pandas = "^1.5.3"
|
|
42
|
+
numpy = "^1.24.2"
|
|
43
|
+
zstandard = "0.19.0"
|
|
44
|
+
pystreamvbyte = "git+https://github.com/Kevinzjy/pystreamvbyte@50eaf184b734fd7d64d18fd9b0847fb210e05b75#egg=pystreamvbyte"
|
|
45
|
+
|
|
46
|
+
[tool.poetry.dev-dependencies]
|
|
47
|
+
pytest = "^7.2.2"
|
|
48
|
+
|
|
49
|
+
#[tool.poetry.plugins."scripts"]
|
|
50
|
+
#example_etl = "pysz.api:check_szmain"
|
pysz-0.1.0/setup.py
ADDED
|
@@ -0,0 +1,37 @@
|
|
|
1
|
+
# -*- coding: utf-8 -*-
|
|
2
|
+
from setuptools import setup
|
|
3
|
+
|
|
4
|
+
package_dir = \
|
|
5
|
+
{'': 'src'}
|
|
6
|
+
|
|
7
|
+
packages = \
|
|
8
|
+
['pysz']
|
|
9
|
+
|
|
10
|
+
package_data = \
|
|
11
|
+
{'': ['*'], 'pysz': ['.pytest_cache/*', '.pytest_cache/v/cache/*']}
|
|
12
|
+
|
|
13
|
+
install_requires = \
|
|
14
|
+
['numpy>=1.24.2,<2.0.0',
|
|
15
|
+
'pandas>=1.5.3,<2.0.0',
|
|
16
|
+
'pystreamvbyte',
|
|
17
|
+
'zstandard==0.19.0']
|
|
18
|
+
|
|
19
|
+
setup_kwargs = {
|
|
20
|
+
'name': 'pysz',
|
|
21
|
+
'version': '0.1.0',
|
|
22
|
+
'description': 'Package for reading and writing svb-zstd compressed data',
|
|
23
|
+
'long_description': None,
|
|
24
|
+
'author': 'Jinyang Zhang',
|
|
25
|
+
'author_email': 'zhangjinyang@biols.ac.cn',
|
|
26
|
+
'maintainer': None,
|
|
27
|
+
'maintainer_email': None,
|
|
28
|
+
'url': None,
|
|
29
|
+
'package_dir': package_dir,
|
|
30
|
+
'packages': packages,
|
|
31
|
+
'package_data': package_data,
|
|
32
|
+
'install_requires': install_requires,
|
|
33
|
+
'python_requires': '>=3.10,<4.0',
|
|
34
|
+
}
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
setup(**setup_kwargs)
|
|
@@ -0,0 +1,8 @@
|
|
|
1
|
+
# pytest cache directory #
|
|
2
|
+
|
|
3
|
+
This directory contains data from the pytest's cache plugin,
|
|
4
|
+
which provides the `--lf` and `--ff` options, as well as the `cache` fixture.
|
|
5
|
+
|
|
6
|
+
**Do not** commit this to version control.
|
|
7
|
+
|
|
8
|
+
See [the docs](https://docs.pytest.org/en/stable/how-to/cache.html) for more information.
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
[]
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
__version__ = "0.1.0"
|
|
@@ -0,0 +1,325 @@
|
|
|
1
|
+
import json
|
|
2
|
+
import numpy as np
|
|
3
|
+
import pandas as pd
|
|
4
|
+
from pathlib import Path
|
|
5
|
+
from collections import OrderedDict, namedtuple
|
|
6
|
+
from zstandard import ZstdDecompressor
|
|
7
|
+
|
|
8
|
+
from pysz import __version__
|
|
9
|
+
from pysz.compression import svb_decode, svb_encode, zstd_decode, zstd_encode, str_decode, str_encode
|
|
10
|
+
from pysz.utils import mkdir, assert_file_exists, assert_dir_exists
|
|
11
|
+
|
|
12
|
+
from multiprocessing import Manager, Process
|
|
13
|
+
|
|
14
|
+
# Default attributes & datasets for chunk data
|
|
15
|
+
Attributes = [
|
|
16
|
+
('ID', str),
|
|
17
|
+
('Offset', np.int32),
|
|
18
|
+
('Raw_unit', np.uint32),
|
|
19
|
+
]
|
|
20
|
+
Datasets = [
|
|
21
|
+
('Raw', np.uint32),
|
|
22
|
+
('Fastq', str),
|
|
23
|
+
('Move', np.uint16),
|
|
24
|
+
]
|
|
25
|
+
|
|
26
|
+
# For dtype conversion
|
|
27
|
+
Dtypes_names = {
|
|
28
|
+
str: "S",
|
|
29
|
+
np.int16: "I16", np.int32: "I32",
|
|
30
|
+
np.uint16: "U16", np.uint32: "U32",
|
|
31
|
+
np.float16: "F16", np.float32: "F32",
|
|
32
|
+
}
|
|
33
|
+
Dtypes = {j: i for i, j in Dtypes_names.items()}
|
|
34
|
+
|
|
35
|
+
# Dtypes that can be svb compressed
|
|
36
|
+
Svb_dtypes = [np.int16, np.int32, np.uint16, np.uint32]
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
class CompressedFile(object):
|
|
40
|
+
"""
|
|
41
|
+
Class for parsing SVB-ZSTD compressed data
|
|
42
|
+
"""
|
|
43
|
+
|
|
44
|
+
def __init__(self, data_dir, mode:str="r", header:list=None, attributes:list=None, datasets:list=None, overwrite:bool=False, n_threads=8, allow_multiprocessing=True):
|
|
45
|
+
"""
|
|
46
|
+
Init function of CompressedFile Class
|
|
47
|
+
|
|
48
|
+
Args:
|
|
49
|
+
data_dir: str / Path,
|
|
50
|
+
path to SZ directory, required
|
|
51
|
+
mode: str,
|
|
52
|
+
"r" for read and "w" for writing sz data
|
|
53
|
+
header: [(key, value)]:
|
|
54
|
+
list of of keys and values in header, defaults to [(version, '0.0.1')]
|
|
55
|
+
attributes: [(attr_id, attr_dtype)],
|
|
56
|
+
list of ID and dtype for attributes, defaults to [('ID', str), ('Offset', np.int32), ('Raw_unit', np.uint32)]
|
|
57
|
+
datasets: [(dataset_id, dataset_dtype)],
|
|
58
|
+
list of ID and dtype for datasets, defaults to [('Raw', np.uint32), ('Fastq', np.uint32), ('Move', np.uint32)]
|
|
59
|
+
overwrite: bool,
|
|
60
|
+
whether overwrite existed sz data, defaults to False
|
|
61
|
+
n_threads: int,
|
|
62
|
+
number of threads for parallelized data compression, defaults to 8
|
|
63
|
+
allow_multiprocessing: bool,
|
|
64
|
+
set to True if want to access SZ data using multiprocessing, defaults to False
|
|
65
|
+
"""
|
|
66
|
+
self.dir = data_dir if isinstance(data_dir, Path) else Path(data_dir)
|
|
67
|
+
self.idx_path = self.dir / "index"
|
|
68
|
+
self.dat_path = self.dir / "dat"
|
|
69
|
+
self.mode = mode
|
|
70
|
+
|
|
71
|
+
# For Record class
|
|
72
|
+
self.idx, self.header, self.attr, self.datasets = None, None, None, None
|
|
73
|
+
|
|
74
|
+
# For multiprocessing
|
|
75
|
+
self.threads = n_threads
|
|
76
|
+
self.q_in = Manager().Queue()
|
|
77
|
+
self.q_out = Manager().Queue()
|
|
78
|
+
self.pool = []
|
|
79
|
+
self.worker, self.decompressor, self.fh = None, None, None
|
|
80
|
+
|
|
81
|
+
# Get attributes and datasets
|
|
82
|
+
if self.mode == "r":
|
|
83
|
+
assert_dir_exists(data_dir)
|
|
84
|
+
assert_file_exists(self.idx_path)
|
|
85
|
+
assert_file_exists(self.dat_path)
|
|
86
|
+
self.load_header()
|
|
87
|
+
elif self.mode == 'w':
|
|
88
|
+
mkdir(data_dir, overwrite=overwrite)
|
|
89
|
+
self.save_header(header, attributes, datasets)
|
|
90
|
+
else:
|
|
91
|
+
raise KeyError("Mode should be either 'w' or 'r'.")
|
|
92
|
+
|
|
93
|
+
# Init Record class
|
|
94
|
+
keys = []
|
|
95
|
+
if self.attr is not None:
|
|
96
|
+
keys += list(self.attr)
|
|
97
|
+
if self.datasets is not None:
|
|
98
|
+
keys += list(self.datasets)
|
|
99
|
+
self.record = namedtuple("Record", keys)
|
|
100
|
+
|
|
101
|
+
# Init multiprocessing related variables
|
|
102
|
+
if self.mode == 'w':
|
|
103
|
+
self.worker = Process(target=self.writer)
|
|
104
|
+
self.worker.start()
|
|
105
|
+
for _ in np.arange(self.threads):
|
|
106
|
+
p = Process(target=self.encoder)
|
|
107
|
+
p.start()
|
|
108
|
+
self.pool.append(p)
|
|
109
|
+
else:
|
|
110
|
+
if allow_multiprocessing:
|
|
111
|
+
self.decompressor = ZstdDecompressor()
|
|
112
|
+
self.fh = open(self.dat_path, 'rb')
|
|
113
|
+
|
|
114
|
+
def save_header(self, header, attributes, datasets):
|
|
115
|
+
"""
|
|
116
|
+
Save header, attributes and dataset information to the index file
|
|
117
|
+
|
|
118
|
+
Args:
|
|
119
|
+
header: [(key, value)]:
|
|
120
|
+
list of of keys and values in header, defaults to [(version, '0.0.1')]
|
|
121
|
+
attributes: [(attr_id, attr_dtype)],
|
|
122
|
+
list of ID and dtype for attributes, defaults to [('ID', str), ('Offset', np.int32), ('Raw_unit', np.uint32)]
|
|
123
|
+
datasets: [(dataset_id, dataset_dtype)],
|
|
124
|
+
list of ID and dtype for datasets, defaults to [('Raw', np.uint32), ('Fastq', np.uint32), ('Move', np.uint32)]
|
|
125
|
+
"""
|
|
126
|
+
self.header = {"version": __version__} if header is None else dict(header)
|
|
127
|
+
self.attr = OrderedDict(Attributes) if attributes is None else OrderedDict(attributes)
|
|
128
|
+
self.datasets = OrderedDict(Datasets) if datasets is None else OrderedDict(datasets)
|
|
129
|
+
|
|
130
|
+
with open(self.idx_path, 'w') as out:
|
|
131
|
+
header_str = json.dumps(self.header)
|
|
132
|
+
out.write("#" + header_str + '\n')
|
|
133
|
+
|
|
134
|
+
cols = ["LENGTH", "OFFSET",] + \
|
|
135
|
+
[f"{i}:A:{Dtypes_names[j]}" for i,j in self.attr.items()] + \
|
|
136
|
+
[f"{i}:D:{Dtypes_names[j]}" for i,j in self.datasets.items()]
|
|
137
|
+
out.write('\t'.join(cols) + '\n')
|
|
138
|
+
|
|
139
|
+
def load_header(self):
|
|
140
|
+
"""
|
|
141
|
+
Load header, attributes and datasets information for SZ file
|
|
142
|
+
"""
|
|
143
|
+
attributes = []
|
|
144
|
+
datasets = []
|
|
145
|
+
index = []
|
|
146
|
+
with open(self.idx_path, 'r') as f:
|
|
147
|
+
# Get first row of header
|
|
148
|
+
header = json.loads(f.readline().rstrip().lstrip('#'))
|
|
149
|
+
|
|
150
|
+
# Get attributes and datasets
|
|
151
|
+
cols = f.readline().rstrip().split('\t')
|
|
152
|
+
for col in cols[2:]:
|
|
153
|
+
key_id, key_group, key_dtype = col.split(':')
|
|
154
|
+
if key_group == 'A':
|
|
155
|
+
attributes.append((key_id, Dtypes[key_dtype]))
|
|
156
|
+
elif key_group == "D":
|
|
157
|
+
datasets.append((key_id, Dtypes[key_dtype]))
|
|
158
|
+
else:
|
|
159
|
+
raise KeyError(f"Unsupported data type: {key_group}")
|
|
160
|
+
|
|
161
|
+
# Get index dataframe
|
|
162
|
+
for line in f:
|
|
163
|
+
index.append(line.rstrip().split('\t'))
|
|
164
|
+
|
|
165
|
+
self.header = header
|
|
166
|
+
self.attr = OrderedDict(attributes)
|
|
167
|
+
self.datasets = OrderedDict(datasets)
|
|
168
|
+
|
|
169
|
+
# Convert into pandas dataframe and parser dtype
|
|
170
|
+
self.idx = pd.DataFrame(index)
|
|
171
|
+
self.idx.columns = ['LENGTH', 'OFFSET'] + list(self.attr) + list(self.datasets)
|
|
172
|
+
self.idx['LENGTH'] = self.idx['LENGTH'].astype(int)
|
|
173
|
+
self.idx['OFFSET'] = self.idx['OFFSET'].astype(int)
|
|
174
|
+
|
|
175
|
+
def put(self, *data):
|
|
176
|
+
"""
|
|
177
|
+
Put new data into queue for compressing and saving
|
|
178
|
+
|
|
179
|
+
Args:
|
|
180
|
+
*data: data must be in specific order consistent to the attributes and datasets.
|
|
181
|
+
The default SZ contains the following attributes:
|
|
182
|
+
|
|
183
|
+
ID: read ID, str
|
|
184
|
+
Offset: Offset for raw current data, np.int32
|
|
185
|
+
Raw_unit: Raw unit for raw current data, np.float32
|
|
186
|
+
|
|
187
|
+
And the following datasets:
|
|
188
|
+
|
|
189
|
+
Raw: Raw current data points, np.uint32
|
|
190
|
+
Fastq: Basecalled fastq, str
|
|
191
|
+
Move: move tables, np.uint16
|
|
192
|
+
"""
|
|
193
|
+
self.q_in.put(data)
|
|
194
|
+
|
|
195
|
+
def encode(self, data):
|
|
196
|
+
"""
|
|
197
|
+
Encode input data into SVB-ZSTD compressed binary format
|
|
198
|
+
Args:
|
|
199
|
+
data: data acquired from queue
|
|
200
|
+
|
|
201
|
+
Returns:
|
|
202
|
+
encoded, byte str,
|
|
203
|
+
byte str of compressed datasets
|
|
204
|
+
idx, list,
|
|
205
|
+
list of items to be stored in index
|
|
206
|
+
"""
|
|
207
|
+
record = self.record(*data)
|
|
208
|
+
dat_encoded = []
|
|
209
|
+
idx_encoded = []
|
|
210
|
+
|
|
211
|
+
# Store indices
|
|
212
|
+
for attr_id, attr_dtype in self.attr.items():
|
|
213
|
+
idx_encoded.append(str(attr_dtype(getattr(record, attr_id))))
|
|
214
|
+
|
|
215
|
+
# Compress datasets according to the dtypes
|
|
216
|
+
for dataset_id, dataset_dtype in self.datasets.items():
|
|
217
|
+
if dataset_dtype in Svb_dtypes:
|
|
218
|
+
d_bstr, d_size, _ = svb_encode(np.array(getattr(record, dataset_id)).astype(dataset_dtype))
|
|
219
|
+
elif dataset_dtype == str:
|
|
220
|
+
d_bstr = str_encode(dataset_dtype(getattr(record, dataset_id)))
|
|
221
|
+
d_size = ''
|
|
222
|
+
else:
|
|
223
|
+
d_bstr = np.array(getattr(record, dataset_id)).astype(dataset_dtype).tobytes()
|
|
224
|
+
d_size = ''
|
|
225
|
+
|
|
226
|
+
d_encoded = zstd_encode(d_bstr)
|
|
227
|
+
dat_encoded.append(d_encoded)
|
|
228
|
+
idx_encoded.append(f"{len(d_encoded)}:{d_size}")
|
|
229
|
+
|
|
230
|
+
# Convert encoded into simple binary format
|
|
231
|
+
encoded = b''.join(dat_encoded)
|
|
232
|
+
|
|
233
|
+
return encoded, idx_encoded
|
|
234
|
+
|
|
235
|
+
def encoder(self):
|
|
236
|
+
"""
|
|
237
|
+
Process for compressing input data and pass to SZ writer
|
|
238
|
+
"""
|
|
239
|
+
while True:
|
|
240
|
+
raw = self.q_in.get()
|
|
241
|
+
if raw is None:
|
|
242
|
+
break
|
|
243
|
+
dat_line, idx_line = self.encode(raw)
|
|
244
|
+
self.q_out.put((dat_line, idx_line))
|
|
245
|
+
self.q_out.put(None)
|
|
246
|
+
|
|
247
|
+
def writer(self):
|
|
248
|
+
"""
|
|
249
|
+
Process for handling compressed data and save it to disk
|
|
250
|
+
"""
|
|
251
|
+
n_alive = self.threads
|
|
252
|
+
cursor = 0
|
|
253
|
+
with open(self.idx_path, 'a') as idx_out, open(self.dat_path, 'wb') as dat_out:
|
|
254
|
+
while True:
|
|
255
|
+
# Wait for all encode workers to stop
|
|
256
|
+
if n_alive == 0:
|
|
257
|
+
break
|
|
258
|
+
res = self.q_out.get()
|
|
259
|
+
if res is None:
|
|
260
|
+
n_alive -= 1
|
|
261
|
+
continue
|
|
262
|
+
|
|
263
|
+
# Record cursor position everytime
|
|
264
|
+
dat_line, idx_line = res
|
|
265
|
+
dat_out.write(dat_line)
|
|
266
|
+
shift, length = cursor, len(dat_line)
|
|
267
|
+
cursor += length
|
|
268
|
+
idx_line = [str(length), str(shift), ] + idx_line
|
|
269
|
+
idx_out.write('\t'.join(idx_line) + '\n')
|
|
270
|
+
|
|
271
|
+
def get(self, idx):
|
|
272
|
+
"""
|
|
273
|
+
Get specified index from compressed SZ data
|
|
274
|
+
|
|
275
|
+
Args:
|
|
276
|
+
idx: int / list
|
|
277
|
+
numeric int or list of integer index, items will be acquired using 'df.iloc';
|
|
278
|
+
|
|
279
|
+
Returns:
|
|
280
|
+
list of Record (namedTuple), containing attributes and datasets information specified in the SZ header
|
|
281
|
+
|
|
282
|
+
"""
|
|
283
|
+
reads = []
|
|
284
|
+
rows = self.idx.iloc[[idx]] if isinstance(idx, int) else self.idx.iloc[idx]
|
|
285
|
+
rows = rows.sort_values(by='OFFSET')
|
|
286
|
+
|
|
287
|
+
zdc = ZstdDecompressor() if self.decompressor is None else self.decompressor
|
|
288
|
+
f = open(self.dat_path, 'rb') if self.fh is None else self.fh
|
|
289
|
+
|
|
290
|
+
for _, row in rows.iterrows():
|
|
291
|
+
offset, length = row['OFFSET'], row['LENGTH']
|
|
292
|
+
attr = row[2:2+len(self.attr)].tolist()
|
|
293
|
+
|
|
294
|
+
f.seek(offset)
|
|
295
|
+
encoded = f.read(length)
|
|
296
|
+
|
|
297
|
+
datasets = []
|
|
298
|
+
p = 0
|
|
299
|
+
for d_name, d_col in zip(self.datasets, row[2+len(self.attr):]):
|
|
300
|
+
d_size, d_shape = d_col.split(':')
|
|
301
|
+
bstr = zstd_decode(encoded[p:p+int(d_size)], zdc)
|
|
302
|
+
p += int(d_size)
|
|
303
|
+
|
|
304
|
+
d_dtype = self.datasets[d_name]
|
|
305
|
+
if d_dtype in Svb_dtypes:
|
|
306
|
+
d_decoded = svb_decode(bstr, int(d_shape), dtype=d_dtype)
|
|
307
|
+
elif d_dtype == str:
|
|
308
|
+
d_decoded = str_decode(bstr)
|
|
309
|
+
else:
|
|
310
|
+
d_decoded = np.frombuffer(bstr, dtype=d_dtype)
|
|
311
|
+
datasets.append(d_decoded)
|
|
312
|
+
record = attr + datasets
|
|
313
|
+
reads.append(self.record(*record))
|
|
314
|
+
return reads
|
|
315
|
+
|
|
316
|
+
def close(self):
|
|
317
|
+
"""
|
|
318
|
+
Functions for closing and wanting for process to end.
|
|
319
|
+
"""
|
|
320
|
+
if self.mode == 'w':
|
|
321
|
+
for _ in np.arange(self.threads):
|
|
322
|
+
self.q_in.put(None)
|
|
323
|
+
for p in self.pool:
|
|
324
|
+
p.join()
|
|
325
|
+
self.worker.join()
|
|
@@ -0,0 +1,108 @@
|
|
|
1
|
+
import numpy as np
|
|
2
|
+
import streamvbyte
|
|
3
|
+
from zstandard import ZstdDecompressor, ZstdCompressor
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
def str_to_bytes(bytes_or_str):
|
|
7
|
+
"""
|
|
8
|
+
Convert str to byte string
|
|
9
|
+
:param bytes_or_str:
|
|
10
|
+
:return:
|
|
11
|
+
"""
|
|
12
|
+
if isinstance(bytes_or_str, str):
|
|
13
|
+
value = bytes_or_str.encode('utf-8')
|
|
14
|
+
else:
|
|
15
|
+
value = bytes_or_str
|
|
16
|
+
return value
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def bytes_to_str(bytes_or_str):
|
|
20
|
+
"""
|
|
21
|
+
Convert byte string to string
|
|
22
|
+
:param bytes_or_str:
|
|
23
|
+
:return:
|
|
24
|
+
"""
|
|
25
|
+
if isinstance(bytes_or_str, bytes):
|
|
26
|
+
value = bytes_or_str.decode('utf-8')
|
|
27
|
+
else:
|
|
28
|
+
value = bytes_or_str
|
|
29
|
+
return value
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def str_encode(data):
|
|
33
|
+
"""
|
|
34
|
+
Convert str to byte
|
|
35
|
+
:param data:
|
|
36
|
+
:return:
|
|
37
|
+
"""
|
|
38
|
+
return str_to_bytes(data)
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def str_decode(data):
|
|
42
|
+
"""
|
|
43
|
+
Convert byte to str
|
|
44
|
+
:param data:
|
|
45
|
+
:return:
|
|
46
|
+
"""
|
|
47
|
+
return bytes_to_str(data)
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def is_gzipped(file_name):
|
|
51
|
+
"""
|
|
52
|
+
Examine whether a file is gzipped.
|
|
53
|
+
Reference: https://stackoverflow.com/questions/3703276/how-to-tell-if-a-file-is-gzip-compressed
|
|
54
|
+
"""
|
|
55
|
+
with open(file_name, 'rb') as f:
|
|
56
|
+
return f.read(2) == b'\x1f\x8b'
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def zstd_encode(bstr, cctx=None):
|
|
60
|
+
"""
|
|
61
|
+
Compress byte string using Zstd library
|
|
62
|
+
:param data: byte str
|
|
63
|
+
data for compression
|
|
64
|
+
:param cctx: zstandard.ZstdCompressor
|
|
65
|
+
instance of zstd compressor
|
|
66
|
+
:return: byte str
|
|
67
|
+
compressed data
|
|
68
|
+
"""
|
|
69
|
+
_cctx = ZstdCompressor() if cctx is None else cctx
|
|
70
|
+
compressed = _cctx.compress(bstr)
|
|
71
|
+
return compressed
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def zstd_decode(compressed, zdc=None):
|
|
75
|
+
"""
|
|
76
|
+
Decompress zstd compressed data
|
|
77
|
+
:param compressed:
|
|
78
|
+
:param zdc:
|
|
79
|
+
:return:
|
|
80
|
+
"""
|
|
81
|
+
_zdc = ZstdDecompressor() if zdc is None else zdc
|
|
82
|
+
decompressed = _zdc.decompress(compressed)
|
|
83
|
+
return decompressed
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
def svb_encode(data):
|
|
87
|
+
"""
|
|
88
|
+
Compress data using streamvbyte algorithm, int16, uint16, int32, uint32 is supported
|
|
89
|
+
:param data:
|
|
90
|
+
:return:
|
|
91
|
+
"""
|
|
92
|
+
if data.dtype not in [np.uint16, np.int16, np.int32, np.uint32]:
|
|
93
|
+
raise TypeError("Streamvbyte only support int16/uint16/int32/uint32 formats!")
|
|
94
|
+
compressed = streamvbyte.encode(data).tobytes()
|
|
95
|
+
return compressed, data.shape[0], data.dtype
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
def svb_decode(data, size, dtype=np.int32):
|
|
99
|
+
"""
|
|
100
|
+
Decompress streamvbyte encoded data
|
|
101
|
+
:param data:
|
|
102
|
+
:param size:
|
|
103
|
+
:param dtype:
|
|
104
|
+
:return:
|
|
105
|
+
"""
|
|
106
|
+
decoded = np.frombuffer(data, dtype=np.uint8)
|
|
107
|
+
decompressed = streamvbyte.decode(decoded, size, dtype=dtype)
|
|
108
|
+
return decompressed
|
|
@@ -0,0 +1,35 @@
|
|
|
1
|
+
import sys
|
|
2
|
+
import warnings
|
|
3
|
+
from pathlib import Path
|
|
4
|
+
|
|
5
|
+
def mkdir(dir_name, overwrite=False):
|
|
6
|
+
dir_path = dir_name if isinstance(dir_name, Path) else Path(dir_name)
|
|
7
|
+
if dir_path.exists():
|
|
8
|
+
if dir_path.is_dir():
|
|
9
|
+
if overwrite is True:
|
|
10
|
+
sys.stderr.write(f'Overwriting existed {dir_name}\n')
|
|
11
|
+
else:
|
|
12
|
+
raise OSError(f'{dir_name} already exists, please use --overwrite to force overwrite')
|
|
13
|
+
else:
|
|
14
|
+
raise OSError(f'{dir_name} conflict with existing files!')
|
|
15
|
+
else:
|
|
16
|
+
dir_path.mkdir()
|
|
17
|
+
return dir_path
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def assert_dir_exists(dir_name):
|
|
21
|
+
dir_path = dir_name if isinstance(dir_name, Path) else Path(dir_name)
|
|
22
|
+
if dir_path.exists() and dir_path.is_dir():
|
|
23
|
+
return dir_path
|
|
24
|
+
raise OSError(f"Directory not found: {dir_path}.")
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def assert_file_exists(file_name):
|
|
28
|
+
file_path = file_name if isinstance(file_name, Path) else Path(file_name)
|
|
29
|
+
if file_path.exists() and file_path.is_file():
|
|
30
|
+
return file_path
|
|
31
|
+
raise OSError(f"File not found: {file_path}.")
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def error_callback(e):
|
|
35
|
+
sys.stderr.write(f"Error in worker process: {e.__cause__}: {e}")
|