immlib 1.0.0.dev2__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- immlib/__init__.py +131 -0
- immlib/_init.py +108 -0
- immlib/_version.py +235 -0
- immlib/doc/__init__.py +38 -0
- immlib/doc/_core.py +311 -0
- immlib/iolib/__init__.py +29 -0
- immlib/iolib/_core.py +720 -0
- immlib/pathlib/__init__.py +69 -0
- immlib/pathlib/_cache.py +152 -0
- immlib/pathlib/_core.py +869 -0
- immlib/pathlib/_osf.py +538 -0
- immlib/test/__init__.py +16 -0
- immlib/test/__main__.py +10 -0
- immlib/test/doc/__init__.py +6 -0
- immlib/test/doc/test_core.py +91 -0
- immlib/test/iolib/__init__.py +7 -0
- immlib/test/iolib/test_core.py +81 -0
- immlib/test/pathlib/__init__.py +11 -0
- immlib/test/pathlib/test_core.py +146 -0
- immlib/test/pathlib/test_osf.py +54 -0
- immlib/test/types/__init__.py +5 -0
- immlib/test/types/test_core.py +110 -0
- immlib/test/util/__init__.py +11 -0
- immlib/test/util/test_core.py +681 -0
- immlib/test/util/test_numeric.py +1374 -0
- immlib/test/util/test_quantity.py +218 -0
- immlib/test/util/test_url.py +51 -0
- immlib/test/workflow/__init__.py +9 -0
- immlib/test/workflow/test_core.py +418 -0
- immlib/test/workflow/test_plantype.py +248 -0
- immlib/types/__init__.py +29 -0
- immlib/types/_core.py +333 -0
- immlib/util/__init__.py +283 -0
- immlib/util/_core.py +2524 -0
- immlib/util/_numeric.py +2651 -0
- immlib/util/_quantity.py +523 -0
- immlib/util/_url.py +114 -0
- immlib/workflow/__init__.py +48 -0
- immlib/workflow/_core.py +1635 -0
- immlib/workflow/_plantype.py +334 -0
- immlib-1.0.0.dev2.dist-info/METADATA +76 -0
- immlib-1.0.0.dev2.dist-info/RECORD +45 -0
- immlib-1.0.0.dev2.dist-info/WHEEL +5 -0
- immlib-1.0.0.dev2.dist-info/licenses/LICENSE +21 -0
- immlib-1.0.0.dev2.dist-info/top_level.txt +1 -0
immlib/iolib/_core.py
ADDED
|
@@ -0,0 +1,720 @@
|
|
|
1
|
+
# -*- coding: utf-8 -*-
|
|
2
|
+
################################################################################
|
|
3
|
+
# immlib/iolib/_core.py
|
|
4
|
+
|
|
5
|
+
"""Input and output operations via the load() and save() functions."""
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
# Dependencies #################################################################
|
|
9
|
+
|
|
10
|
+
import os, io, gzip, numbers
|
|
11
|
+
from pathlib import Path
|
|
12
|
+
|
|
13
|
+
import numpy as np
|
|
14
|
+
from pcollections import pdict, plist, pset, ldict, lazy
|
|
15
|
+
|
|
16
|
+
from ..doc import docwrap
|
|
17
|
+
from ..util import is_str, is_amap, is_aseq
|
|
18
|
+
from ..pathlib import path, is_path, like_path
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
# Format and Formatter #########################################################
|
|
22
|
+
|
|
23
|
+
# The Format class, an inset class for managing the various stream formats.
|
|
24
|
+
class Format:
|
|
25
|
+
"""Manages the details of a stream format for use with `save` and `load`.
|
|
26
|
+
|
|
27
|
+
`Format` objects are typically created with the `save.register` or
|
|
28
|
+
`load.register` methods. This class should be handled internally but should
|
|
29
|
+
not be instantiated outside of the `Save` or `Load` classes.
|
|
30
|
+
|
|
31
|
+
`Format` objects can be treated as save functions: the `__call__` method
|
|
32
|
+
on a `Format` object accepts a stream or path-like object representing
|
|
33
|
+
the destination to which the format should be written, an object to save
|
|
34
|
+
in the format, and any optional arguments understood by the type.
|
|
35
|
+
"""
|
|
36
|
+
def __init__(self, name, function, *suffixes, mode='b', gzip_suffix=None):
|
|
37
|
+
self.name = name
|
|
38
|
+
self.function = function
|
|
39
|
+
self.suffixes = []
|
|
40
|
+
for suff in suffixes:
|
|
41
|
+
if is_str(suff):
|
|
42
|
+
suff = (suff,)
|
|
43
|
+
elif all(map(is_str, suff)):
|
|
44
|
+
suff = tuple(suff)
|
|
45
|
+
else:
|
|
46
|
+
raise ValueError(f"invalid path suffix: {suff}")
|
|
47
|
+
self.suffixes.append(suff)
|
|
48
|
+
if mode != 'b' and mode != 't':
|
|
49
|
+
raise ValueError("format mode must be 't' or 'b'")
|
|
50
|
+
self.mode = mode
|
|
51
|
+
if gzip_suffix is None:
|
|
52
|
+
gzip_suffix = ()
|
|
53
|
+
elif is_str(gzip_suffix):
|
|
54
|
+
gzip_suffix = ((gzip_suffix,),)
|
|
55
|
+
elif is_aseq(gzip_suffix):
|
|
56
|
+
if all(map(is_str, gzip_suffix)):
|
|
57
|
+
gzip_suffix = tuple((s,) for s in gzip_suffix)
|
|
58
|
+
else:
|
|
59
|
+
g = []
|
|
60
|
+
for suff in gzip_suffix:
|
|
61
|
+
if is_str(suff):
|
|
62
|
+
g.append((suff,))
|
|
63
|
+
elif is_aseq(suff) and all(map(is_str, suff)):
|
|
64
|
+
g.append(tuple(suff))
|
|
65
|
+
else:
|
|
66
|
+
raise ValueError(f"invalid gzip suffix: {suff}")
|
|
67
|
+
gzip_suffix = tuple(g)
|
|
68
|
+
else:
|
|
69
|
+
raise ValueError(f"invalid gzip suffix: {gzip_suffix}")
|
|
70
|
+
self.gzip_suffix = gzip_suffix
|
|
71
|
+
# We also want to change the documentation.
|
|
72
|
+
self.__doc__ = self.function.__doc__
|
|
73
|
+
def __call__(self, stream, stream_mode, *args, **kwargs):
|
|
74
|
+
# If dest isn't a stream, we need to open it as a path.
|
|
75
|
+
if isinstance(stream, io.IOBase):
|
|
76
|
+
return self.function(stream, *args, **kwargs)
|
|
77
|
+
else:
|
|
78
|
+
p = path(stream)
|
|
79
|
+
with p.open(stream_mode + self.mode) as s:
|
|
80
|
+
return self.function(s, *args, **kwargs)
|
|
81
|
+
class Formatter:
|
|
82
|
+
"""Base class for the Save and Load classes.
|
|
83
|
+
|
|
84
|
+
Handles common operations between saving and loading systems.
|
|
85
|
+
"""
|
|
86
|
+
# The mode use for opening streams from paths ('r' or 'w' typically).
|
|
87
|
+
stream_mode = ''
|
|
88
|
+
# Data/Details of the Save/Load classes.
|
|
89
|
+
__slots__ = ("formats", "_format_by_suffix")
|
|
90
|
+
# Constructor.
|
|
91
|
+
def __init__(self, template=None):
|
|
92
|
+
self.formats = {}
|
|
93
|
+
self._format_by_suffix = {}
|
|
94
|
+
if template is not None:
|
|
95
|
+
cls = type(self)
|
|
96
|
+
if isinstance(template, cls):
|
|
97
|
+
formats = template.formats
|
|
98
|
+
elif is_amap(template):
|
|
99
|
+
formats = templates
|
|
100
|
+
else:
|
|
101
|
+
raise TypeError(f"template must be a {cls} object or a mapping")
|
|
102
|
+
for (k,format) in formats.items():
|
|
103
|
+
if not isinstance(format, Format):
|
|
104
|
+
raise TypeError(f"template contains non-format named {k}")
|
|
105
|
+
self.register(format)
|
|
106
|
+
def deduce_format(self, arg, ignore_gz=True):
|
|
107
|
+
"""Deduces the file format for a given path or suffix.
|
|
108
|
+
|
|
109
|
+
`save.deduce_format(arg)` deduces the format implied by `arg`. The
|
|
110
|
+
argument may be a path, a string representing a path, or a sequence of
|
|
111
|
+
strings representing the suffixes of the path. If the final suffix is
|
|
112
|
+
`.gz` then this suffix is ignored.
|
|
113
|
+
"""
|
|
114
|
+
if is_str(arg):
|
|
115
|
+
suffixes = path(arg).suffixes
|
|
116
|
+
elif is_path(arg):
|
|
117
|
+
suffixes = arg.suffixes
|
|
118
|
+
elif is_aseq(arg) and all(map(is_str, arg)):
|
|
119
|
+
suffixes = arg
|
|
120
|
+
else:
|
|
121
|
+
raise ValueError(
|
|
122
|
+
f"deduce_format given invalid argument of type {type(arg)}")
|
|
123
|
+
suff = tuple(suffixes)
|
|
124
|
+
if ignore_gz and suff[-1] == '.gz':
|
|
125
|
+
suff = suff[:-1]
|
|
126
|
+
while suff:
|
|
127
|
+
format = self._format_by_suffix.get(suff)
|
|
128
|
+
if format:
|
|
129
|
+
return format
|
|
130
|
+
suff = suff[1:]
|
|
131
|
+
# No format found.
|
|
132
|
+
return None
|
|
133
|
+
@staticmethod
|
|
134
|
+
def _expandall(p):
|
|
135
|
+
p = path(p)
|
|
136
|
+
if isinstance(p, Path):
|
|
137
|
+
p = Path(os.path.expanduser(os.path.expandvars(os.fspath(p))))
|
|
138
|
+
return p
|
|
139
|
+
def _call(self, target, format, gzip, /, *args, **kwargs):
|
|
140
|
+
if isinstance(target, io.IOBase):
|
|
141
|
+
target_path = None
|
|
142
|
+
target_stream = target
|
|
143
|
+
if gzip is Ellipsis:
|
|
144
|
+
gzip = False
|
|
145
|
+
else:
|
|
146
|
+
target_path = path(target)
|
|
147
|
+
if isinstance(target_path, Path):
|
|
148
|
+
target_path = Formatter._expandall(target_path)
|
|
149
|
+
target_stream = None
|
|
150
|
+
if format is None:
|
|
151
|
+
if target_path is None:
|
|
152
|
+
raise ValueError("format can't be None when target is a stream")
|
|
153
|
+
# We can walk through the formats and attempt to deduce the correct
|
|
154
|
+
# format from the ending of the dest_str.
|
|
155
|
+
format = self.deduce_format(target_path)
|
|
156
|
+
if not format:
|
|
157
|
+
raise ValueError(f"format can't be deduced from path: {target}")
|
|
158
|
+
# At this point we have a format name or we've raised an error.
|
|
159
|
+
if is_str(format):
|
|
160
|
+
fmt = self.formats.get(format)
|
|
161
|
+
if fmt is None:
|
|
162
|
+
raise ValueError(f"format not recognized: {format}")
|
|
163
|
+
else:
|
|
164
|
+
format = fmt
|
|
165
|
+
elif not isinstance(format, Format):
|
|
166
|
+
raise TypeError("format arg must be a format name or Format object")
|
|
167
|
+
# Now we know the format and thus have a format function; if we were
|
|
168
|
+
# given a path instead of a stream, we need to open the path for the
|
|
169
|
+
# formatting function. We also need to handle gzipping--if the format is
|
|
170
|
+
# a gzip format then gzip will be True at this point.
|
|
171
|
+
if gzip is Ellipsis:
|
|
172
|
+
suff = tuple(target_path.suffixes)
|
|
173
|
+
gzip = (target_path.suffix == '.gz') or suff in format.gzip_suffix
|
|
174
|
+
if gzip is True:
|
|
175
|
+
if target_stream:
|
|
176
|
+
with self._gzip(target_stream, format.mode) as act_stream:
|
|
177
|
+
r = format.function(act_stream, *args, **kwargs)
|
|
178
|
+
return (r, target_stream)
|
|
179
|
+
else:
|
|
180
|
+
with target_path.open(self.stream_mode + 'b') as stream:
|
|
181
|
+
with self._gzip(stream, format.mode) as act_stream:
|
|
182
|
+
r = format.function(act_stream, *args, **kwargs)
|
|
183
|
+
return (r, target_path)
|
|
184
|
+
elif target_stream:
|
|
185
|
+
r = format.function(target_stream, *args, **kwargs)
|
|
186
|
+
return (r, target_stream)
|
|
187
|
+
else:
|
|
188
|
+
with target_path.open(self.stream_mode + format.mode) as stream:
|
|
189
|
+
r = format.function(stream, *args, **kwargs)
|
|
190
|
+
return (r, target_path)
|
|
191
|
+
def _gzip(self, stream, mode,
|
|
192
|
+
compresslevel=9,
|
|
193
|
+
encoding=None,
|
|
194
|
+
errors=None,
|
|
195
|
+
newline=None):
|
|
196
|
+
return gzip.open(
|
|
197
|
+
stream,
|
|
198
|
+
mode=(self.stream_mode + mode),
|
|
199
|
+
compresslevel=compresslevel,
|
|
200
|
+
encoding=encoding,
|
|
201
|
+
errors=errors,
|
|
202
|
+
newline=newline)
|
|
203
|
+
def register(self, name, /, *suffixes, mode='b', gzip_suffix=None):
|
|
204
|
+
"""Registers a format type with the `save`/`load` interface.
|
|
205
|
+
|
|
206
|
+
The `immlib.save.register` method is intended to be used as a decorator:
|
|
207
|
+
`@immlib.save.register(format_name, suffix1, suffix2...)` ensures that
|
|
208
|
+
the function that follows it is registered as a format managed by the
|
|
209
|
+
`immlib.save` system. The function that is decorated should be written
|
|
210
|
+
to accept a stream object (though the decorator will ensure that it
|
|
211
|
+
works when called with a path-like object or a stream). The function's
|
|
212
|
+
signature must always match `f(output_stream, object_to_save, **opts)`
|
|
213
|
+
where the `**opts` must be any set of named parameters. The return value
|
|
214
|
+
of the function is ignored, but it should raise an error if the object
|
|
215
|
+
cannot be saved.
|
|
216
|
+
|
|
217
|
+
The `immlib.load.register` method is identical except that the function
|
|
218
|
+
that follows is not passed an object to load and instead must return the
|
|
219
|
+
object that is loaded from the given stream.
|
|
220
|
+
|
|
221
|
+
Parameters
|
|
222
|
+
----------
|
|
223
|
+
name : str
|
|
224
|
+
The name of the format.
|
|
225
|
+
*suffixes : strings or tuples of strings, optional
|
|
226
|
+
Any number of suffixes that this format uses. Suffixes must be
|
|
227
|
+
unique, and an error is raised if a requested suffix is already
|
|
228
|
+
registered to another format. Each suffix may be a string such as
|
|
229
|
+
`.tgz` or a tuple such as `('.tar', '.gz')`.
|
|
230
|
+
mode : 't' or 'b', optional
|
|
231
|
+
Whether any stream that is opened in order to save the file should
|
|
232
|
+
be opened in text mode (`'t'`) or binary mode (`'b'`). The default
|
|
233
|
+
is `'b'`.
|
|
234
|
+
gzip_suffix : None or str or tuple of str
|
|
235
|
+
A sequence of suffixes that indicate that the format is additionally
|
|
236
|
+
encoded using gzip. Files that end with `.gz` are automatically
|
|
237
|
+
interpreted as gzipped files by the `save` system, but if an
|
|
238
|
+
additional file ending needs to be specified as a gzip format, such
|
|
239
|
+
as the suffix `.npz` for gzipped-numpy files (equivalent to
|
|
240
|
+
`.npy.gz`), then it should be specified in this option. If a single
|
|
241
|
+
string is given (such as `'.npz'`), then it is interpreted as a
|
|
242
|
+
single suffix (`[['.npz']]`); if a sequence of strings are given
|
|
243
|
+
(`['.npz', '.nz']`), then they are interpreted as individual
|
|
244
|
+
suffixes (`[['.npz'], ['.nz']]`). For compound suffixes, they must
|
|
245
|
+
be elements of a sequence, e.g. `[['.json', '.gz']]` not `['.json',
|
|
246
|
+
'.gz']`. Note, however, that `[['.json', '.gz']]` is automatically
|
|
247
|
+
interpreted as a gzip file by virtue of its `.gz` suffix.
|
|
248
|
+
"""
|
|
249
|
+
# There are actually two uses: (1) that described in the help above and
|
|
250
|
+
# (2) save.register(format) where format is already a Format object. In
|
|
251
|
+
# the latter case we aren't a decorator.
|
|
252
|
+
if isinstance(name, Format):
|
|
253
|
+
# We're in case 2, so we register this format specifically.
|
|
254
|
+
format = name
|
|
255
|
+
if format.name in self.formats:
|
|
256
|
+
raise RuntimeError(f"format {format.name} already registered")
|
|
257
|
+
suffs = tuple(format.suffixes) + format.gzip_suffix
|
|
258
|
+
for suff in suffs:
|
|
259
|
+
ex = self._format_by_suffix.get(suff)
|
|
260
|
+
if ex is not None:
|
|
261
|
+
raise RuntimeError(
|
|
262
|
+
f"suffix {suff} already mapped to format {ex.name}")
|
|
263
|
+
# We pass the criteria; go ahead and add this format.
|
|
264
|
+
self.formats[format.name] = format
|
|
265
|
+
for suff in suffs:
|
|
266
|
+
self._format_by_suffix[suff] = format
|
|
267
|
+
# Return the format itself.
|
|
268
|
+
return format
|
|
269
|
+
else:
|
|
270
|
+
# We're in case 1, so we return a decorator for the function.
|
|
271
|
+
def _formatter_register_dec(f):
|
|
272
|
+
if isinstance(f, Format):
|
|
273
|
+
f = f.function
|
|
274
|
+
format = Format(
|
|
275
|
+
name, f, *suffixes,
|
|
276
|
+
mode=mode,
|
|
277
|
+
gzip_suffix=gzip_suffix)
|
|
278
|
+
return self.register(format)
|
|
279
|
+
return _formatter_register_dec
|
|
280
|
+
# That's it for the register function.
|
|
281
|
+
def unregister(self, name, *, error_on_missing=False):
|
|
282
|
+
"""Unregisters the format with the given name from the save/load system.
|
|
283
|
+
|
|
284
|
+
The format with the given name is unregistered from the `immlib.save`
|
|
285
|
+
system, and the `Format` object that is removed is returned. If the
|
|
286
|
+
format is not found, then `None` is returned, but no error is raised by
|
|
287
|
+
default.
|
|
288
|
+
|
|
289
|
+
The `immlib.load.unregister` method works identically to the
|
|
290
|
+
`immlib.save` version.
|
|
291
|
+
|
|
292
|
+
Parameters
|
|
293
|
+
----------
|
|
294
|
+
name : str
|
|
295
|
+
The name of the format to unregister.
|
|
296
|
+
error_on_missing : boolean, optional
|
|
297
|
+
Whether to throw a `RuntimeError` if the named format is not found
|
|
298
|
+
in the save manager. By default this is `False`.
|
|
299
|
+
|
|
300
|
+
Returns
|
|
301
|
+
-------
|
|
302
|
+
Format or None
|
|
303
|
+
The format object that is unregistered or `None` if the given name
|
|
304
|
+
was not a registered format.
|
|
305
|
+
|
|
306
|
+
Raises
|
|
307
|
+
------
|
|
308
|
+
RuntimeError
|
|
309
|
+
If the given name does not map to a format in the save manager and
|
|
310
|
+
the `error_on_missing` option is `True`.
|
|
311
|
+
"""
|
|
312
|
+
format = self.formats.get(name)
|
|
313
|
+
if format is None:
|
|
314
|
+
if error_on_missing:
|
|
315
|
+
fnnm = type(self).__name__.lower()
|
|
316
|
+
raise RuntimeError(
|
|
317
|
+
f"format {name} not found in immlib {fnnm} system")
|
|
318
|
+
else:
|
|
319
|
+
return None
|
|
320
|
+
# Actually remove things:
|
|
321
|
+
for suff in (format.suffixes + format.gzip_suffix):
|
|
322
|
+
del self._format_by_suffix[suff]
|
|
323
|
+
del self.formats[name]
|
|
324
|
+
return format
|
|
325
|
+
def copy(self):
|
|
326
|
+
"""Returns a copy of the given save manager.
|
|
327
|
+
|
|
328
|
+
`immlib.save.copy()` can be used to return a copy of the save manager,
|
|
329
|
+
for instances where a single save manager is not ideal.
|
|
330
|
+
"""
|
|
331
|
+
cls = type(self)
|
|
332
|
+
return cls(self)
|
|
333
|
+
|
|
334
|
+
|
|
335
|
+
# Save #########################################################################
|
|
336
|
+
|
|
337
|
+
class Save(Formatter):
|
|
338
|
+
"""Saves a Python object to a stream or path then returns the stream/path.
|
|
339
|
+
|
|
340
|
+
`immlib.save(path, object, format)` saves the given `object` to the given
|
|
341
|
+
`path` using the named file `format` and returns the path on success or
|
|
342
|
+
rasies an error on failure. If the format can be deduced from the path
|
|
343
|
+
suffix, then it may be omitted.
|
|
344
|
+
|
|
345
|
+
`immlib.save(stream, object, format)` writes the given `object` to the given
|
|
346
|
+
stream object using the named `format` and returns the stream.
|
|
347
|
+
|
|
348
|
+
In fact, `immlib.save` is an object of the `immlib.pathlib.Save` type that
|
|
349
|
+
primarily behaves like a function. The `immlib.save.register` and
|
|
350
|
+
`immlib.save.unregister` methods can be used to add understood formats.
|
|
351
|
+
|
|
352
|
+
Parameters
|
|
353
|
+
----------
|
|
354
|
+
dest : path-like or stream
|
|
355
|
+
The output destination to which the object is to be saved. This may be
|
|
356
|
+
either a path-like object or a writeable `IOBase` stream.
|
|
357
|
+
obj : object
|
|
358
|
+
Any object that can be saved in the given format.
|
|
359
|
+
format : str or None, optional
|
|
360
|
+
If provided, `format` must be a string that names a format that has been
|
|
361
|
+
previously registered with `save` using the `save.register` method. If
|
|
362
|
+
`format` is not provided or is `None` (the default), then an attempt is
|
|
363
|
+
made to deduce the format using the suffix of the `dest` argument,
|
|
364
|
+
assuming that `dest` is a path and not a stream. If `dest` is a stream
|
|
365
|
+
or if the suffix does not indicate a specific format, then a
|
|
366
|
+
`ValueError` is raised.
|
|
367
|
+
**kwargs
|
|
368
|
+
Any additional parameters are passed to the registered export function.
|
|
369
|
+
|
|
370
|
+
Returns
|
|
371
|
+
-------
|
|
372
|
+
path-like or stream
|
|
373
|
+
The destination stream or path. If the `dest` argument is a path name,
|
|
374
|
+
then a path object is returned instead of the path name.
|
|
375
|
+
|
|
376
|
+
Raises
|
|
377
|
+
------
|
|
378
|
+
TypeError
|
|
379
|
+
If the destination is not a stream or path-like object or if the format
|
|
380
|
+
is not a string.
|
|
381
|
+
ValueError
|
|
382
|
+
If the format is not recognized or if it cannot be deduced from the
|
|
383
|
+
destination object.
|
|
384
|
+
"""
|
|
385
|
+
stream_mode = 'w'
|
|
386
|
+
def __call__(self, dest, obj, format=None, /, gzip=Ellipsis, **kwargs):
|
|
387
|
+
(loadret, saveret) = self._call(dest, format, gzip, obj, **kwargs)
|
|
388
|
+
return saveret
|
|
389
|
+
|
|
390
|
+
# We can go ahead and declare a single global save object for the immlib
|
|
391
|
+
# library. We will later use this object to register various basic format
|
|
392
|
+
# types.
|
|
393
|
+
save = Save()
|
|
394
|
+
|
|
395
|
+
# Register a few basic file formats.
|
|
396
|
+
@save.register('str', mode='t')
|
|
397
|
+
def save_str(stream, obj, append_nl=False):
|
|
398
|
+
"""Saves `str(obj)` to the given stream."""
|
|
399
|
+
stream.write(str(obj))
|
|
400
|
+
if append_nl:
|
|
401
|
+
stream.write('\n')
|
|
402
|
+
@save.register('bytes', mode='b')
|
|
403
|
+
def save_bytes(stream, obj):
|
|
404
|
+
"""Saves `bytes(obj)` to the given stream."""
|
|
405
|
+
stream.write(bytes(obj))
|
|
406
|
+
@save.register('repr', mode='t')
|
|
407
|
+
def save_repr(stream, obj, append_nl=False):
|
|
408
|
+
"""Saves `repr(obj)` to the given stream."""
|
|
409
|
+
stream.write(repr(obj))
|
|
410
|
+
if append_nl:
|
|
411
|
+
stream.write('\n')
|
|
412
|
+
@save.register('text', '.txt', '.text', mode='t')
|
|
413
|
+
def save_text(stream, lines, append_nls=False):
|
|
414
|
+
"""Saves a blob of text or a sequence of lines to a stream or path.
|
|
415
|
+
|
|
416
|
+
Parameters
|
|
417
|
+
----------
|
|
418
|
+
stream : stream or path-like
|
|
419
|
+
The stream or path to which to write the lines.
|
|
420
|
+
lines : string or sequence of strings
|
|
421
|
+
The text that should be written to the file or stream.
|
|
422
|
+
append_nls : boolean, optional
|
|
423
|
+
Whether to append newlines to the end of each line. If `False` (the
|
|
424
|
+
default), then the lines are written as they are; otherwise, a newline
|
|
425
|
+
character is appended after each line in the `lines` argument.
|
|
426
|
+
"""
|
|
427
|
+
if is_str(lines):
|
|
428
|
+
lines = (lines,)
|
|
429
|
+
if not is_aseq(lines):
|
|
430
|
+
raise TypeError("lines must be text or a sequence of text objects")
|
|
431
|
+
if append_nls:
|
|
432
|
+
for ln in lines:
|
|
433
|
+
stream.write(ln)
|
|
434
|
+
stream.write('\n')
|
|
435
|
+
else:
|
|
436
|
+
for ln in lines:
|
|
437
|
+
stream.write(ln)
|
|
438
|
+
@save.register('pickle', '.pickle', '.pkl', '.pcl', mode='b')
|
|
439
|
+
def save_pickle(stream, obj, protocol=None, **kwargs):
|
|
440
|
+
"""Saves a pickled object to a destination path or stream.
|
|
441
|
+
|
|
442
|
+
All keyword options are forwarded to the `pickle.dump` function. The
|
|
443
|
+
`protocol` keyword option is provided as the `protocol` positional argument
|
|
444
|
+
to the `pickle.dump` function.
|
|
445
|
+
"""
|
|
446
|
+
import pickle
|
|
447
|
+
pickle.dump(obj, stream, protocol, **kwargs)
|
|
448
|
+
@save.register('numpy', '.npy', '.np', '.numpy', mode='b', gzip_suffix='.npz')
|
|
449
|
+
def save_numpy(stream, obj, **kwargs):
|
|
450
|
+
"""Saves a numpy object to a destination path or stream.
|
|
451
|
+
|
|
452
|
+
All keyword options are forwarded to the `numpy.save` function.
|
|
453
|
+
"""
|
|
454
|
+
import numpy as np
|
|
455
|
+
np.save(stream, obj, **kwargs)
|
|
456
|
+
def json_default(obj):
|
|
457
|
+
"""Converts an object to a json-formattable object or raises TypeError.
|
|
458
|
+
"""
|
|
459
|
+
if is_str(obj) or obj is None or obj is True or obj is False:
|
|
460
|
+
return obj
|
|
461
|
+
elif isinstance(obj, numbers.Integral):
|
|
462
|
+
return int(obj)
|
|
463
|
+
elif isinstance(obj, numbers.Real):
|
|
464
|
+
return float(obj)
|
|
465
|
+
elif is_amap(obj):
|
|
466
|
+
return dict(obj)
|
|
467
|
+
elif is_aseq(obj):
|
|
468
|
+
return list(obj)
|
|
469
|
+
elif isinstance(obj, np.ndarray):
|
|
470
|
+
return obj.tolist()
|
|
471
|
+
#elif is_planobject(obj):
|
|
472
|
+
# cls = type(obj)
|
|
473
|
+
# return {
|
|
474
|
+
# '__plantype__': f'{cls.__module__}.{cls.__name__}',
|
|
475
|
+
# '__params__': dict(obj.__plandict__.params)}
|
|
476
|
+
else:
|
|
477
|
+
raise TypeError(type(obj))
|
|
478
|
+
@save.register('json', '.json', mode='t')
|
|
479
|
+
def save_json(stream, obj, /, default=json_default, **kwargs):
|
|
480
|
+
"""Saves an object as a JSON string or raises a TypeError if not possible.
|
|
481
|
+
|
|
482
|
+
All keywords are passed along to the `json.dump` function. The `default`
|
|
483
|
+
option uses the `immlib.iolib.json_default` function, which is different
|
|
484
|
+
than the default used by `json.dump`, but all other options are unaltered.
|
|
485
|
+
"""
|
|
486
|
+
import json
|
|
487
|
+
json.dump(obj, stream, default=default, **kwargs)
|
|
488
|
+
def yaml_prepare(obj):
|
|
489
|
+
"""Returns a version of the argument that can be JSON/YAML serialized."""
|
|
490
|
+
if is_str(obj) or obj is None or obj is True or obj is False:
|
|
491
|
+
return obj
|
|
492
|
+
elif isinstance(obj, numbers.Integral):
|
|
493
|
+
return int(obj)
|
|
494
|
+
elif isinstance(obj, numbers.Real):
|
|
495
|
+
return float(obj)
|
|
496
|
+
elif isinstance(obj, np.ndarray):
|
|
497
|
+
return obj.tolist()
|
|
498
|
+
elif is_amap(obj):
|
|
499
|
+
r = {}
|
|
500
|
+
for (k,v) in obj.items():
|
|
501
|
+
if not is_str(k):
|
|
502
|
+
raise TypeError("JSON/YAML dict keys must be strings")
|
|
503
|
+
r[k] = yaml_prepare(v)
|
|
504
|
+
return r
|
|
505
|
+
elif is_aseq(obj):
|
|
506
|
+
return [yaml_prepare(u) for u in obj]
|
|
507
|
+
else:
|
|
508
|
+
raise TypeError(type(obj))
|
|
509
|
+
@save.register('yaml', '.yaml', mode='t')
|
|
510
|
+
def save_yaml(stream, obj, /, **kwargs):
|
|
511
|
+
"""Saves an object as a YAML string or raises a TypeError if not possible.
|
|
512
|
+
|
|
513
|
+
All keywords are passed along to the `yaml.Dumper` object that is used for
|
|
514
|
+
dumping `obj`. Note that unlike with the `yaml.dump` function, object are
|
|
515
|
+
only serialized if they are JSON/YAML serializable; classes that are not
|
|
516
|
+
mappings, sequences, numbers, strings, or booleans cannot be serialized
|
|
517
|
+
using this function and instead raise TypeErrors.
|
|
518
|
+
"""
|
|
519
|
+
import yaml
|
|
520
|
+
yaml.dump(yaml_prepare(obj), stream, **kwargs)
|
|
521
|
+
@save.register('csv', '.csv', mode='t')
|
|
522
|
+
def save_csv(stream, obj, /, index=False, **kwargs):
|
|
523
|
+
"""Saves a pandas DataFrame to a CSV file.
|
|
524
|
+
|
|
525
|
+
All options are passed along to `pandas.DataFrame.to_csv()`. The option
|
|
526
|
+
`index` has the default value of `False` in this function.
|
|
527
|
+
"""
|
|
528
|
+
import pandas
|
|
529
|
+
obj = pandas.DataFrame(obj)
|
|
530
|
+
obj.to_csv(stream, index=index, **kwargs)
|
|
531
|
+
@save.register('tsv', '.tsv', mode='t')
|
|
532
|
+
def save_tsv(stream, obj, /, sep="\t", index=False, **kwargs):
|
|
533
|
+
"""Saves a pandas DataFrame to a TSV file.
|
|
534
|
+
|
|
535
|
+
All options are passed along to `pandas.DataFrame.to_csv()`. The option
|
|
536
|
+
`index` has the default value of `False` in this function, and the option
|
|
537
|
+
`sep` has the default value `"\t"`.
|
|
538
|
+
"""
|
|
539
|
+
import pandas
|
|
540
|
+
obj = pandas.DataFrame(obj)
|
|
541
|
+
obj.to_csv(stream, sep=sep, index=index, **kwargs)
|
|
542
|
+
|
|
543
|
+
|
|
544
|
+
# Load #########################################################################
|
|
545
|
+
|
|
546
|
+
class Load(Formatter):
|
|
547
|
+
"""Loads a Python object from a stream or path then returns the object.
|
|
548
|
+
|
|
549
|
+
`immlib.load(path, format)` loads a Python object from the given `path`
|
|
550
|
+
using the named file `format` and returns the loaded object or rasies an
|
|
551
|
+
error on failure. If the format can be deduced from the path suffix, then it
|
|
552
|
+
may be omitted.
|
|
553
|
+
|
|
554
|
+
`immlib.load(stream, format)` reads a Python object from the given stream
|
|
555
|
+
object using the named `format` and returns the object. The format argument
|
|
556
|
+
cannot be omitted when the first argument is a stream.
|
|
557
|
+
|
|
558
|
+
In fact, `immlib.load` is an object of the `immlib.pathlib.Load` type that
|
|
559
|
+
primarily behaves like a function. The `immlib.load.register` and
|
|
560
|
+
`immlib.load.unregister` methods can be used to add understood formats.
|
|
561
|
+
|
|
562
|
+
Parameters
|
|
563
|
+
----------
|
|
564
|
+
source : path-like or stream
|
|
565
|
+
The input source from which the object is to be loaded. This may be
|
|
566
|
+
either a path-like object or a readable `IOBase` stream.
|
|
567
|
+
format : str or None, optional
|
|
568
|
+
If provided, `format` must be a string that names a format that has been
|
|
569
|
+
previously registered with `load` using the `load.register` method. If
|
|
570
|
+
`format` is not provided or is `None` (the default), then an attempt is
|
|
571
|
+
made to deduce the format using the suffix of the `source` argument,
|
|
572
|
+
assuming that `source` is a path and not a stream. If `source` is a
|
|
573
|
+
stream or if the suffix does not indicate a specific format, then a
|
|
574
|
+
`ValueError` is raised.
|
|
575
|
+
**kwargs
|
|
576
|
+
Any additional parameters are passed to the registered export function.
|
|
577
|
+
|
|
578
|
+
Returns
|
|
579
|
+
-------
|
|
580
|
+
object
|
|
581
|
+
The object that was loaded from the given stream or path.
|
|
582
|
+
|
|
583
|
+
Raises
|
|
584
|
+
------
|
|
585
|
+
TypeError
|
|
586
|
+
If the input source is not a stream or path-like object or if the format
|
|
587
|
+
is not a string.
|
|
588
|
+
ValueError
|
|
589
|
+
If the format is not recognized or if it cannot be deduced from the
|
|
590
|
+
source.
|
|
591
|
+
"""
|
|
592
|
+
stream_mode = 'r'
|
|
593
|
+
def __call__(self, src, format=None, /, gzip=Ellipsis, **kwargs):
|
|
594
|
+
# There's a special case for loading: if we're given a path, and it
|
|
595
|
+
# refers to a directory, we load it as a lazy dictionary.
|
|
596
|
+
if format is None:
|
|
597
|
+
if not isinstance(src, io.IOBase):
|
|
598
|
+
p = Formatter._expandall(src)
|
|
599
|
+
if p.is_dir():
|
|
600
|
+
format = 'dir'
|
|
601
|
+
if format == 'dir':
|
|
602
|
+
return Load.from_dir(src)
|
|
603
|
+
(loadret, saveret) = self._call(src, format, gzip, **kwargs)
|
|
604
|
+
return loadret
|
|
605
|
+
@staticmethod
|
|
606
|
+
def from_dir(src, filter=None):
|
|
607
|
+
"""Loads a nested dictionary structure of a directory.
|
|
608
|
+
|
|
609
|
+
`Load.from_dir(path)` returns `path` if `path` refers to a file.
|
|
610
|
+
Alternatively, if `path` refers to a directory, this function returns a
|
|
611
|
+
lazy dictionary whose keys are the names of the entries in the directory
|
|
612
|
+
and whose values are the result of calling `Load.from_dir` on their
|
|
613
|
+
paths.
|
|
614
|
+
"""
|
|
615
|
+
src = path(src)
|
|
616
|
+
if src.is_file():
|
|
617
|
+
raise NotADirectoryError(src)
|
|
618
|
+
if filter:
|
|
619
|
+
d = {
|
|
620
|
+
p.name: (
|
|
621
|
+
lazy(Load.from_dir, p, filter=filter) if p.is_dir() else p)
|
|
622
|
+
for p in src.iterdir()
|
|
623
|
+
if filter(p)}
|
|
624
|
+
else:
|
|
625
|
+
d = {
|
|
626
|
+
p.name: (lazy(Load.from_dir, p) if p.is_dir() else p)
|
|
627
|
+
for p in src.iterdir()}
|
|
628
|
+
return ldict(d)
|
|
629
|
+
|
|
630
|
+
# We can go ahead and declare a single global save object for the immlib
|
|
631
|
+
# library. We will later use this object to register various basic format
|
|
632
|
+
# types.
|
|
633
|
+
load = Load()
|
|
634
|
+
|
|
635
|
+
# Register a few basic file formats.
|
|
636
|
+
@load.register('str', mode='t')
|
|
637
|
+
def load_str(stream, strip_nl=False, size=-1):
|
|
638
|
+
"""Loads a string from the given source."""
|
|
639
|
+
s = stream.read(size)
|
|
640
|
+
if strip_nl:
|
|
641
|
+
s = s.rstrip('\n')
|
|
642
|
+
return s
|
|
643
|
+
@load.register('bytes', mode='b')
|
|
644
|
+
def load_bytes(stream, size=-1):
|
|
645
|
+
"""Loads a string from the given source."""
|
|
646
|
+
return stream.read(size)
|
|
647
|
+
@load.register('repr', mode='t')
|
|
648
|
+
def load_repr(stream, size=-1):
|
|
649
|
+
"""Loads an object using the `ast.literal_eval` function."""
|
|
650
|
+
from ast import literal_eval
|
|
651
|
+
return literal_eval(stream.read(size))
|
|
652
|
+
@load.register('text', '.txt', '.text', mode='t')
|
|
653
|
+
def load_text(stream, strip_nls=True, size=-1):
|
|
654
|
+
"""Loads a blob of text or a sequence of lines from a stream or path.
|
|
655
|
+
|
|
656
|
+
Parameters
|
|
657
|
+
----------
|
|
658
|
+
stream : stream or path-like
|
|
659
|
+
The stream or path from which to read the lines.
|
|
660
|
+
strip_nls : boolean, optional
|
|
661
|
+
Whether to strip newlines from the end of each line. If `False`, then
|
|
662
|
+
the lines are returned as they are; if `True` (the default), any newline
|
|
663
|
+
character is stripped from each line in the `lines` argument.
|
|
664
|
+
"""
|
|
665
|
+
if strip_nls:
|
|
666
|
+
return stream.read(size).splitlines()
|
|
667
|
+
else:
|
|
668
|
+
return stream.readlines(size)
|
|
669
|
+
@load.register('pickle', '.pickle', '.pkl', '.pcl', mode='b')
|
|
670
|
+
def load_pickle(stream, **kwargs):
|
|
671
|
+
"""Loads a pickled object from a path or stream and returns the object.
|
|
672
|
+
|
|
673
|
+
All keyword options are forwarded to the `pickle.load` function.
|
|
674
|
+
"""
|
|
675
|
+
import pickle
|
|
676
|
+
return pickle.load(stream, **kwargs)
|
|
677
|
+
@load.register('numpy', '.npy', '.np', '.numpy', mode='b', gzip_suffix='.npz')
|
|
678
|
+
def load_numpy(stream, **kwargs):
|
|
679
|
+
"""Loads a numpy object from a path or stream and returns the object.
|
|
680
|
+
|
|
681
|
+
All keyword options are forwarded to the `numpy.load` function.
|
|
682
|
+
"""
|
|
683
|
+
import numpy as np
|
|
684
|
+
return np.load(stream, **kwargs)
|
|
685
|
+
@load.register('json', '.json', mode='t')
|
|
686
|
+
def load_json(stream, /, **kwargs):
|
|
687
|
+
"""Loads an object from a JSON stream or path and returns the object.
|
|
688
|
+
|
|
689
|
+
All keywords are passed along to the `json.load` function.
|
|
690
|
+
"""
|
|
691
|
+
import json
|
|
692
|
+
return json.load(stream, **kwargs)
|
|
693
|
+
@load.register('yaml', '.yaml', '.yml', mode='t')
|
|
694
|
+
def load_yaml(stream, /, safe=True):
|
|
695
|
+
"""Loads an object from a YAML stream or path and returns the object.
|
|
696
|
+
|
|
697
|
+
The optional argument `safe` may be set to `False` to use unsafe YAML
|
|
698
|
+
loading via the `yaml.load()` function.
|
|
699
|
+
"""
|
|
700
|
+
import yaml
|
|
701
|
+
if safe:
|
|
702
|
+
return yaml.safe_load(stream)
|
|
703
|
+
else:
|
|
704
|
+
return yaml.load(stream)
|
|
705
|
+
@load.register('csv', '.csv', mode='t')
|
|
706
|
+
def load_csv(stream, /, sep=',', **kwargs):
|
|
707
|
+
"""Loads a pandas DataFrame from a CSV file.
|
|
708
|
+
|
|
709
|
+
All options are passed along to `pandas.read_csv()`.
|
|
710
|
+
"""
|
|
711
|
+
import pandas
|
|
712
|
+
return pandas.read_csv(stream, sep=sep, **kwargs)
|
|
713
|
+
@load.register('tsv', '.tsv', mode='t')
|
|
714
|
+
def load_tsv(stream, /, sep="\t", **kwargs):
|
|
715
|
+
"""Loads a pandas DataFrame from a TSV file.
|
|
716
|
+
|
|
717
|
+
All options are passed along to `pandas.read_csv()`.
|
|
718
|
+
"""
|
|
719
|
+
import pandas
|
|
720
|
+
return pandas.read_csv(stream, sep=sep, **kwargs)
|