immlib 1.0.0.dev2__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (45) hide show
  1. immlib/__init__.py +131 -0
  2. immlib/_init.py +108 -0
  3. immlib/_version.py +235 -0
  4. immlib/doc/__init__.py +38 -0
  5. immlib/doc/_core.py +311 -0
  6. immlib/iolib/__init__.py +29 -0
  7. immlib/iolib/_core.py +720 -0
  8. immlib/pathlib/__init__.py +69 -0
  9. immlib/pathlib/_cache.py +152 -0
  10. immlib/pathlib/_core.py +869 -0
  11. immlib/pathlib/_osf.py +538 -0
  12. immlib/test/__init__.py +16 -0
  13. immlib/test/__main__.py +10 -0
  14. immlib/test/doc/__init__.py +6 -0
  15. immlib/test/doc/test_core.py +91 -0
  16. immlib/test/iolib/__init__.py +7 -0
  17. immlib/test/iolib/test_core.py +81 -0
  18. immlib/test/pathlib/__init__.py +11 -0
  19. immlib/test/pathlib/test_core.py +146 -0
  20. immlib/test/pathlib/test_osf.py +54 -0
  21. immlib/test/types/__init__.py +5 -0
  22. immlib/test/types/test_core.py +110 -0
  23. immlib/test/util/__init__.py +11 -0
  24. immlib/test/util/test_core.py +681 -0
  25. immlib/test/util/test_numeric.py +1374 -0
  26. immlib/test/util/test_quantity.py +218 -0
  27. immlib/test/util/test_url.py +51 -0
  28. immlib/test/workflow/__init__.py +9 -0
  29. immlib/test/workflow/test_core.py +418 -0
  30. immlib/test/workflow/test_plantype.py +248 -0
  31. immlib/types/__init__.py +29 -0
  32. immlib/types/_core.py +333 -0
  33. immlib/util/__init__.py +283 -0
  34. immlib/util/_core.py +2524 -0
  35. immlib/util/_numeric.py +2651 -0
  36. immlib/util/_quantity.py +523 -0
  37. immlib/util/_url.py +114 -0
  38. immlib/workflow/__init__.py +48 -0
  39. immlib/workflow/_core.py +1635 -0
  40. immlib/workflow/_plantype.py +334 -0
  41. immlib-1.0.0.dev2.dist-info/METADATA +76 -0
  42. immlib-1.0.0.dev2.dist-info/RECORD +45 -0
  43. immlib-1.0.0.dev2.dist-info/WHEEL +5 -0
  44. immlib-1.0.0.dev2.dist-info/licenses/LICENSE +21 -0
  45. immlib-1.0.0.dev2.dist-info/top_level.txt +1 -0
immlib/iolib/_core.py ADDED
@@ -0,0 +1,720 @@
1
+ # -*- coding: utf-8 -*-
2
+ ################################################################################
3
+ # immlib/iolib/_core.py
4
+
5
+ """Input and output operations via the load() and save() functions."""
6
+
7
+
8
+ # Dependencies #################################################################
9
+
10
+ import os, io, gzip, numbers
11
+ from pathlib import Path
12
+
13
+ import numpy as np
14
+ from pcollections import pdict, plist, pset, ldict, lazy
15
+
16
+ from ..doc import docwrap
17
+ from ..util import is_str, is_amap, is_aseq
18
+ from ..pathlib import path, is_path, like_path
19
+
20
+
21
+ # Format and Formatter #########################################################
22
+
23
+ # The Format class, an inset class for managing the various stream formats.
24
+ class Format:
25
+ """Manages the details of a stream format for use with `save` and `load`.
26
+
27
+ `Format` objects are typically created with the `save.register` or
28
+ `load.register` methods. This class should be handled internally but should
29
+ not be instantiated outside of the `Save` or `Load` classes.
30
+
31
+ `Format` objects can be treated as save functions: the `__call__` method
32
+ on a `Format` object accepts a stream or path-like object representing
33
+ the destination to which the format should be written, an object to save
34
+ in the format, and any optional arguments understood by the type.
35
+ """
36
+ def __init__(self, name, function, *suffixes, mode='b', gzip_suffix=None):
37
+ self.name = name
38
+ self.function = function
39
+ self.suffixes = []
40
+ for suff in suffixes:
41
+ if is_str(suff):
42
+ suff = (suff,)
43
+ elif all(map(is_str, suff)):
44
+ suff = tuple(suff)
45
+ else:
46
+ raise ValueError(f"invalid path suffix: {suff}")
47
+ self.suffixes.append(suff)
48
+ if mode != 'b' and mode != 't':
49
+ raise ValueError("format mode must be 't' or 'b'")
50
+ self.mode = mode
51
+ if gzip_suffix is None:
52
+ gzip_suffix = ()
53
+ elif is_str(gzip_suffix):
54
+ gzip_suffix = ((gzip_suffix,),)
55
+ elif is_aseq(gzip_suffix):
56
+ if all(map(is_str, gzip_suffix)):
57
+ gzip_suffix = tuple((s,) for s in gzip_suffix)
58
+ else:
59
+ g = []
60
+ for suff in gzip_suffix:
61
+ if is_str(suff):
62
+ g.append((suff,))
63
+ elif is_aseq(suff) and all(map(is_str, suff)):
64
+ g.append(tuple(suff))
65
+ else:
66
+ raise ValueError(f"invalid gzip suffix: {suff}")
67
+ gzip_suffix = tuple(g)
68
+ else:
69
+ raise ValueError(f"invalid gzip suffix: {gzip_suffix}")
70
+ self.gzip_suffix = gzip_suffix
71
+ # We also want to change the documentation.
72
+ self.__doc__ = self.function.__doc__
73
+ def __call__(self, stream, stream_mode, *args, **kwargs):
74
+ # If dest isn't a stream, we need to open it as a path.
75
+ if isinstance(stream, io.IOBase):
76
+ return self.function(stream, *args, **kwargs)
77
+ else:
78
+ p = path(stream)
79
+ with p.open(stream_mode + self.mode) as s:
80
+ return self.function(s, *args, **kwargs)
81
+ class Formatter:
82
+ """Base class for the Save and Load classes.
83
+
84
+ Handles common operations between saving and loading systems.
85
+ """
86
+ # The mode use for opening streams from paths ('r' or 'w' typically).
87
+ stream_mode = ''
88
+ # Data/Details of the Save/Load classes.
89
+ __slots__ = ("formats", "_format_by_suffix")
90
+ # Constructor.
91
+ def __init__(self, template=None):
92
+ self.formats = {}
93
+ self._format_by_suffix = {}
94
+ if template is not None:
95
+ cls = type(self)
96
+ if isinstance(template, cls):
97
+ formats = template.formats
98
+ elif is_amap(template):
99
+ formats = templates
100
+ else:
101
+ raise TypeError(f"template must be a {cls} object or a mapping")
102
+ for (k,format) in formats.items():
103
+ if not isinstance(format, Format):
104
+ raise TypeError(f"template contains non-format named {k}")
105
+ self.register(format)
106
+ def deduce_format(self, arg, ignore_gz=True):
107
+ """Deduces the file format for a given path or suffix.
108
+
109
+ `save.deduce_format(arg)` deduces the format implied by `arg`. The
110
+ argument may be a path, a string representing a path, or a sequence of
111
+ strings representing the suffixes of the path. If the final suffix is
112
+ `.gz` then this suffix is ignored.
113
+ """
114
+ if is_str(arg):
115
+ suffixes = path(arg).suffixes
116
+ elif is_path(arg):
117
+ suffixes = arg.suffixes
118
+ elif is_aseq(arg) and all(map(is_str, arg)):
119
+ suffixes = arg
120
+ else:
121
+ raise ValueError(
122
+ f"deduce_format given invalid argument of type {type(arg)}")
123
+ suff = tuple(suffixes)
124
+ if ignore_gz and suff[-1] == '.gz':
125
+ suff = suff[:-1]
126
+ while suff:
127
+ format = self._format_by_suffix.get(suff)
128
+ if format:
129
+ return format
130
+ suff = suff[1:]
131
+ # No format found.
132
+ return None
133
+ @staticmethod
134
+ def _expandall(p):
135
+ p = path(p)
136
+ if isinstance(p, Path):
137
+ p = Path(os.path.expanduser(os.path.expandvars(os.fspath(p))))
138
+ return p
139
+ def _call(self, target, format, gzip, /, *args, **kwargs):
140
+ if isinstance(target, io.IOBase):
141
+ target_path = None
142
+ target_stream = target
143
+ if gzip is Ellipsis:
144
+ gzip = False
145
+ else:
146
+ target_path = path(target)
147
+ if isinstance(target_path, Path):
148
+ target_path = Formatter._expandall(target_path)
149
+ target_stream = None
150
+ if format is None:
151
+ if target_path is None:
152
+ raise ValueError("format can't be None when target is a stream")
153
+ # We can walk through the formats and attempt to deduce the correct
154
+ # format from the ending of the dest_str.
155
+ format = self.deduce_format(target_path)
156
+ if not format:
157
+ raise ValueError(f"format can't be deduced from path: {target}")
158
+ # At this point we have a format name or we've raised an error.
159
+ if is_str(format):
160
+ fmt = self.formats.get(format)
161
+ if fmt is None:
162
+ raise ValueError(f"format not recognized: {format}")
163
+ else:
164
+ format = fmt
165
+ elif not isinstance(format, Format):
166
+ raise TypeError("format arg must be a format name or Format object")
167
+ # Now we know the format and thus have a format function; if we were
168
+ # given a path instead of a stream, we need to open the path for the
169
+ # formatting function. We also need to handle gzipping--if the format is
170
+ # a gzip format then gzip will be True at this point.
171
+ if gzip is Ellipsis:
172
+ suff = tuple(target_path.suffixes)
173
+ gzip = (target_path.suffix == '.gz') or suff in format.gzip_suffix
174
+ if gzip is True:
175
+ if target_stream:
176
+ with self._gzip(target_stream, format.mode) as act_stream:
177
+ r = format.function(act_stream, *args, **kwargs)
178
+ return (r, target_stream)
179
+ else:
180
+ with target_path.open(self.stream_mode + 'b') as stream:
181
+ with self._gzip(stream, format.mode) as act_stream:
182
+ r = format.function(act_stream, *args, **kwargs)
183
+ return (r, target_path)
184
+ elif target_stream:
185
+ r = format.function(target_stream, *args, **kwargs)
186
+ return (r, target_stream)
187
+ else:
188
+ with target_path.open(self.stream_mode + format.mode) as stream:
189
+ r = format.function(stream, *args, **kwargs)
190
+ return (r, target_path)
191
+ def _gzip(self, stream, mode,
192
+ compresslevel=9,
193
+ encoding=None,
194
+ errors=None,
195
+ newline=None):
196
+ return gzip.open(
197
+ stream,
198
+ mode=(self.stream_mode + mode),
199
+ compresslevel=compresslevel,
200
+ encoding=encoding,
201
+ errors=errors,
202
+ newline=newline)
203
+ def register(self, name, /, *suffixes, mode='b', gzip_suffix=None):
204
+ """Registers a format type with the `save`/`load` interface.
205
+
206
+ The `immlib.save.register` method is intended to be used as a decorator:
207
+ `@immlib.save.register(format_name, suffix1, suffix2...)` ensures that
208
+ the function that follows it is registered as a format managed by the
209
+ `immlib.save` system. The function that is decorated should be written
210
+ to accept a stream object (though the decorator will ensure that it
211
+ works when called with a path-like object or a stream). The function's
212
+ signature must always match `f(output_stream, object_to_save, **opts)`
213
+ where the `**opts` must be any set of named parameters. The return value
214
+ of the function is ignored, but it should raise an error if the object
215
+ cannot be saved.
216
+
217
+ The `immlib.load.register` method is identical except that the function
218
+ that follows is not passed an object to load and instead must return the
219
+ object that is loaded from the given stream.
220
+
221
+ Parameters
222
+ ----------
223
+ name : str
224
+ The name of the format.
225
+ *suffixes : strings or tuples of strings, optional
226
+ Any number of suffixes that this format uses. Suffixes must be
227
+ unique, and an error is raised if a requested suffix is already
228
+ registered to another format. Each suffix may be a string such as
229
+ `.tgz` or a tuple such as `('.tar', '.gz')`.
230
+ mode : 't' or 'b', optional
231
+ Whether any stream that is opened in order to save the file should
232
+ be opened in text mode (`'t'`) or binary mode (`'b'`). The default
233
+ is `'b'`.
234
+ gzip_suffix : None or str or tuple of str
235
+ A sequence of suffixes that indicate that the format is additionally
236
+ encoded using gzip. Files that end with `.gz` are automatically
237
+ interpreted as gzipped files by the `save` system, but if an
238
+ additional file ending needs to be specified as a gzip format, such
239
+ as the suffix `.npz` for gzipped-numpy files (equivalent to
240
+ `.npy.gz`), then it should be specified in this option. If a single
241
+ string is given (such as `'.npz'`), then it is interpreted as a
242
+ single suffix (`[['.npz']]`); if a sequence of strings are given
243
+ (`['.npz', '.nz']`), then they are interpreted as individual
244
+ suffixes (`[['.npz'], ['.nz']]`). For compound suffixes, they must
245
+ be elements of a sequence, e.g. `[['.json', '.gz']]` not `['.json',
246
+ '.gz']`. Note, however, that `[['.json', '.gz']]` is automatically
247
+ interpreted as a gzip file by virtue of its `.gz` suffix.
248
+ """
249
+ # There are actually two uses: (1) that described in the help above and
250
+ # (2) save.register(format) where format is already a Format object. In
251
+ # the latter case we aren't a decorator.
252
+ if isinstance(name, Format):
253
+ # We're in case 2, so we register this format specifically.
254
+ format = name
255
+ if format.name in self.formats:
256
+ raise RuntimeError(f"format {format.name} already registered")
257
+ suffs = tuple(format.suffixes) + format.gzip_suffix
258
+ for suff in suffs:
259
+ ex = self._format_by_suffix.get(suff)
260
+ if ex is not None:
261
+ raise RuntimeError(
262
+ f"suffix {suff} already mapped to format {ex.name}")
263
+ # We pass the criteria; go ahead and add this format.
264
+ self.formats[format.name] = format
265
+ for suff in suffs:
266
+ self._format_by_suffix[suff] = format
267
+ # Return the format itself.
268
+ return format
269
+ else:
270
+ # We're in case 1, so we return a decorator for the function.
271
+ def _formatter_register_dec(f):
272
+ if isinstance(f, Format):
273
+ f = f.function
274
+ format = Format(
275
+ name, f, *suffixes,
276
+ mode=mode,
277
+ gzip_suffix=gzip_suffix)
278
+ return self.register(format)
279
+ return _formatter_register_dec
280
+ # That's it for the register function.
281
+ def unregister(self, name, *, error_on_missing=False):
282
+ """Unregisters the format with the given name from the save/load system.
283
+
284
+ The format with the given name is unregistered from the `immlib.save`
285
+ system, and the `Format` object that is removed is returned. If the
286
+ format is not found, then `None` is returned, but no error is raised by
287
+ default.
288
+
289
+ The `immlib.load.unregister` method works identically to the
290
+ `immlib.save` version.
291
+
292
+ Parameters
293
+ ----------
294
+ name : str
295
+ The name of the format to unregister.
296
+ error_on_missing : boolean, optional
297
+ Whether to throw a `RuntimeError` if the named format is not found
298
+ in the save manager. By default this is `False`.
299
+
300
+ Returns
301
+ -------
302
+ Format or None
303
+ The format object that is unregistered or `None` if the given name
304
+ was not a registered format.
305
+
306
+ Raises
307
+ ------
308
+ RuntimeError
309
+ If the given name does not map to a format in the save manager and
310
+ the `error_on_missing` option is `True`.
311
+ """
312
+ format = self.formats.get(name)
313
+ if format is None:
314
+ if error_on_missing:
315
+ fnnm = type(self).__name__.lower()
316
+ raise RuntimeError(
317
+ f"format {name} not found in immlib {fnnm} system")
318
+ else:
319
+ return None
320
+ # Actually remove things:
321
+ for suff in (format.suffixes + format.gzip_suffix):
322
+ del self._format_by_suffix[suff]
323
+ del self.formats[name]
324
+ return format
325
+ def copy(self):
326
+ """Returns a copy of the given save manager.
327
+
328
+ `immlib.save.copy()` can be used to return a copy of the save manager,
329
+ for instances where a single save manager is not ideal.
330
+ """
331
+ cls = type(self)
332
+ return cls(self)
333
+
334
+
335
+ # Save #########################################################################
336
+
337
+ class Save(Formatter):
338
+ """Saves a Python object to a stream or path then returns the stream/path.
339
+
340
+ `immlib.save(path, object, format)` saves the given `object` to the given
341
+ `path` using the named file `format` and returns the path on success or
342
+ rasies an error on failure. If the format can be deduced from the path
343
+ suffix, then it may be omitted.
344
+
345
+ `immlib.save(stream, object, format)` writes the given `object` to the given
346
+ stream object using the named `format` and returns the stream.
347
+
348
+ In fact, `immlib.save` is an object of the `immlib.pathlib.Save` type that
349
+ primarily behaves like a function. The `immlib.save.register` and
350
+ `immlib.save.unregister` methods can be used to add understood formats.
351
+
352
+ Parameters
353
+ ----------
354
+ dest : path-like or stream
355
+ The output destination to which the object is to be saved. This may be
356
+ either a path-like object or a writeable `IOBase` stream.
357
+ obj : object
358
+ Any object that can be saved in the given format.
359
+ format : str or None, optional
360
+ If provided, `format` must be a string that names a format that has been
361
+ previously registered with `save` using the `save.register` method. If
362
+ `format` is not provided or is `None` (the default), then an attempt is
363
+ made to deduce the format using the suffix of the `dest` argument,
364
+ assuming that `dest` is a path and not a stream. If `dest` is a stream
365
+ or if the suffix does not indicate a specific format, then a
366
+ `ValueError` is raised.
367
+ **kwargs
368
+ Any additional parameters are passed to the registered export function.
369
+
370
+ Returns
371
+ -------
372
+ path-like or stream
373
+ The destination stream or path. If the `dest` argument is a path name,
374
+ then a path object is returned instead of the path name.
375
+
376
+ Raises
377
+ ------
378
+ TypeError
379
+ If the destination is not a stream or path-like object or if the format
380
+ is not a string.
381
+ ValueError
382
+ If the format is not recognized or if it cannot be deduced from the
383
+ destination object.
384
+ """
385
+ stream_mode = 'w'
386
+ def __call__(self, dest, obj, format=None, /, gzip=Ellipsis, **kwargs):
387
+ (loadret, saveret) = self._call(dest, format, gzip, obj, **kwargs)
388
+ return saveret
389
+
390
+ # We can go ahead and declare a single global save object for the immlib
391
+ # library. We will later use this object to register various basic format
392
+ # types.
393
+ save = Save()
394
+
395
+ # Register a few basic file formats.
396
+ @save.register('str', mode='t')
397
+ def save_str(stream, obj, append_nl=False):
398
+ """Saves `str(obj)` to the given stream."""
399
+ stream.write(str(obj))
400
+ if append_nl:
401
+ stream.write('\n')
402
+ @save.register('bytes', mode='b')
403
+ def save_bytes(stream, obj):
404
+ """Saves `bytes(obj)` to the given stream."""
405
+ stream.write(bytes(obj))
406
+ @save.register('repr', mode='t')
407
+ def save_repr(stream, obj, append_nl=False):
408
+ """Saves `repr(obj)` to the given stream."""
409
+ stream.write(repr(obj))
410
+ if append_nl:
411
+ stream.write('\n')
412
+ @save.register('text', '.txt', '.text', mode='t')
413
+ def save_text(stream, lines, append_nls=False):
414
+ """Saves a blob of text or a sequence of lines to a stream or path.
415
+
416
+ Parameters
417
+ ----------
418
+ stream : stream or path-like
419
+ The stream or path to which to write the lines.
420
+ lines : string or sequence of strings
421
+ The text that should be written to the file or stream.
422
+ append_nls : boolean, optional
423
+ Whether to append newlines to the end of each line. If `False` (the
424
+ default), then the lines are written as they are; otherwise, a newline
425
+ character is appended after each line in the `lines` argument.
426
+ """
427
+ if is_str(lines):
428
+ lines = (lines,)
429
+ if not is_aseq(lines):
430
+ raise TypeError("lines must be text or a sequence of text objects")
431
+ if append_nls:
432
+ for ln in lines:
433
+ stream.write(ln)
434
+ stream.write('\n')
435
+ else:
436
+ for ln in lines:
437
+ stream.write(ln)
438
+ @save.register('pickle', '.pickle', '.pkl', '.pcl', mode='b')
439
+ def save_pickle(stream, obj, protocol=None, **kwargs):
440
+ """Saves a pickled object to a destination path or stream.
441
+
442
+ All keyword options are forwarded to the `pickle.dump` function. The
443
+ `protocol` keyword option is provided as the `protocol` positional argument
444
+ to the `pickle.dump` function.
445
+ """
446
+ import pickle
447
+ pickle.dump(obj, stream, protocol, **kwargs)
448
+ @save.register('numpy', '.npy', '.np', '.numpy', mode='b', gzip_suffix='.npz')
449
+ def save_numpy(stream, obj, **kwargs):
450
+ """Saves a numpy object to a destination path or stream.
451
+
452
+ All keyword options are forwarded to the `numpy.save` function.
453
+ """
454
+ import numpy as np
455
+ np.save(stream, obj, **kwargs)
456
+ def json_default(obj):
457
+ """Converts an object to a json-formattable object or raises TypeError.
458
+ """
459
+ if is_str(obj) or obj is None or obj is True or obj is False:
460
+ return obj
461
+ elif isinstance(obj, numbers.Integral):
462
+ return int(obj)
463
+ elif isinstance(obj, numbers.Real):
464
+ return float(obj)
465
+ elif is_amap(obj):
466
+ return dict(obj)
467
+ elif is_aseq(obj):
468
+ return list(obj)
469
+ elif isinstance(obj, np.ndarray):
470
+ return obj.tolist()
471
+ #elif is_planobject(obj):
472
+ # cls = type(obj)
473
+ # return {
474
+ # '__plantype__': f'{cls.__module__}.{cls.__name__}',
475
+ # '__params__': dict(obj.__plandict__.params)}
476
+ else:
477
+ raise TypeError(type(obj))
478
+ @save.register('json', '.json', mode='t')
479
+ def save_json(stream, obj, /, default=json_default, **kwargs):
480
+ """Saves an object as a JSON string or raises a TypeError if not possible.
481
+
482
+ All keywords are passed along to the `json.dump` function. The `default`
483
+ option uses the `immlib.iolib.json_default` function, which is different
484
+ than the default used by `json.dump`, but all other options are unaltered.
485
+ """
486
+ import json
487
+ json.dump(obj, stream, default=default, **kwargs)
488
+ def yaml_prepare(obj):
489
+ """Returns a version of the argument that can be JSON/YAML serialized."""
490
+ if is_str(obj) or obj is None or obj is True or obj is False:
491
+ return obj
492
+ elif isinstance(obj, numbers.Integral):
493
+ return int(obj)
494
+ elif isinstance(obj, numbers.Real):
495
+ return float(obj)
496
+ elif isinstance(obj, np.ndarray):
497
+ return obj.tolist()
498
+ elif is_amap(obj):
499
+ r = {}
500
+ for (k,v) in obj.items():
501
+ if not is_str(k):
502
+ raise TypeError("JSON/YAML dict keys must be strings")
503
+ r[k] = yaml_prepare(v)
504
+ return r
505
+ elif is_aseq(obj):
506
+ return [yaml_prepare(u) for u in obj]
507
+ else:
508
+ raise TypeError(type(obj))
509
+ @save.register('yaml', '.yaml', mode='t')
510
+ def save_yaml(stream, obj, /, **kwargs):
511
+ """Saves an object as a YAML string or raises a TypeError if not possible.
512
+
513
+ All keywords are passed along to the `yaml.Dumper` object that is used for
514
+ dumping `obj`. Note that unlike with the `yaml.dump` function, object are
515
+ only serialized if they are JSON/YAML serializable; classes that are not
516
+ mappings, sequences, numbers, strings, or booleans cannot be serialized
517
+ using this function and instead raise TypeErrors.
518
+ """
519
+ import yaml
520
+ yaml.dump(yaml_prepare(obj), stream, **kwargs)
521
+ @save.register('csv', '.csv', mode='t')
522
+ def save_csv(stream, obj, /, index=False, **kwargs):
523
+ """Saves a pandas DataFrame to a CSV file.
524
+
525
+ All options are passed along to `pandas.DataFrame.to_csv()`. The option
526
+ `index` has the default value of `False` in this function.
527
+ """
528
+ import pandas
529
+ obj = pandas.DataFrame(obj)
530
+ obj.to_csv(stream, index=index, **kwargs)
531
+ @save.register('tsv', '.tsv', mode='t')
532
+ def save_tsv(stream, obj, /, sep="\t", index=False, **kwargs):
533
+ """Saves a pandas DataFrame to a TSV file.
534
+
535
+ All options are passed along to `pandas.DataFrame.to_csv()`. The option
536
+ `index` has the default value of `False` in this function, and the option
537
+ `sep` has the default value `"\t"`.
538
+ """
539
+ import pandas
540
+ obj = pandas.DataFrame(obj)
541
+ obj.to_csv(stream, sep=sep, index=index, **kwargs)
542
+
543
+
544
+ # Load #########################################################################
545
+
546
+ class Load(Formatter):
547
+ """Loads a Python object from a stream or path then returns the object.
548
+
549
+ `immlib.load(path, format)` loads a Python object from the given `path`
550
+ using the named file `format` and returns the loaded object or rasies an
551
+ error on failure. If the format can be deduced from the path suffix, then it
552
+ may be omitted.
553
+
554
+ `immlib.load(stream, format)` reads a Python object from the given stream
555
+ object using the named `format` and returns the object. The format argument
556
+ cannot be omitted when the first argument is a stream.
557
+
558
+ In fact, `immlib.load` is an object of the `immlib.pathlib.Load` type that
559
+ primarily behaves like a function. The `immlib.load.register` and
560
+ `immlib.load.unregister` methods can be used to add understood formats.
561
+
562
+ Parameters
563
+ ----------
564
+ source : path-like or stream
565
+ The input source from which the object is to be loaded. This may be
566
+ either a path-like object or a readable `IOBase` stream.
567
+ format : str or None, optional
568
+ If provided, `format` must be a string that names a format that has been
569
+ previously registered with `load` using the `load.register` method. If
570
+ `format` is not provided or is `None` (the default), then an attempt is
571
+ made to deduce the format using the suffix of the `source` argument,
572
+ assuming that `source` is a path and not a stream. If `source` is a
573
+ stream or if the suffix does not indicate a specific format, then a
574
+ `ValueError` is raised.
575
+ **kwargs
576
+ Any additional parameters are passed to the registered export function.
577
+
578
+ Returns
579
+ -------
580
+ object
581
+ The object that was loaded from the given stream or path.
582
+
583
+ Raises
584
+ ------
585
+ TypeError
586
+ If the input source is not a stream or path-like object or if the format
587
+ is not a string.
588
+ ValueError
589
+ If the format is not recognized or if it cannot be deduced from the
590
+ source.
591
+ """
592
+ stream_mode = 'r'
593
+ def __call__(self, src, format=None, /, gzip=Ellipsis, **kwargs):
594
+ # There's a special case for loading: if we're given a path, and it
595
+ # refers to a directory, we load it as a lazy dictionary.
596
+ if format is None:
597
+ if not isinstance(src, io.IOBase):
598
+ p = Formatter._expandall(src)
599
+ if p.is_dir():
600
+ format = 'dir'
601
+ if format == 'dir':
602
+ return Load.from_dir(src)
603
+ (loadret, saveret) = self._call(src, format, gzip, **kwargs)
604
+ return loadret
605
+ @staticmethod
606
+ def from_dir(src, filter=None):
607
+ """Loads a nested dictionary structure of a directory.
608
+
609
+ `Load.from_dir(path)` returns `path` if `path` refers to a file.
610
+ Alternatively, if `path` refers to a directory, this function returns a
611
+ lazy dictionary whose keys are the names of the entries in the directory
612
+ and whose values are the result of calling `Load.from_dir` on their
613
+ paths.
614
+ """
615
+ src = path(src)
616
+ if src.is_file():
617
+ raise NotADirectoryError(src)
618
+ if filter:
619
+ d = {
620
+ p.name: (
621
+ lazy(Load.from_dir, p, filter=filter) if p.is_dir() else p)
622
+ for p in src.iterdir()
623
+ if filter(p)}
624
+ else:
625
+ d = {
626
+ p.name: (lazy(Load.from_dir, p) if p.is_dir() else p)
627
+ for p in src.iterdir()}
628
+ return ldict(d)
629
+
630
+ # We can go ahead and declare a single global save object for the immlib
631
+ # library. We will later use this object to register various basic format
632
+ # types.
633
+ load = Load()
634
+
635
+ # Register a few basic file formats.
636
+ @load.register('str', mode='t')
637
+ def load_str(stream, strip_nl=False, size=-1):
638
+ """Loads a string from the given source."""
639
+ s = stream.read(size)
640
+ if strip_nl:
641
+ s = s.rstrip('\n')
642
+ return s
643
+ @load.register('bytes', mode='b')
644
+ def load_bytes(stream, size=-1):
645
+ """Loads a string from the given source."""
646
+ return stream.read(size)
647
+ @load.register('repr', mode='t')
648
+ def load_repr(stream, size=-1):
649
+ """Loads an object using the `ast.literal_eval` function."""
650
+ from ast import literal_eval
651
+ return literal_eval(stream.read(size))
652
+ @load.register('text', '.txt', '.text', mode='t')
653
+ def load_text(stream, strip_nls=True, size=-1):
654
+ """Loads a blob of text or a sequence of lines from a stream or path.
655
+
656
+ Parameters
657
+ ----------
658
+ stream : stream or path-like
659
+ The stream or path from which to read the lines.
660
+ strip_nls : boolean, optional
661
+ Whether to strip newlines from the end of each line. If `False`, then
662
+ the lines are returned as they are; if `True` (the default), any newline
663
+ character is stripped from each line in the `lines` argument.
664
+ """
665
+ if strip_nls:
666
+ return stream.read(size).splitlines()
667
+ else:
668
+ return stream.readlines(size)
669
+ @load.register('pickle', '.pickle', '.pkl', '.pcl', mode='b')
670
+ def load_pickle(stream, **kwargs):
671
+ """Loads a pickled object from a path or stream and returns the object.
672
+
673
+ All keyword options are forwarded to the `pickle.load` function.
674
+ """
675
+ import pickle
676
+ return pickle.load(stream, **kwargs)
677
+ @load.register('numpy', '.npy', '.np', '.numpy', mode='b', gzip_suffix='.npz')
678
+ def load_numpy(stream, **kwargs):
679
+ """Loads a numpy object from a path or stream and returns the object.
680
+
681
+ All keyword options are forwarded to the `numpy.load` function.
682
+ """
683
+ import numpy as np
684
+ return np.load(stream, **kwargs)
685
+ @load.register('json', '.json', mode='t')
686
+ def load_json(stream, /, **kwargs):
687
+ """Loads an object from a JSON stream or path and returns the object.
688
+
689
+ All keywords are passed along to the `json.load` function.
690
+ """
691
+ import json
692
+ return json.load(stream, **kwargs)
693
+ @load.register('yaml', '.yaml', '.yml', mode='t')
694
+ def load_yaml(stream, /, safe=True):
695
+ """Loads an object from a YAML stream or path and returns the object.
696
+
697
+ The optional argument `safe` may be set to `False` to use unsafe YAML
698
+ loading via the `yaml.load()` function.
699
+ """
700
+ import yaml
701
+ if safe:
702
+ return yaml.safe_load(stream)
703
+ else:
704
+ return yaml.load(stream)
705
+ @load.register('csv', '.csv', mode='t')
706
+ def load_csv(stream, /, sep=',', **kwargs):
707
+ """Loads a pandas DataFrame from a CSV file.
708
+
709
+ All options are passed along to `pandas.read_csv()`.
710
+ """
711
+ import pandas
712
+ return pandas.read_csv(stream, sep=sep, **kwargs)
713
+ @load.register('tsv', '.tsv', mode='t')
714
+ def load_tsv(stream, /, sep="\t", **kwargs):
715
+ """Loads a pandas DataFrame from a TSV file.
716
+
717
+ All options are passed along to `pandas.read_csv()`.
718
+ """
719
+ import pandas
720
+ return pandas.read_csv(stream, sep=sep, **kwargs)