flupy 1.2.2__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
flupy/__init__.py ADDED
@@ -0,0 +1,9 @@
1
+ from importlib.metadata import version
2
+
3
+ from flupy.cli.utils import walk_dirs, walk_files
4
+ from flupy.fluent import flu
5
+
6
+ __project__ = "flupy"
7
+ __version__ = version(__project__)
8
+
9
+ __all__ = ["flu", "walk_files", "walk_dirs"]
flupy/cli/__init__.py ADDED
File without changes
flupy/cli/cli.py ADDED
@@ -0,0 +1,89 @@
1
+ import argparse
2
+ import importlib
3
+ import sys
4
+ from signal import SIG_DFL, SIGPIPE, signal
5
+ from typing import Any, Dict, Generator, List, Optional
6
+
7
+ from flupy import __version__, flu, walk_dirs, walk_files
8
+
9
+
10
+ def read_file(path: str) -> Generator[str, None, None]:
11
+ """Yield lines from a file given its path"""
12
+ with open(path, "r") as f:
13
+ yield from f
14
+
15
+
16
+ def parse_args(args: List[str]) -> argparse.Namespace:
17
+ """Parse input arguments"""
18
+ parser = argparse.ArgumentParser(
19
+ description="flupy: a fluent interface for python collections",
20
+ formatter_class=argparse.RawTextHelpFormatter,
21
+ )
22
+ parser.add_argument("-v", "--version", action="version", version="%(prog)s " + __version__)
23
+ parser.add_argument("command", help="command to execute against input")
24
+ parser.add_argument("-f", "--file", help="path to input file")
25
+ parser.add_argument(
26
+ "-i",
27
+ "--import",
28
+ nargs="*",
29
+ default=[],
30
+ help="modules to import\n"
31
+ "Syntax: <module>:<object>:<alias>\n"
32
+ "Examples:\n"
33
+ "\t'import os' = '-i os'\n"
34
+ "\t'import os as op_sys' = '-i os::op_sys'\n"
35
+ "\t'from os import environ' = '-i os:environ'\n"
36
+ "\t'from os import environ as env' = '-i os:environ:env'\n",
37
+ )
38
+ return parser.parse_args(args)
39
+
40
+
41
+ def build_import_dict(imps: List[str]) -> Dict[str, Any]:
42
+ """Execute CLI scoped imports"""
43
+ import_dict = {}
44
+ for imp_stx in imps:
45
+ module, _, obj_alias = imp_stx.partition(":")
46
+ obj, _, alias = obj_alias.partition(":")
47
+
48
+ if not obj:
49
+ import_dict[alias or module] = importlib.import_module(module)
50
+ else:
51
+ _garb = importlib.import_module(module)
52
+ import_dict[alias or obj] = getattr(_garb, obj)
53
+ return import_dict
54
+
55
+
56
+ def main(argv: Optional[List[str]] = None) -> None:
57
+ """CLI Entrypoint"""
58
+ args = parse_args(argv[1:] if argv is not None else sys.argv[1:])
59
+
60
+ _command = args.command
61
+ _file = args.file
62
+ _import = getattr(args, "import")
63
+
64
+ import_dict = build_import_dict(_import)
65
+
66
+ if _file:
67
+ _ = flu(read_file(_file)).map(str.rstrip)
68
+ else:
69
+ # Do not raise exception for Broken Pipe
70
+ signal(SIGPIPE, SIG_DFL)
71
+ _ = flu(sys.stdin).map(str.rstrip)
72
+
73
+ locals_dict = {
74
+ "flu": flu,
75
+ "_": _,
76
+ "walk_files": walk_files,
77
+ "walk_dirs": walk_dirs,
78
+ }
79
+
80
+ pipeline = eval(_command, import_dict, locals_dict)
81
+
82
+ if hasattr(pipeline, "__iter__") and not isinstance(pipeline, (str, bytes)):
83
+ for r in pipeline:
84
+ sys.stdout.write(str(r) + "\n")
85
+
86
+ elif pipeline is None:
87
+ pass
88
+ else:
89
+ sys.stdout.write(str(pipeline) + "\n")
flupy/cli/utils.py ADDED
@@ -0,0 +1,34 @@
1
+ # pylint: disable=invalid-name
2
+ import os
3
+ from typing import Generator
4
+
5
+ from flupy.fluent import Fluent, flu
6
+
7
+
8
+ def walk_files(*pathes: str, abspath: bool = True) -> "Fluent[str]":
9
+ """Yield files recursively starting from each location in *pathes"""
10
+
11
+ if pathes == ():
12
+ pathes = (".",)
13
+
14
+ def _impl() -> Generator[str, None, None]:
15
+ for path in pathes:
16
+ for d, _, files in os.walk(path):
17
+ for x in files:
18
+ rel_path = os.path.join(d, x)
19
+ if abspath:
20
+ yield os.path.abspath(rel_path)
21
+ else:
22
+ yield rel_path
23
+
24
+ return flu(_impl())
25
+
26
+
27
+ def walk_dirs(path: str = ".") -> "Fluent[str]":
28
+ """Yield files recursively starting from *path"""
29
+
30
+ def _impl() -> Generator[str, None, None]:
31
+ for d, _, _ in os.walk(path):
32
+ yield d
33
+
34
+ return flu(_impl())
flupy/fluent.py ADDED
@@ -0,0 +1,801 @@
1
+ # pylint: disable=invalid-name
2
+ import time
3
+ from collections import defaultdict, deque
4
+ from collections.abc import Iterable as IterableType
5
+ from functools import reduce
6
+ from itertools import dropwhile, groupby, islice, product, takewhile, tee, zip_longest
7
+ from random import sample
8
+ from typing import (
9
+ Any,
10
+ Callable,
11
+ Collection,
12
+ Deque,
13
+ Generator,
14
+ Generic,
15
+ Hashable,
16
+ Iterable,
17
+ Iterator,
18
+ List,
19
+ Optional,
20
+ Set,
21
+ Tuple,
22
+ Type,
23
+ TypeVar,
24
+ Union,
25
+ overload,
26
+ )
27
+
28
+ from typing_extensions import Concatenate, ParamSpec, Protocol
29
+
30
+ __all__ = ["flu"]
31
+
32
+
33
+ T = TypeVar("T")
34
+ T_co = TypeVar("T_co", covariant=True)
35
+ T_contra = TypeVar("T_contra", contravariant=True)
36
+ _T1 = TypeVar("_T1")
37
+ _T2 = TypeVar("_T2")
38
+ _T3 = TypeVar("_T3")
39
+ S = TypeVar("S")
40
+ P = ParamSpec("P")
41
+
42
+ CallableTakesIterable = Callable[[Iterable[T]], Collection[T]]
43
+
44
+
45
+ class SupportsEquality(Protocol):
46
+ def __eq__(self, __other: object) -> bool:
47
+ pass
48
+
49
+
50
+ class SupportsGetItem(Protocol[T_co]):
51
+ def __getitem__(self, __k: Hashable) -> T_co:
52
+ pass
53
+
54
+
55
+ class SupportsIteration(Protocol[T_co]):
56
+ def __iter__(self) -> Iterator[T]:
57
+ pass
58
+
59
+
60
+ class SupportsLessThan(Protocol):
61
+ def __lt__(self, __other: Any) -> bool:
62
+ pass
63
+
64
+
65
+ SupportsLessThanT = TypeVar("SupportsLessThanT", bound="SupportsLessThan")
66
+
67
+
68
+ class Empty:
69
+ pass
70
+
71
+
72
+ def identity(x: T) -> T:
73
+ return x
74
+
75
+
76
+ class Fluent(Generic[T]):
77
+ """A fluent interface to lazy generator functions
78
+
79
+ >>> from flupy import flu
80
+ >>> (
81
+ flu(range(100))
82
+ .map(lambda x: x**2)
83
+ .filter(lambda x: x % 3 == 0)
84
+ .chunk(3)
85
+ .take(2)
86
+ .to_list()
87
+ )
88
+ [[0, 9, 36], [81, 144, 225]]
89
+ """
90
+
91
+ def __init__(self, iterable: Iterable[T]) -> None:
92
+ iterator = iter(iterable)
93
+ self._iterator: Iterator[T] = iterator
94
+
95
+ @overload
96
+ def __getitem__(self, index: int) -> T:
97
+ pass
98
+
99
+ @overload
100
+ def __getitem__(self, index: slice) -> "Fluent[T]":
101
+ pass
102
+
103
+ def __getitem__(self, key: Union[int, slice]) -> Union[T, "Fluent[T]"]:
104
+ if isinstance(key, int) and key >= 0:
105
+ try:
106
+ return next(islice(self._iterator, key, key + 1))
107
+ except StopIteration:
108
+ raise IndexError("flu index out of range")
109
+ elif isinstance(key, slice):
110
+ return flu(islice(self._iterator, key.start, key.stop, key.step))
111
+ else:
112
+ raise TypeError(f"Indices must be non-negative integers or slices, not {type(key).__name__}")
113
+
114
+ ### Summary ###
115
+ def collect(self, n: Optional[int] = None, container_type: CallableTakesIterable[T] = list) -> Collection[T]:
116
+ """Collect items from iterable into a container
117
+
118
+ >>> flu(range(4)).collect()
119
+ [0, 1, 2, 3]
120
+
121
+ >>> flu(range(4)).collect(container_type=set)
122
+ {0, 1, 2, 3}
123
+
124
+ >>> flu(range(4)).collect(n=2)
125
+ [0, 1]
126
+ """
127
+ return container_type(self.take(n))
128
+
129
+ def to_list(self) -> List[T]:
130
+ """Collect items from iterable into a list
131
+
132
+ >>> flu(range(4)).to_list()
133
+ [0, 1, 2, 3]
134
+ """
135
+ return list(self)
136
+
137
+ def sum(self) -> Union[T, int]:
138
+ """Sum of elements in the iterable
139
+
140
+ >>> flu([1,2,3]).sum()
141
+ 6
142
+
143
+ """
144
+ return sum(self) # type: ignore
145
+
146
+ def count(self) -> int:
147
+ """Count of elements in the iterable
148
+
149
+ >>> flu(['a','b','c']).count()
150
+ 3
151
+ """
152
+ return sum(1 for _ in self)
153
+
154
+ def min(self: "Fluent[SupportsLessThanT]") -> SupportsLessThanT:
155
+ """Smallest element in the interable
156
+
157
+ >>> flu([1, 3, 0, 2]).min()
158
+ 0
159
+ """
160
+ return min(self)
161
+
162
+ def max(self: "Fluent[SupportsLessThanT]") -> SupportsLessThanT:
163
+ """Largest element in the interable
164
+
165
+ >>> flu([0, 3, 2, 1]).max()
166
+ 3
167
+ """
168
+ return max(self)
169
+
170
+ def first(self, default: Any = Empty()) -> T:
171
+ """Return the first item of the iterable. Raise IndexError if empty, or return default if provided.
172
+
173
+ >>> flu([0, 1, 2, 3]).first()
174
+ 0
175
+ >>> flu([]).first(default="some_default")
176
+ 'some_default'
177
+ """
178
+ x: Union[Empty, T] = default
179
+ for x in self:
180
+ return x
181
+ if isinstance(x, Empty):
182
+ raise IndexError("Empty iterator")
183
+ return x
184
+
185
+ def last(self, default: Any = Empty()) -> T:
186
+ """Return the last item of the iterble. Raise IndexError if empty or default if provided.
187
+
188
+ >>> flu([0, 1, 2, 3]).last()
189
+ 3
190
+ >>> flu([]).last(default='some_default')
191
+ 'some_default'
192
+ """
193
+ x: Union[Empty, T] = default
194
+ for x in self:
195
+ pass
196
+ if isinstance(x, Empty):
197
+ raise IndexError("Empty iterator")
198
+ return x
199
+
200
+ def head(self, n: int = 10, container_type: CallableTakesIterable[T] = list) -> Collection[T]:
201
+ """Returns up to the first *n* elements from the iterable.
202
+
203
+ >>> flu(range(20)).head()
204
+ [0, 1, 2, 3, 4, 5, 6, 7, 8, 9]
205
+
206
+ >>> flu(range(15)).head(n=2)
207
+ [0, 1]
208
+
209
+ >>> flu([]).head()
210
+ []
211
+ """
212
+ return self.take(n).collect(container_type=container_type)
213
+
214
+ def tail(self, n: int = 10, container_type: CallableTakesIterable[T] = list) -> Collection[T]:
215
+ """Return up to the last *n* elements from the iterable
216
+
217
+ >>> flu(range(20)).tail()
218
+ [10, 11, 12, 13, 14, 15, 16, 17, 18, 19]
219
+
220
+ >>> flu(range(15)).tail(n=2)
221
+ [13, 14]
222
+ """
223
+ val: Union[List[Empty], Tuple[Any, ...]] = [Empty()]
224
+ for val in self.window(n, fill_value=Empty()):
225
+ pass
226
+ return container_type([x for x in val if not isinstance(x, Empty)])
227
+
228
+ ### End Summary ###
229
+
230
+ ### Non-Constant Memory ###
231
+ def sort(
232
+ self: "Fluent[SupportsLessThanT]",
233
+ key: Optional[Callable[[Any], Any]] = None,
234
+ reverse: bool = False,
235
+ ) -> "Fluent[SupportsLessThanT]":
236
+ """Sort iterable by *key* function if provided or identity otherwise
237
+
238
+ Note: sorting loads the entire iterable into memory
239
+
240
+ >>> flu([3,6,1]).sort().to_list()
241
+ [1, 3, 6]
242
+
243
+ >>> flu([3,6,1]).sort(reverse=True).to_list()
244
+ [6, 3, 1]
245
+
246
+ >>> flu([3,-6,1]).sort(key=abs).to_list()
247
+ [1, 3, -6]
248
+ """
249
+ return Fluent(sorted(self, key=key, reverse=reverse))
250
+
251
+ def join_left(
252
+ self,
253
+ other: Iterable[_T1],
254
+ key: Callable[[T], Hashable] = identity,
255
+ other_key: Callable[[_T1], Hashable] = identity,
256
+ ) -> "Fluent[Tuple[T, Union[_T1, None]]]":
257
+ """Join the iterable with another iterable using equality between *key* applied to self and *other_key* applied to *other* to identify matching entries
258
+
259
+ When no matching entry is found in *other*, entries in the iterable are paired with None
260
+
261
+ Note: join_left loads *other* into memory
262
+
263
+ >>> flu(range(6)).join_left(range(0, 6, 2)).to_list()
264
+ [(0, 0), (1, None), (2, 2), (3, None), (4, 4), (5, None)]
265
+ """
266
+
267
+ def _impl() -> Generator[Tuple[T, Union[_T1, None]], None, None]:
268
+
269
+ other_lookup = defaultdict(list)
270
+
271
+ for entry_other in other:
272
+ other_lookup[other_key(entry_other)].append(entry_other)
273
+
274
+ for entry in self:
275
+ matches: Optional[List[_T1]] = other_lookup.get(key(entry))
276
+
277
+ if matches:
278
+ for match in matches:
279
+ yield (entry, match)
280
+ else:
281
+ yield (entry, None)
282
+
283
+ return Fluent(_impl())
284
+
285
+ def join_inner(
286
+ self,
287
+ other: Iterable[_T1],
288
+ key: Callable[[T], Hashable] = identity,
289
+ other_key: Callable[[_T1], Hashable] = identity,
290
+ ) -> "Fluent[Tuple[T, _T1]]":
291
+ """Join the iterable with another iterable using equality between *key* applied to self and *other_key* applied to *other* to identify matching entries
292
+
293
+ When no matching entry is found in *other*, entries in the iterable are filtered from the results
294
+
295
+ Note: join_inner loads *other* into memory
296
+
297
+ >>> flu(range(6)).join_inner(range(0, 6, 2)).to_list()
298
+ [(0, 0), (2, 2), (4, 4)]
299
+
300
+ """
301
+
302
+ def _impl() -> Generator[Tuple[T, _T1], None, None]:
303
+
304
+ other_lookup = defaultdict(list)
305
+
306
+ for entry_other in other:
307
+ other_lookup[other_key(entry_other)].append(entry_other)
308
+
309
+ for entry in self:
310
+ matches: List[_T1] = other_lookup[key(entry)]
311
+
312
+ for match in matches:
313
+ yield (entry, match)
314
+
315
+ return Fluent(_impl())
316
+
317
+ def shuffle(self) -> "Fluent[T]":
318
+ """Randomize the order of elements in the interable
319
+
320
+ Note: shuffle loads the entire iterable into memory
321
+
322
+ >>> flu([3,6,1]).shuffle().to_list()
323
+ [6, 1, 3]
324
+ """
325
+ dat: List[T] = self.to_list()
326
+ return Fluent(sample(dat, len(dat)))
327
+
328
+ def group_by(
329
+ self, key: Callable[[T], Union[T, _T1]] = identity, sort: bool = True
330
+ ) -> "Fluent[Tuple[Union[T,_T1], Fluent[T]]]":
331
+ """Yield consecutive keys and groups from the iterable
332
+
333
+ *key* is a function to compute a key value used in grouping and sorting for each element. *key* defaults to an identity function which returns the unchaged element
334
+
335
+ When the iterable is pre-sorted according to *key*, setting *sort* to False will prevent loading the dataset into memory and improve performance
336
+
337
+ >>> flu([2, 4, 2, 4]).group_by().to_list()
338
+ [(2, <flu object>), (4, <flu object>)]
339
+
340
+ Or, if the iterable is pre-sorted
341
+
342
+ >>> flu([2, 2, 5, 5]).group_by(sort=False).to_list()
343
+ [(2, <flu object>), (5, <flu object>)]
344
+
345
+ Using a key function
346
+
347
+ >>> points = [
348
+ {'x': 1, 'y': 0},
349
+ {'x': 4, 'y': 3},
350
+ {'x': 1, 'y': 5}
351
+ ]
352
+ >>> key_func = lambda u: u['x']
353
+ >>> flu(points).group_by(key=key_func, sort=True).to_list()
354
+ [(1, <flu object>), (4, <flu object>)]
355
+ """
356
+
357
+ gen = self.sort(key) if sort else self
358
+ return Fluent(groupby(gen, key)).map(lambda x: (x[0], flu([y for y in x[1]])))
359
+
360
+ def unique(self, key: Callable[[T], Hashable] = identity) -> "Fluent[T]":
361
+ """Yield elements that are unique by a *key*.
362
+
363
+ >>> flu([2, 3, 2, 3]).unique().to_list()
364
+ [2, 3]
365
+
366
+ >>> flu([2, -3, -2, 3]).unique(key=abs).to_list()
367
+ [2, -3]
368
+ """
369
+
370
+ def _impl() -> Generator[T, None, None]:
371
+ seen: Set[Any] = set()
372
+ for x in self:
373
+ x_hash = key(x)
374
+ if x_hash in seen:
375
+ continue
376
+ else:
377
+ seen.add(x_hash)
378
+ yield x
379
+
380
+ return Fluent(_impl())
381
+
382
+ ### End Non-Constant Memory ###
383
+
384
+ ### Side Effect ###
385
+ def rate_limit(self, per_second: Union[int, float] = 100) -> "Fluent[T]":
386
+ """Restrict consumption of iterable to n item *per_second*
387
+
388
+ >>> import time
389
+ >>> start_time = time.time()
390
+ >>> _ = flu(range(3)).rate_limit(3).to_list()
391
+ >>> print('Runtime', int(time.time() - start_time))
392
+ 1.00126 # approximately 1 second for 3 items
393
+ """
394
+
395
+ def _impl() -> Generator[T, None, None]:
396
+ wait_time = 1.0 / per_second
397
+ for val in self:
398
+ start_time = time.time()
399
+ yield val
400
+ call_duration = time.time() - start_time
401
+ time.sleep(max(wait_time - call_duration, 0.0))
402
+
403
+ return Fluent(_impl())
404
+
405
+ def side_effect(
406
+ self,
407
+ func: Callable[[T], Any],
408
+ before: Optional[Callable[[], Any]] = None,
409
+ after: Optional[Callable[[], Any]] = None,
410
+ ) -> "Fluent[T]":
411
+ """Invoke *func* for each item in the iterable before yielding the item.
412
+ *func* takes a single argument and the output is discarded
413
+ *before* and *after* are optional functions that take no parameters and are executed once before iteration begins
414
+ and after iteration ends respectively. Each will be called exactly once.
415
+
416
+
417
+ >>> flu(range(2)).side_effect(lambda x: print(f'Collected {x}')).to_list()
418
+ Collected 0
419
+ Collected 1
420
+ [0, 1]
421
+ """
422
+
423
+ def _impl() -> Generator[T, None, None]:
424
+ try:
425
+ if before is not None:
426
+ before()
427
+
428
+ for x in self:
429
+ func(x)
430
+ yield x
431
+
432
+ finally:
433
+ if after is not None:
434
+ after()
435
+
436
+ return Fluent(_impl())
437
+
438
+ ### End Side Effect ###
439
+
440
+ def map(self, func: Callable[Concatenate[T, P], _T1], *args: Any, **kwargs: Any) -> "Fluent[_T1]":
441
+ """Apply *func* to each element of iterable
442
+
443
+ >>> flu(range(5)).map(lambda x: x*x).to_list()
444
+ [0, 1, 4, 9, 16]
445
+ """
446
+
447
+ def _impl() -> Generator[_T1, None, None]:
448
+ for val in self._iterator:
449
+ yield func(val, *args, **kwargs)
450
+
451
+ return Fluent(_impl())
452
+
453
+ def map_item(self: "Fluent[SupportsGetItem[T]]", item: Hashable) -> "Fluent[T]":
454
+ """Extracts *item* from every element of the iterable
455
+
456
+ >>> flu([(2, 4), (2, 5)]).map_item(1).to_list()
457
+ [4, 5]
458
+
459
+ >>> flu([{'mykey': 8}, {'mykey': 5}]).map_item('mykey').to_list()
460
+ [8, 5]
461
+ """
462
+
463
+ def _impl() -> Generator[T, None, None]:
464
+ for x in self:
465
+ yield x[item]
466
+
467
+ return Fluent(_impl())
468
+
469
+ def map_attr(self, attr: str) -> "Fluent[Any]":
470
+ """Extracts the attribute *attr* from each element of the iterable
471
+
472
+ >>> from collections import namedtuple
473
+ >>> MyTup = namedtuple('MyTup', ['value', 'backup_val'])
474
+ >>> flu([MyTup(1, 5), MyTup(2, 4)]).map_attr('value').to_list()
475
+ [1, 2]
476
+ """
477
+ return self.map(lambda x: getattr(x, attr))
478
+
479
+ def filter(self, func: Callable[Concatenate[T, P], bool], *args: Any, **kwargs: Any) -> "Fluent[T]":
480
+ """Yield elements of iterable where *func* returns truthy
481
+
482
+ >>> flu(range(10)).filter(lambda x: x % 2 == 0).to_list()
483
+ [0, 2, 4, 6, 8]
484
+ """
485
+
486
+ def _impl() -> Generator[T, None, None]:
487
+ for val in self._iterator:
488
+ if func(val, *args, **kwargs):
489
+ yield val
490
+
491
+ return Fluent(_impl())
492
+
493
+ def reduce(self, func: Callable[[T, T], T]) -> T:
494
+ """Apply a function of two arguments cumulatively to the items of the iterable,
495
+ from left to right, so as to reduce the sequence to a single value
496
+
497
+ >>> flu(range(5)).reduce(lambda x, y: x + y)
498
+ 10
499
+ """
500
+ return reduce(func, self)
501
+
502
+ def fold_left(self, func: Callable[[S, T], S], initial: S) -> S:
503
+ """Apply a function of two arguments cumulatively to the items of the iterable,
504
+ from left to right, starting with *initial*, so as to fold the sequence to
505
+ a single value
506
+
507
+ >>> flu(range(5)).fold_left(lambda x, y: x + str(y), "")
508
+ '01234'
509
+ """
510
+ return reduce(func, self, initial)
511
+
512
+ @overload
513
+ def zip(self, __iter1: Iterable[_T1]) -> "Fluent[Tuple[T, _T1]]": ...
514
+
515
+ @overload
516
+ def zip(self, __iter1: Iterable[_T1], __iter2: Iterable[_T2]) -> "Fluent[Tuple[T, _T1, _T2]]": ...
517
+
518
+ @overload
519
+ def zip(
520
+ self, __iter1: Iterable[_T1], __iter2: Iterable[_T2], __iter3: Iterable[_T3]
521
+ ) -> "Fluent[Tuple[T, _T1, _T2, _T3]]": ...
522
+
523
+ @overload
524
+ def zip(
525
+ self,
526
+ __iter1: Iterable[Any],
527
+ __iter2: Iterable[Any],
528
+ __iter3: Iterable[Any],
529
+ __iter4: Iterable[Any],
530
+ *iterable: Iterable[Any],
531
+ ) -> "Fluent[Tuple[T, ...]]": ...
532
+
533
+ def zip(self, *iterable: Iterable[Any]) -> Union[
534
+ "Fluent[Tuple[T, ...]]",
535
+ "Fluent[Tuple[T, _T1]]",
536
+ "Fluent[Tuple[T, _T1, _T2]]",
537
+ "Fluent[Tuple[T, _T1, _T2, _T3]]",
538
+ ]:
539
+ """Yields tuples containing the i-th element from the i-th
540
+ argument in the instance, and the iterable
541
+
542
+ >>> flu(range(5)).zip(range(3, 0, -1)).to_list()
543
+ [(0, 3), (1, 2), (2, 1)]
544
+ """
545
+ # @self_to_flu is not compatible with @overload
546
+ # make sure any usage of self supports arbitrary iterables
547
+ tup_iter = zip(iter(self), *iterable)
548
+ return Fluent(tup_iter)
549
+
550
+ def zip_longest(self, *iterable: Iterable[_T1], fill_value: Any = None) -> "Fluent[Tuple[T, ...]]":
551
+ """Yields tuples containing the i-th element from the i-th
552
+ argument in the instance, and the iterable
553
+ Iteration continues until the longest iterable is exhaused.
554
+ If iterables are uneven in length, missing values are filled in with fill value
555
+
556
+ >>> flu(range(5)).zip_longest(range(3, 0, -1)).to_list()
557
+ [(0, 3), (1, 2), (2, 1), (3, None), (4, None)]
558
+
559
+
560
+ >>> flu(range(5)).zip_longest(range(3, 0, -1), fill_value='a').to_list()
561
+ [(0, 3), (1, 2), (2, 1), (3, 'a'), (4, 'a')]
562
+ """
563
+ return Fluent(zip_longest(self, *iterable, fillvalue=fill_value))
564
+
565
+ def enumerate(self, start: int = 0) -> "Fluent[Tuple[int, T]]":
566
+ """Yields tuples from the instance where the first element
567
+ is a count from initial value *start*.
568
+
569
+ >>> flu([3,4,5]).enumerate().to_list()
570
+ [(0, 3), (1, 4), (2, 5)]
571
+ """
572
+ return Fluent(enumerate(self, start=start))
573
+
574
+ def take(self, n: Optional[int] = None) -> "Fluent[T]":
575
+ """Yield first *n* items of the iterable
576
+
577
+ >>> flu(range(10)).take(2).to_list()
578
+ [0, 1]
579
+ """
580
+ return Fluent(islice(self._iterator, n))
581
+
582
+ def take_while(self, predicate: Callable[[T], bool]) -> "Fluent[T]":
583
+ """Yield elements from the chainable so long as the predicate is true
584
+
585
+ >>> flu(range(10)).take_while(lambda x: x < 3).to_list()
586
+ [0, 1, 2]
587
+ """
588
+ return Fluent(takewhile(predicate, self._iterator))
589
+
590
+ def drop_while(self, predicate: Callable[[T], bool]) -> "Fluent[T]":
591
+ """Drop elements from the chainable as long as the predicate is true;
592
+ afterwards, return every element
593
+
594
+ >>> flu(range(10)).drop_while(lambda x: x < 3).to_list()
595
+ [3, 4, 5, 6, 7, 8, 9]
596
+ """
597
+ return Fluent(dropwhile(predicate, self._iterator))
598
+
599
+ def chunk(self, n: int) -> "Fluent[List[T]]":
600
+ """Yield lists of elements from iterable in groups of *n*
601
+
602
+ if the iterable is not evenly divisiible by *n*, the final list will be shorter
603
+
604
+ >>> flu(range(10)).chunk(3).to_list()
605
+ [[0, 1, 2], [3, 4, 5], [6, 7, 8], [9]]
606
+ """
607
+
608
+ def _impl() -> Generator[List[T], None, None]:
609
+
610
+ while True:
611
+ vals: List[T] = list(self.take(n))
612
+ if vals:
613
+ yield vals
614
+ else:
615
+ return
616
+
617
+ return Fluent(_impl())
618
+
619
+ def flatten(
620
+ self,
621
+ depth: int = 1,
622
+ base_type: Optional[Type[object]] = None,
623
+ iterate_strings: bool = False,
624
+ ) -> "Fluent[Any]":
625
+ """Recursively flatten nested iterables (e.g., a list of lists of tuples)
626
+ into non-iterable type or an optional user-defined base_type
627
+
628
+ Strings are treated as non-iterable for convenience. set iterate_string=True
629
+ to change that behavior.
630
+
631
+ >>> flu([[0, 1, 2], [3, 4, 5]]).flatten().to_list()
632
+ [0, 1, 2, 3, 4, 5]
633
+
634
+ >>> flu([[0, [1, 2]], [[3, 4], 5]]).flatten().to_list()
635
+ [0, [1, 2], [3, 4], 5]
636
+
637
+ >>> flu([[0, [1, 2]], [[3, 4], 5]]).flatten(depth=2).to_list()
638
+ [0, 1, 2, 3, 4, 5]
639
+
640
+ >>> flu([[0, [1, 2]], [[3, 4], 5]]).flatten(depth=2).to_list()
641
+ [0, 1, 2, 3, 4, 5]
642
+
643
+ >>> flu([1, (2, 2), 4, [5, (6, 6, 6)]]).flatten(base_type=tuple).to_list()
644
+ [1, (2, 2), 4, 5, (6, 6, 6)]
645
+
646
+ >>> flu([[2, 0], 'abc', 3, [4]]).flatten(iterate_strings=True).to_list()
647
+ [2, 0, 'a', 'b', 'c', 3, 4]
648
+ """
649
+
650
+ # TODO(OR): Reimplement with strong types
651
+ def walk(node: Any, level: int) -> Generator[T, None, None]:
652
+ if (
653
+ ((depth is not None) and (level > depth))
654
+ or (isinstance(node, str) and not iterate_strings)
655
+ or ((base_type is not None) and isinstance(node, base_type))
656
+ ):
657
+ yield node
658
+ return
659
+ try:
660
+ tree = iter(node)
661
+ except TypeError:
662
+ yield node
663
+ return
664
+ else:
665
+ for child in tree:
666
+ for val in walk(child, level + 1):
667
+ yield val
668
+
669
+ return Fluent(walk(self, level=0))
670
+
671
+ def denormalize(self: "Fluent[SupportsIteration[Any]]", iterate_strings: bool = False) -> "Fluent[Tuple[Any, ...]]":
672
+ """Denormalize iterable components of each record
673
+
674
+ >>> flu([("abc", [1, 2, 3])]).denormalize().to_list()
675
+ [('abc', 1), ('abc', 2), ('abc', 3)]
676
+
677
+ >>> flu([("abc", [1, 2])]).denormalize(iterate_strings=True).to_list()
678
+ [('a', 1), ('a', 2), ('b', 1), ('b', 2), ('c', 1), ('c', 2)]
679
+
680
+ >>> flu([("abc", [])]).denormalize().to_list()
681
+ []
682
+ """
683
+
684
+ def _impl() -> Generator[Tuple[Any, ...], None, None]:
685
+ for record in self:
686
+ iter_elements: List[Iterable[Any]] = []
687
+ element: Any
688
+ for element in record:
689
+
690
+ # Check for string and string iteration is allowed
691
+ if isinstance(element, str) and iterate_strings:
692
+ iter_elements.append(element)
693
+
694
+ # Check for string and string iteration is not allowed
695
+ elif isinstance(element, str):
696
+ iter_elements.append([element])
697
+
698
+ # Check for iterable
699
+ elif isinstance(element, IterableType):
700
+ iter_elements.append(element)
701
+
702
+ # Check for non-iterable
703
+ else:
704
+ iter_elements.append([element])
705
+
706
+ for row in product(*iter_elements):
707
+ yield row
708
+
709
+ return Fluent(_impl())
710
+
711
+ def window(self, n: int, step: int = 1, fill_value: Any = None) -> "Fluent[Tuple[Any, ...]]":
712
+ """Yield a sliding window of width *n* over the given iterable.
713
+
714
+ Each window will advance in increments of *step*:
715
+
716
+ If the length of the iterable does not evenly divide by the *step*
717
+ the final output is padded with *fill_value*
718
+
719
+ >>> flu(range(5)).window(3).to_list()
720
+ [(0, 1, 2), (1, 2, 3), (2, 3, 4)]
721
+
722
+ >>> flu(range(5)).window(n=3, step=2).to_list()
723
+ [(0, 1, 2), (2, 3, 4)]
724
+
725
+ >>> flu(range(9)).window(n=4, step=3).to_list()
726
+ [(0, 1, 2, 3), (3, 4, 5, 6), (6, 7, 8, None)]
727
+
728
+ >>> flu(range(9)).window(n=4, step=3, fill_value=-1).to_list()
729
+ [(0, 1, 2, 3), (3, 4, 5, 6), (6, 7, 8, -1)]
730
+ """
731
+
732
+ def _impl() -> Generator[Tuple[Any, ...], None, None]:
733
+ if n < 0:
734
+ raise ValueError("n must be >= 0")
735
+ elif n == 0:
736
+ yield tuple()
737
+ return
738
+ if step < 1:
739
+ raise ValueError("step must be >= 1")
740
+
741
+ window: Deque[Any] = deque([], n)
742
+ append = window.append
743
+
744
+ # Initial deque fill
745
+ for _ in range(n):
746
+ append(next(self, fill_value))
747
+ yield tuple(window)
748
+
749
+ # Appending new items to the right causes old items to fall off the left
750
+ i = 0
751
+ for item in self:
752
+ append(item)
753
+ i = (i + 1) % step
754
+ if i % step == 0:
755
+ yield tuple(window)
756
+
757
+ # If there are items from the iterable in the window, pad with the given
758
+ # value and emit them.
759
+ if (i % step) and (step - i < n):
760
+ for _ in range(step - i):
761
+ append(fill_value)
762
+ yield tuple(window)
763
+
764
+ return Fluent(_impl())
765
+
766
+ def __iter__(self) -> "Fluent[T]":
767
+ return self
768
+
769
+ def __next__(self) -> T:
770
+ return next(self._iterator)
771
+
772
+ def tee(self, n: int = 2) -> "Fluent[Fluent[T]]":
773
+ """Return n independent iterators from a single iterable
774
+
775
+ once tee() has made a split, the original iterable should not be used
776
+ anywhere else; otherwise, the iterable could get advanced without the
777
+ tee objects being informed
778
+
779
+ >>> copy1, copy2 = flu(range(5)).tee()
780
+ >>> copy1.sum()
781
+ 10
782
+ >>> copy2.to_list()
783
+ [0, 1, 2, 3, 4]
784
+ """
785
+ return Fluent((Fluent(x) for x in tee(self, n)))
786
+
787
+
788
+ class flu(Fluent[T]):
789
+ """A fluent interface to lazy generator functions
790
+
791
+ >>> from flupy import flu
792
+ >>> (
793
+ flu(range(100))
794
+ .map(lambda x: x**2)
795
+ .filter(lambda x: x % 3 == 0)
796
+ .chunk(3)
797
+ .take(2)
798
+ .to_list()
799
+ )
800
+ [[0, 9, 36], [81, 144, 225]]
801
+ """
flupy/py.typed ADDED
File without changes
@@ -0,0 +1,23 @@
1
+ # License
2
+
3
+ **The MIT License (MIT)**
4
+
5
+ Copyright &copy; 2017, Oliver Rice
6
+
7
+ Permission is hereby granted, free of charge, to any person obtaining a copy
8
+ of this software and associated documentation files (the "Software"), to deal
9
+ in the Software without restriction, including without limitation the rights
10
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
11
+ copies of the Software, and to permit persons to whom the Software is
12
+ furnished to do so, subject to the following conditions:
13
+
14
+ The above copyright notice and this permission notice shall be included in
15
+ all copies or substantial portions of the Software.
16
+
17
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
18
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
19
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
20
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
21
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
22
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
23
+ THE SOFTWARE.
@@ -0,0 +1,114 @@
1
+ Metadata-Version: 2.3
2
+ Name: flupy
3
+ Version: 1.2.2
4
+ Summary: Fluent data processing in Python - a chainable stream processing library for expressive data manipulation using method chaining
5
+ License: MIT
6
+ Author: Oliver Rice
7
+ Author-email: oliver@oliverrice.com
8
+ Requires-Python: >=3.9
9
+ Classifier: Development Status :: 4 - Beta
10
+ Classifier: License :: OSI Approved :: MIT License
11
+ Classifier: Natural Language :: English
12
+ Classifier: Operating System :: OS Independent
13
+ Classifier: Programming Language :: Python
14
+ Classifier: Programming Language :: Python :: 3
15
+ Classifier: Programming Language :: Python :: 3.9
16
+ Classifier: Programming Language :: Python :: 3.10
17
+ Classifier: Programming Language :: Python :: 3.11
18
+ Classifier: Programming Language :: Python :: 3.12
19
+ Classifier: Programming Language :: Python :: 3.13
20
+ Requires-Dist: typing_extensions (>=4)
21
+ Project-URL: Repository, https://github.com/olirice/flupy
22
+ Description-Content-Type: text/markdown
23
+
24
+ # flupy
25
+
26
+ <p>
27
+
28
+ <a href="https://flupy.readthedocs.io/en/latest/?badge=latest"><img src="https://readthedocs.org/projects/flupy/badge/?version=latest" alt="Tests" height="18"></a>
29
+ <a href="https://codecov.io/gh/olirice/flupy"><img src="https://codecov.io/gh/olirice/flupy/branch/master/graph/badge.svg" height="18"></a>
30
+ <a href="https://github.com/psf/black">
31
+ <img src="https://img.shields.io/badge/code%20style-black-000000.svg" alt="Codestyle Black" height="18">
32
+ </a>
33
+ </p>
34
+
35
+ <p>
36
+ <a href="https://www.python.org/downloads/"><img src="https://img.shields.io/badge/python-3.6+-blue.svg" alt="Python version" height="18"></a>
37
+ <a href="https://badge.fury.io/py/flupy"><img src="https://badge.fury.io/py/flupy.svg" alt="PyPI version" height="18"></a>
38
+ <a href="https://github.com/olirice/flupy/blob/master/LICENSE"><img src="https://img.shields.io/pypi/l/markdown-subtemplate.svg" alt="License" height="18"></a>
39
+ <a href="https://pypi.org/project/flupy/"><img src="https://img.shields.io/pypi/dm/flupy.svg" alt="Download count" height="18"></a>
40
+ </p>
41
+
42
+ ---
43
+
44
+ **Documentation**: <a href="https://flupy.readthedocs.io/en/latest/" target="_blank">https://flupy.readthedocs.io/en/latest/</a>
45
+
46
+ **Source Code**: <a href="https://github.com/olirice/flupy" target="_blank">https://github.com/olirice/flupy</a>
47
+
48
+ ---
49
+
50
+ ## Overview
51
+ Flupy implements a [fluent interface](https://en.wikipedia.org/wiki/Fluent_interface) for operating on python iterables. All flupy methods return generators and are evaluated lazily. This allows expressions to transform arbitrary size data in extremely limited memory.
52
+
53
+ You can think of flupy as a light weight, 0 dependency, pure python alternative to the excellent [Apache Spark](https://spark.apache.org/) project.
54
+
55
+ ## Setup
56
+
57
+ ### Requirements
58
+
59
+ * Python 3.6+
60
+
61
+ ### Installation
62
+
63
+ Install flupy with pip:
64
+ ```sh
65
+ $ pip install flupy
66
+ ```
67
+
68
+ ### Library
69
+ ```python
70
+ from itertools import count
71
+ from flupy import flu
72
+
73
+ # Processing an infinite sequence in constant memory
74
+ pipeline = (
75
+ flu(count())
76
+ .map(lambda x: x**2)
77
+ .filter(lambda x: x % 517 == 0)
78
+ .chunk(5)
79
+ .take(3)
80
+ )
81
+
82
+ for item in pipeline:
83
+ print(item)
84
+
85
+ # Returns:
86
+ # [0, 267289, 1069156, 2405601, 4276624]
87
+ # [6682225, 9622404, 13097161, 17106496, 21650409]
88
+ # [26728900, 32341969, 38489616, 45171841, 52388644]
89
+ ```
90
+
91
+ ### CLI
92
+ The flupy command line interface brings the same syntax for lazy piplines to your shell. Inputs to the `flu` command are auto-populated into a `Fluent` context named `_`.
93
+ ````
94
+ $ flu -h
95
+ usage: flu [-h] [-f FILE] [-i [IMPORT [IMPORT ...]]] command
96
+
97
+ flupy: a fluent interface for python
98
+
99
+ positional arguments:
100
+ command flupy command to execute on input
101
+
102
+ optional arguments:
103
+ -h, --help show this help message and exit
104
+ -f FILE, --file FILE path to input file
105
+ -i [IMPORT [IMPORT ...]], --import [IMPORT [IMPORT ...]]
106
+ modules to import
107
+ Syntax: <module>:<object>:<alias>
108
+ Examples:
109
+ 'import os' = '-i os'
110
+ 'import os as op_sys' = '-i os::op_sys'
111
+ 'from os import environ' = '-i os:environ'
112
+ 'from os import environ as env' = '-i os:environ:env'
113
+ ````
114
+
@@ -0,0 +1,11 @@
1
+ flupy/__init__.py,sha256=q1_r7-VgAm6qDCAP5prkxy604N7wxZloIHXQE-XuHxQ,223
2
+ flupy/cli/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
3
+ flupy/cli/cli.py,sha256=G_bVsJh6BuchnBor2Jml_4rAIr9etHpDpMazXp5JxY0,2731
4
+ flupy/cli/utils.py,sha256=JZdikTjCM8pJsePd8aaabdNcKOqwBLEvCKX-m_Jj8d4,916
5
+ flupy/fluent.py,sha256=vIfOTKLgxV3DVBHZ57ECheDdbo19YeYY_N9jEEd-Fic,25158
6
+ flupy/py.typed,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
7
+ flupy-1.2.2.dist-info/LICENSE.md,sha256=PGi8-kwuKJBXdNHfDwSYRQkfnRZS6kTmbPiJcEfhVlM,1097
8
+ flupy-1.2.2.dist-info/METADATA,sha256=tq0PImF-TGnygtcIj2VFBe42Pw0fEuflWMFTyUFawcc,4235
9
+ flupy-1.2.2.dist-info/WHEEL,sha256=fGIA9gx4Qxk2KDKeNJCbOEwSrmLtjWCwzBz351GyrPQ,88
10
+ flupy-1.2.2.dist-info/entry_points.txt,sha256=jJkgvssRM-Hi1Az1gh9zznv4wOYa1YEWubX5Ksgg0Ks,80
11
+ flupy-1.2.2.dist-info/RECORD,,
@@ -0,0 +1,4 @@
1
+ Wheel-Version: 1.0
2
+ Generator: poetry-core 2.1.2
3
+ Root-Is-Purelib: true
4
+ Tag: py3-none-any
@@ -0,0 +1,4 @@
1
+ [console_scripts]
2
+ flu=flupy.cli.cli:main
3
+ flu_precommit=flupy.cli.cli:precommit
4
+