lazysort 1.0.0.dev0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
lazysort/__init__.py ADDED
@@ -0,0 +1,9 @@
1
+ """
2
+ >>> from lazysort import LazySort
3
+ >>> ls = LazySort([1, 44, 32, 1, 434, 22, 11, 8, 66, 10], work_objs_limit=5)
4
+ >>> ls[4]
5
+ 11
6
+ """
7
+
8
+ from .main import LazySort as LazySort
9
+ from .clusters import Cluster as Cluster
lazysort/clusters.py ADDED
@@ -0,0 +1,167 @@
1
+ from collections.abc import Iterable, Iterator
2
+ from typing import TypeVar, Generic
3
+ import dataclasses
4
+
5
+ _V = TypeVar('_V')
6
+
7
+
8
+ @dataclasses.dataclass
9
+ class Cluster(Generic[_V]):
10
+ """
11
+ A cluster of values.
12
+
13
+ :param val_min: Minimum value in the cluster.
14
+ :param val_max: Maximum value in the cluster.
15
+ :param src_idx_from: The index at which the first value of the cluster is located in a source sequence.
16
+ :param src_idx_to: The index at which the last value of the cluster is located in a source sequence.
17
+ :param idx: The index at which the cluster begins in a sorted sequence.
18
+ :param size: Number of values in the cluster.
19
+ :param next: The next cluster in the chain.
20
+ """
21
+ val_min: _V
22
+ val_max: _V
23
+ src_idx_from: int
24
+ src_idx_to: int
25
+ idx: int = -1
26
+ size: int = 1
27
+ next: 'Cluster[_V] | None' = dataclasses.field(default=None, compare=False, repr=False)
28
+
29
+ @classmethod
30
+ def new(cls, val: _V, src_idx: int = 0) -> 'Cluster[_V]':
31
+ """
32
+ Create a new cluster of values.
33
+
34
+ :param val: A value the cluster will contain.
35
+ :param src_idx: The index at which the value is located in a source sequence.
36
+ :return: A new cluster.
37
+ """
38
+ return cls(val_min=val, val_max=val, src_idx_from=src_idx, src_idx_to=src_idx + 1)
39
+
40
+ def add_value(self, val: _V, src_idx: int) -> None:
41
+ """
42
+ Add a new value to the cluster.
43
+
44
+ :param val: A new value the cluster will contain.
45
+ :param src_idx: The index at which the value is located in a source sequence.
46
+ """
47
+ self.size += 1
48
+ if val < self.val_min:
49
+ self.val_min = val
50
+ elif val > self.val_max:
51
+ self.val_max = val
52
+ if src_idx + 1 > self.src_idx_to:
53
+ self.src_idx_to = src_idx + 1
54
+ elif src_idx < self.src_idx_from:
55
+ self.src_idx_from = src_idx
56
+
57
+ def is_value_contained(self, val: _V) -> bool:
58
+ """
59
+ Return True if the cluster contains a value.
60
+
61
+ :param val: A value.
62
+ """
63
+ return self.val_min <= val <= self.val_max
64
+
65
+ def is_homogeneous(self) -> bool:
66
+ """
67
+ Return True if the cluster contains only identical values.
68
+ """
69
+ return self.val_min == self.val_max
70
+
71
+ def find_by_index(self, idx: int) -> 'Cluster[_V]':
72
+ """
73
+ Find a cluster in the chain by idx attribute.
74
+
75
+ :param idx: An index of an item in a sorted sequence.
76
+ """
77
+ cluster: Cluster[_V] = self
78
+ while not cluster.is_index_contained(idx):
79
+ if cluster.next is None:
80
+ raise ValueError(f'A cluster with the item index "{idx}" not found.')
81
+ cluster = cluster.next
82
+ return cluster
83
+
84
+ def is_index_contained(self, idx: int) -> bool:
85
+ """
86
+ Return True if the cluster contains an item with the given index.
87
+
88
+ :param idx: An index of an item in a sorted sequence.
89
+ """
90
+ return self.idx <= idx < self.idx + self.size
91
+
92
+ def last(self) -> 'Cluster[_V]':
93
+ """
94
+ Return the last cluster in the chain.
95
+ """
96
+ cluster: Cluster[_V] = self
97
+ while cluster.next is not None:
98
+ cluster = cluster.next
99
+ return cluster
100
+
101
+ def join_next(self) -> None:
102
+ """
103
+ Join the current cluster with the next one.
104
+ """
105
+ if self.next is None:
106
+ raise ValueError('No next cluster to join.')
107
+ self.size += self.next.size
108
+ self.val_max = max(self.val_max, self.next.val_max)
109
+ self.val_min = min(self.val_min, self.next.val_min)
110
+ self.src_idx_from = min(self.src_idx_from, self.next.src_idx_from)
111
+ self.src_idx_to = max(self.src_idx_to, self.next.src_idx_to)
112
+ self.next = self.next.next
113
+
114
+ def replace(self, old: 'Cluster[_V]', new: 'Cluster[_V]') -> 'Cluster[_V]':
115
+ """
116
+ Replace the old cluster with the new one.
117
+
118
+ :param old: The cluster that will be replaced.
119
+ :param new: The cluster that will replace the old one.
120
+ :return: A head of the new cluster chain.
121
+ """
122
+ if self is old:
123
+ if self.next is not None:
124
+ new.last().next = self.next
125
+ return new
126
+ cluster: Cluster[_V] = self
127
+ while cluster.next is not None:
128
+ if cluster.next is old:
129
+ cluster.next = new
130
+ if old.next is not None:
131
+ new.last().next = old.next
132
+ return self
133
+ cluster = cluster.next
134
+ raise LookupError('Cluster not found.')
135
+
136
+ def __iter__(self) -> Iterator['Cluster']:
137
+ cluster: Cluster[_V] = self
138
+ yield cluster
139
+ while cluster.next is not None:
140
+ cluster = cluster.next
141
+ yield cluster
142
+
143
+ def __len__(self) -> int:
144
+ length: int = 1
145
+ cluster: Cluster[_V] = self
146
+ while cluster.next is not None:
147
+ length += 1
148
+ cluster = cluster.next
149
+ return length
150
+
151
+
152
+ def bond_all(clusters: Iterable[Cluster[_V]]) -> Cluster[_V]:
153
+ """
154
+ Bond a list of clusters.
155
+
156
+ :param clusters: List of clusters.
157
+ :return: Chain of clusters.
158
+ """
159
+ if not clusters:
160
+ raise ValueError('No clusters to bond.')
161
+ cluster_iter: Iterator[Cluster[_V]] = iter(clusters)
162
+ cluster: Cluster[_V] = next(cluster_iter)
163
+ head = cluster
164
+ for next_group in cluster_iter:
165
+ cluster.next = next_group
166
+ cluster = next_group
167
+ return head
lazysort/main.py ADDED
@@ -0,0 +1,244 @@
1
+ from collections.abc import Iterable, Callable, Iterator
2
+ from typing import TypeVar, Generic, overload
3
+
4
+ from .clusters import Cluster, bond_all
5
+ from .protocols import RandomAccessSource, SortFunction
6
+
7
+ _T = TypeVar('_T')
8
+ _V = TypeVar('_V')
9
+
10
+
11
+ class LazySort(Generic[_T, _V]):
12
+ clusters: Cluster[_V]
13
+ work_cluster: Cluster[_V]
14
+ work_items: tuple[_T, ...]
15
+ _seq: RandomAccessSource[_T]
16
+ _seq_length: int
17
+ _clusters_limit: int
18
+ _key_fn: Callable[[_T], _V] | None
19
+ _sort_fn: SortFunction[_T, _V]
20
+ _reverse: bool
21
+ _work_items_limit: int
22
+
23
+ def __init__(
24
+ self,
25
+ seq: RandomAccessSource[_T],
26
+ work_items_limit: int,
27
+ clusters_limit: int | None = None,
28
+ key_fn: Callable[[_T], _V] | None = None,
29
+ sort_fn: SortFunction[_T, _V] = sorted,
30
+ reverse: bool = False,
31
+ ):
32
+ """
33
+ Splits a sequence into parts as needed and sorts them.
34
+
35
+ sort_fn must take arguments: iterable_obj as Iterable, key as Callable, reverse as bool.
36
+
37
+ :param seq: Sequence of items.
38
+ :param work_items_limit: Limit of items loaded into memory for sorting.
39
+ :param clusters_limit: Limit of clusters. It must be greater than 5.
40
+ :param key_fn: A function that takes an item and returns the value to sort by.
41
+ :param sort_fn: A sorting function.
42
+ :param reverse: Sort in reverse order. Passed to the sort function as is.
43
+ """
44
+ self._seq = seq
45
+ self._seq_length = len(seq)
46
+ self.work_items = tuple()
47
+ self._work_items_limit = work_items_limit
48
+ if clusters_limit is None:
49
+ self._clusters_limit = max(len(seq) // (work_items_limit // 2), 6)
50
+ else:
51
+ if clusters_limit < 6:
52
+ raise ValueError('Clusters limit must be greater than 5.')
53
+ self._clusters_limit = clusters_limit
54
+ self._key_fn = key_fn
55
+ self._sort_fn = sort_fn
56
+ self._reverse = reverse
57
+ if self._key_fn is None:
58
+ values: Iterable[_V] = self._seq
59
+ else:
60
+ values: Iterable[_V] = map(self._key_fn, self._seq)
61
+ self.clusters = self._determine_clusters(values=values, limit=self._clusters_limit, reverse=self._reverse)
62
+
63
+ def __iter__(self) -> Iterator[_T]:
64
+ for idx in range(self._seq_length):
65
+ yield self[idx]
66
+
67
+ def __len__(self) -> int:
68
+ return self._seq_length
69
+
70
+ def __repr__(self) -> str:
71
+ return (
72
+ f'<{self.__class__.__name__}'
73
+ f' id: {id(self)},'
74
+ f' items: {len(self.work_items)}/{self._seq_length},'
75
+ f' clusters: {len(self.clusters)}/{self._clusters_limit}>'
76
+ )
77
+
78
+ @overload
79
+ def __getitem__(self, item: int, /) -> _T: ...
80
+
81
+ @overload
82
+ def __getitem__(self, item: slice, /) -> list[_T]: ...
83
+
84
+ def __getitem__(self, item: int|slice, /) -> _T | list[_T]:
85
+ if isinstance(item, int):
86
+ return self._get_item(item)
87
+ return [self._get_item(x) for x in range(item.start or 0, item.stop, item.step or 1)]
88
+
89
+ def _get_item(self, idx: int) -> _T:
90
+ """
91
+ Return an item by index.
92
+ """
93
+ if not (-self._seq_length <= idx < self._seq_length):
94
+ raise IndexError('Index is out of range.')
95
+ if idx < 0:
96
+ idx += self._seq_length
97
+
98
+ if not (self.work_items and self.work_cluster.is_index_contained(idx)):
99
+ self.work_cluster = self._get_work_cluster(idx)
100
+ self.work_items = self._get_sorted_items(self.work_cluster)
101
+ if self.work_cluster.is_homogeneous():
102
+ return self.work_items[0]
103
+ return self.work_items[idx - self.work_cluster.idx]
104
+
105
+ def _get_work_cluster(self, idx: int) -> Cluster[_V]:
106
+ """
107
+ Find a cluster which contains an item with the defined index and split it if necessary.
108
+ """
109
+ cluster: Cluster[_V] = self.clusters.find_by_index(idx)
110
+ while cluster.size > self._work_items_limit:
111
+ if cluster.is_homogeneous():
112
+ break
113
+ clusters_should_be_saved: int = max(
114
+ self._clusters_limit // 2,
115
+ self._clusters_limit - cluster.size // (self._work_items_limit // 2) - 1
116
+ )
117
+ self.clusters = self._shorten_the_chain_of_clusters(
118
+ clusters=self.clusters,
119
+ protected=cluster,
120
+ max_cluster_size=self._work_items_limit,
121
+ max_clusters=clusters_should_be_saved
122
+ )
123
+ if self._key_fn is None:
124
+ cluster_values: Iterable[_V] = filter(cluster.is_value_contained, self._seq)
125
+ else:
126
+ cluster_values: Iterable[_V] = filter(
127
+ cluster.is_value_contained,
128
+ map(self._key_fn, self._seq[cluster.src_idx_from:cluster.src_idx_to])
129
+ )
130
+ new_clusters: Cluster[_V] = self._determine_clusters(
131
+ values=cluster_values,
132
+ limit=self._clusters_limit - len(self.clusters) + 1,
133
+ reverse=self._reverse
134
+ )
135
+ for new_cluster in new_clusters:
136
+ new_cluster.src_idx_from += cluster.src_idx_from
137
+ new_cluster.src_idx_to = cluster.src_idx_to
138
+ new_cluster.idx += cluster.idx
139
+ self.clusters = self.clusters.replace(cluster, new_clusters)
140
+ cluster = new_clusters.find_by_index(idx)
141
+ return cluster
142
+
143
+ def _get_sorted_items(self, cluster: Cluster[_V]) -> tuple[_T, ...]:
144
+ """
145
+ Sort items that contained by cluster.
146
+ """
147
+ if cluster.is_homogeneous():
148
+ return (self._seq[cluster.src_idx_from],)
149
+
150
+ if self._key_fn is None:
151
+ cluster_items: Iterable[_T] = filter(
152
+ cluster.is_value_contained,
153
+ self._seq[cluster.src_idx_from:cluster.src_idx_to]
154
+ )
155
+ else:
156
+ cluster_items: Iterable[_T] = filter(
157
+ lambda x: cluster.is_value_contained(self._key_fn(x)),
158
+ self._seq[cluster.src_idx_from:cluster.src_idx_to]
159
+ )
160
+ return tuple(self._sort_fn(cluster_items, key=self._key_fn, reverse=self._reverse))
161
+
162
+ @staticmethod
163
+ def _shorten_the_chain_of_clusters(
164
+ clusters: Cluster[_V], protected: Cluster[_V], max_cluster_size: int, max_clusters: int
165
+ ) -> Cluster[_V]:
166
+ """
167
+ Reduce the number of clusters.
168
+ """
169
+ # Join all pairs of clusters whose total size is less than max_cluster_size.
170
+ # The reclustering process is resource-intensive. Therefore, it's more efficient to merge clusters while their
171
+ # size is less than max_cluster_size, even if the number of clusters becomes smaller than required.
172
+ cluster: Cluster[_V] = clusters
173
+ while cluster.next is not None:
174
+ if cluster is not protected:
175
+ while (
176
+ cluster.next is not protected
177
+ and cluster.next is not None
178
+ and cluster.size + cluster.next.size <= max_cluster_size
179
+ ):
180
+ cluster.join_next()
181
+ if cluster.next is not None:
182
+ cluster = cluster.next
183
+
184
+ # Join pairs of clusters whose total size the smallest.
185
+ for _ in range(len(clusters) - max_clusters):
186
+ pairs: Iterable[Cluster[_V]] = filter(
187
+ lambda x:not (x.next is None or x is protected or x.next is protected),
188
+ clusters
189
+ )
190
+ cluster: Cluster[_V] = min(pairs, key=lambda x: x.size + x.next.size)
191
+ cluster.join_next()
192
+ return clusters
193
+
194
+ @staticmethod
195
+ def _determine_clusters(values: Iterable[_V], limit: int, reverse: bool) -> Cluster[_V]:
196
+ """
197
+ Analyze the dataset and identify clusters of values.
198
+ """
199
+ data_iter: Iterator[_V] = iter(values)
200
+ clusters: list[Cluster[_V]] = [Cluster.new(next(data_iter))]
201
+ for src_idx, value in enumerate(data_iter, 1):
202
+ left: int = 0
203
+ right: int = len(clusters) - 1
204
+ cursor: int = right // 2
205
+ while left != right:
206
+ cluster = clusters[cursor]
207
+ if cluster.val_max < value:
208
+ left = cursor + 1
209
+ elif value < cluster.val_min:
210
+ right = max(left, cursor - 1)
211
+ else:
212
+ break
213
+ cursor = left + (right - left) // 2
214
+ cluster = clusters[cursor]
215
+
216
+ if value < cluster.val_min:
217
+ if len(clusters) < limit:
218
+ clusters.insert(cursor, Cluster.new(value, src_idx=src_idx))
219
+ elif cursor == 0:
220
+ cluster.add_value(value, src_idx=src_idx)
221
+ elif clusters[cursor - 1].size <= cluster.size:
222
+ clusters[cursor - 1].add_value(value, src_idx=src_idx)
223
+ else:
224
+ cluster.add_value(value, src_idx=src_idx)
225
+ elif value > cluster.val_max:
226
+ if len(clusters) < limit:
227
+ clusters.insert(cursor + 1, Cluster.new(value, src_idx=src_idx))
228
+ elif cursor == len(clusters) - 1:
229
+ cluster.add_value(value, src_idx=src_idx)
230
+ elif cluster.size <= clusters[cursor + 1].size:
231
+ cluster.add_value(value, src_idx=src_idx)
232
+ else:
233
+ clusters[cursor + 1].add_value(value, src_idx=src_idx)
234
+ else:
235
+ cluster.add_value(value, src_idx=src_idx)
236
+
237
+ if reverse:
238
+ clusters.reverse()
239
+ idx: int = 0
240
+ for cluster in clusters:
241
+ cluster.idx = idx
242
+ idx += cluster.size
243
+ bond_all(clusters)
244
+ return clusters[0]
lazysort/protocols.py ADDED
@@ -0,0 +1,38 @@
1
+ from collections.abc import Iterator, Iterable, Callable
2
+ from typing import Protocol, overload, TypeVar
3
+
4
+ _T_cov = TypeVar('_T_cov', covariant=True)
5
+ _T = TypeVar('_T')
6
+ _V_con = TypeVar('_V_con', contravariant=True)
7
+
8
+
9
+ class RandomAccessSource(Protocol[_T_cov]):
10
+ def __len__(self) -> int: ...
11
+
12
+ def __iter__(self) -> Iterator[_T_cov]: ...
13
+
14
+ @overload
15
+ def __getitem__(self, item: int, /) -> _T_cov: ...
16
+
17
+ @overload
18
+ def __getitem__(self, item: slice, /) -> Iterable[_T_cov]: ...
19
+
20
+ @overload
21
+ def __getitem__(self, item: int | slice, /) -> _T_cov | Iterable[_T_cov]: ...
22
+
23
+
24
+ class SortFunction(Protocol[_T, _V_con]):
25
+ @overload
26
+ def __call__(
27
+ self, iterable: Iterable[_T], /, *, key: Callable[[_T], _V_con], reverse: bool = False
28
+ ) -> Iterable[_T]: ...
29
+
30
+ @overload
31
+ def __call__(
32
+ self, iterable: Iterable[_T], /, *, key: None = None, reverse: bool = False
33
+ ) -> Iterable[_T]: ...
34
+
35
+ @overload
36
+ def __call__(
37
+ self, iterable: Iterable[_T], /, *, key: Callable[[_T], _V_con] | None = None, reverse: bool = False
38
+ ) -> Iterable[_T]: ...
@@ -0,0 +1,8 @@
1
+ Metadata-Version: 2.4
2
+ Name: lazysort
3
+ Version: 1.0.0.dev0
4
+ Summary: Python module for lazy sorting by parts.
5
+ Requires-Python: >=3.10
6
+ Description-Content-Type: text/markdown
7
+
8
+ # LazySort
@@ -0,0 +1,8 @@
1
+ lazysort/__init__.py,sha256=M2gUdHtRpkAmldQm8Be9A4uImQkbu5fKKYMQHqd2ZxU,213
2
+ lazysort/clusters.py,sha256=vA8kOq-gocSjal5W6xBacRGpFh_db_0s3f0m7VZutA8,5493
3
+ lazysort/main.py,sha256=s6xViQOmvIbpYKhPUr96IL-JZaYL1S0uerttrrdXy98,9847
4
+ lazysort/protocols.py,sha256=OcJxekk-lzriweKTQsZrjEJaLkX__yPoMiN5xloZ0QY,1115
5
+ lazysort-1.0.0.dev0.dist-info/METADATA,sha256=T64dzvpYr3EX5Q2b2l0cXzjs5KegEp43y_D88E3XP6w,183
6
+ lazysort-1.0.0.dev0.dist-info/WHEEL,sha256=YVMoNqKzERt-wjUZwJ33xBGAwnFl-4cqbYkTtWa4itE,91
7
+ lazysort-1.0.0.dev0.dist-info/top_level.txt,sha256=_CeyhGl4DF2OzyZIdvxV1CegBnHYad4hxi1VpBmA3y0,9
8
+ lazysort-1.0.0.dev0.dist-info/RECORD,,
@@ -0,0 +1,5 @@
1
+ Wheel-Version: 1.0
2
+ Generator: setuptools (84.0.0)
3
+ Root-Is-Purelib: true
4
+ Tag: py3-none-any
5
+
@@ -0,0 +1 @@
1
+ lazysort