kafi 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- kafi/__init__.py +0 -0
- kafi/addons.py +276 -0
- kafi/chunker.py +63 -0
- kafi/dechunker.py +75 -0
- kafi/deserializer.py +148 -0
- kafi/files.py +85 -0
- kafi/fs/__init__.py +0 -0
- kafi/fs/azureblob/__init__.py +0 -0
- kafi/fs/azureblob/azureblob.py +31 -0
- kafi/fs/azureblob/azureblob_admin.py +96 -0
- kafi/fs/azureblob/azureblob_consumer.py +7 -0
- kafi/fs/azureblob/azureblob_producer.py +11 -0
- kafi/fs/fs.py +68 -0
- kafi/fs/fs_admin.py +415 -0
- kafi/fs/fs_consumer.py +181 -0
- kafi/fs/fs_producer.py +70 -0
- kafi/fs/local/__init__.py +0 -0
- kafi/fs/local/local.py +32 -0
- kafi/fs/local/local_admin.py +73 -0
- kafi/fs/local/local_consumer.py +11 -0
- kafi/fs/local/local_producer.py +12 -0
- kafi/fs/s3/__init__.py +0 -0
- kafi/fs/s3/s3.py +31 -0
- kafi/fs/s3/s3_admin.py +87 -0
- kafi/fs/s3/s3_consumer.py +7 -0
- kafi/fs/s3/s3_producer.py +11 -0
- kafi/functional.py +461 -0
- kafi/helpers.py +441 -0
- kafi/kafi.py +8 -0
- kafi/kafka/__init__.py +0 -0
- kafi/kafka/cluster/__init__.py +0 -0
- kafi/kafka/cluster/cluster.py +44 -0
- kafi/kafka/cluster/cluster_admin.py +656 -0
- kafi/kafka/cluster/cluster_consumer.py +166 -0
- kafi/kafka/cluster/cluster_producer.py +77 -0
- kafi/kafka/kafka.py +120 -0
- kafi/kafka/kafka_admin.py +5 -0
- kafi/kafka/kafka_consumer.py +11 -0
- kafi/kafka/kafka_producer.py +10 -0
- kafi/kafka/restproxy/__init__.py +0 -0
- kafi/kafka/restproxy/restproxy.py +62 -0
- kafi/kafka/restproxy/restproxy_admin.py +421 -0
- kafi/kafka/restproxy/restproxy_consumer.py +212 -0
- kafi/kafka/restproxy/restproxy_producer.py +134 -0
- kafi/pandas.py +46 -0
- kafi/schemaregistry.py +253 -0
- kafi/serializer.py +121 -0
- kafi/shell.py +125 -0
- kafi/storage.py +327 -0
- kafi/storage_admin.py +83 -0
- kafi/storage_consumer.py +213 -0
- kafi/storage_producer.py +101 -0
- kafi/streams/__init__.py +0 -0
- kafi/streams/streams.py +174 -0
- kafi/streams/topologynode.py +731 -0
- kafi-0.1.0.dist-info/METADATA +992 -0
- kafi-0.1.0.dist-info/RECORD +59 -0
- kafi-0.1.0.dist-info/WHEEL +4 -0
- kafi-0.1.0.dist-info/licenses/LICENSE +201 -0
kafi/__init__.py
ADDED
|
File without changes
|
kafi/addons.py
ADDED
|
@@ -0,0 +1,276 @@
|
|
|
1
|
+
import json
|
|
2
|
+
|
|
3
|
+
from kafi.functional import Functional
|
|
4
|
+
from kafi.helpers import copy_kwargs
|
|
5
|
+
|
|
6
|
+
# Constants
|
|
7
|
+
|
|
8
|
+
ALL_MESSAGES = -1
|
|
9
|
+
|
|
10
|
+
#
|
|
11
|
+
|
|
12
|
+
def default_projection_function(message_dict1, message_dict2):
|
|
13
|
+
message_dict = dict(message_dict1)
|
|
14
|
+
message_dict["value"] = message_dict1["value"] | message_dict2["value"]
|
|
15
|
+
return message_dict
|
|
16
|
+
#
|
|
17
|
+
|
|
18
|
+
class AddOns(Functional):
|
|
19
|
+
def compact(self, topic, n=ALL_MESSAGES, **kwargs):
|
|
20
|
+
def foldl_function(acc, message_dict):
|
|
21
|
+
key_hash_int_message_dict_dict = acc
|
|
22
|
+
#
|
|
23
|
+
key = message_dict["key"]
|
|
24
|
+
value = message_dict["value"]
|
|
25
|
+
#
|
|
26
|
+
if key is not None:
|
|
27
|
+
key_hash_int = hash(str(key))
|
|
28
|
+
if value is None:
|
|
29
|
+
if key_hash_int in key_hash_int_message_dict_dict:
|
|
30
|
+
del key_hash_int_message_dict_dict[key_hash_int]
|
|
31
|
+
else:
|
|
32
|
+
key_hash_int_message_dict_dict[key_hash_int] = message_dict
|
|
33
|
+
#
|
|
34
|
+
return key_hash_int_message_dict_dict
|
|
35
|
+
#
|
|
36
|
+
|
|
37
|
+
(key_hash_int_message_dict_dict, _) = self.foldl(topic, foldl_function, {}, n, **kwargs)
|
|
38
|
+
#
|
|
39
|
+
message_dict_list = list(key_hash_int_message_dict_dict.values())
|
|
40
|
+
#
|
|
41
|
+
return message_dict_list
|
|
42
|
+
|
|
43
|
+
def compact_to(self, topic, target_storage, target_topic, n=ALL_MESSAGES, **kwargs):
|
|
44
|
+
source_kwargs = copy_kwargs("source", **kwargs)
|
|
45
|
+
target_kwargs = copy_kwargs("target", **kwargs)
|
|
46
|
+
#
|
|
47
|
+
message_dict_list = self.compact(topic, n, **source_kwargs)
|
|
48
|
+
#
|
|
49
|
+
target_producer = target_storage.producer(target_topic, **target_kwargs)
|
|
50
|
+
key_bytes_list_value_bytes_list_tuple = target_producer.produce_list(message_dict_list, **target_kwargs)
|
|
51
|
+
target_producer.close()
|
|
52
|
+
#
|
|
53
|
+
return key_bytes_list_value_bytes_list_tuple
|
|
54
|
+
|
|
55
|
+
#
|
|
56
|
+
|
|
57
|
+
def join_to(self, source_topic1, source_storage2, source_topic2, target_storage, target_topic, get_key_function1=lambda x: x["key"], get_key_function2=lambda x: x["key"], projection_function=default_projection_function, join="left", n=ALL_MESSAGES, **kwargs):
|
|
58
|
+
join_str = join
|
|
59
|
+
#
|
|
60
|
+
if join_str not in ["inner", "left", "right"]:
|
|
61
|
+
raise Exception("Only \"inner\", \"left\" and \"right\" supported.")
|
|
62
|
+
#
|
|
63
|
+
def zip_foldl_to_function(acc, message_dict1, message_dict2):
|
|
64
|
+
# print(message_dict1["value"])
|
|
65
|
+
# print(message_dict2["value"])
|
|
66
|
+
# print("===")
|
|
67
|
+
(index_dict1, index_dict2) = acc
|
|
68
|
+
#
|
|
69
|
+
key1 = get_key_function1(message_dict1)
|
|
70
|
+
key2 = get_key_function2(message_dict2)
|
|
71
|
+
# DBSP: L join R = deltaL join deltaR + deltaL join R + L join deltaR
|
|
72
|
+
out_message_dict_list = []
|
|
73
|
+
# 1. deltaL join deltaR
|
|
74
|
+
if key1 == key2:
|
|
75
|
+
# Match in deltaL join deltaR.
|
|
76
|
+
out_message_dict_list.append(projection_function(message_dict1, message_dict2))
|
|
77
|
+
else:
|
|
78
|
+
# 2. deltaL join R
|
|
79
|
+
if key1 in index_dict2:
|
|
80
|
+
# Match in deltaL join R.
|
|
81
|
+
out_message_dict_list.append(projection_function(message_dict1, index_dict2[key1]))
|
|
82
|
+
else:
|
|
83
|
+
# Could not find key1 in index_dict2.
|
|
84
|
+
# Only append to the output if the join type is "left"
|
|
85
|
+
if join_str == "left":
|
|
86
|
+
out_message_dict_list.append(message_dict1)
|
|
87
|
+
# 3. L join deltaR
|
|
88
|
+
if key2 in index_dict1:
|
|
89
|
+
# Match in L join deltaR
|
|
90
|
+
out_message_dict_list.append(projection_function(index_dict1[key2], message_dict2))
|
|
91
|
+
else:
|
|
92
|
+
# Could not find key2 in index_dict1.
|
|
93
|
+
# Only append to the output if the join type is "right"
|
|
94
|
+
if join_str == "right":
|
|
95
|
+
out_message_dict_list.append(message_dict2)
|
|
96
|
+
# Depending on the join type, persist:
|
|
97
|
+
# * both sides (inner)
|
|
98
|
+
# * the left side (left)
|
|
99
|
+
# * the right side (right)
|
|
100
|
+
if join_str == "inner":
|
|
101
|
+
index_dict1[key1] = message_dict1
|
|
102
|
+
index_dict2[key2] = message_dict2
|
|
103
|
+
elif join_str == "left":
|
|
104
|
+
index_dict1[key1] = message_dict1
|
|
105
|
+
elif join_str == "right":
|
|
106
|
+
index_dict1[key2] = message_dict2
|
|
107
|
+
#
|
|
108
|
+
return ((index_dict1, index_dict2), list(out_message_dict_list))
|
|
109
|
+
|
|
110
|
+
return self.zip_foldl_to(source_topic1, source_storage2, source_topic2, target_storage, target_topic, zip_foldl_to_function, ({}, {}), n=n, **kwargs)
|
|
111
|
+
|
|
112
|
+
#
|
|
113
|
+
|
|
114
|
+
def repeat(self, topic_str, n=1, **kwargs):
|
|
115
|
+
n_int = n
|
|
116
|
+
#
|
|
117
|
+
message_dict_list = self.tail(topic_str, type="bytes", n=n_int, **kwargs)
|
|
118
|
+
pr = self.producer(topic_str, type="bytes", **kwargs)
|
|
119
|
+
pr.produce_list(message_dict_list, **kwargs)
|
|
120
|
+
pr.close()
|
|
121
|
+
#
|
|
122
|
+
return message_dict_list
|
|
123
|
+
|
|
124
|
+
#
|
|
125
|
+
|
|
126
|
+
def recreate(self, pattern, partitions=None, config={}, **kwargs):
|
|
127
|
+
pattern_str_or_str_list = pattern
|
|
128
|
+
#
|
|
129
|
+
topic_str_list = self.admin.list_topics(pattern_str_or_str_list)
|
|
130
|
+
#
|
|
131
|
+
if topic_str_list:
|
|
132
|
+
for topic_str in topic_str_list:
|
|
133
|
+
if partitions is None:
|
|
134
|
+
partitions_int = self.partitions(topic_str)[topic_str]
|
|
135
|
+
else:
|
|
136
|
+
partitions_int = partitions
|
|
137
|
+
#
|
|
138
|
+
old_config_dict = self.config(topic_str)[topic_str]
|
|
139
|
+
config_dict = {}
|
|
140
|
+
for key_str, value_str in old_config_dict.items():
|
|
141
|
+
if key_str in config:
|
|
142
|
+
config_dict[key_str] = config[key_str]
|
|
143
|
+
else:
|
|
144
|
+
config_dict[key_str] = value_str
|
|
145
|
+
#
|
|
146
|
+
self.delete(topic_str)
|
|
147
|
+
#
|
|
148
|
+
self.create(topic_str, partitions=partitions_int, config=config_dict, **kwargs)
|
|
149
|
+
else:
|
|
150
|
+
if isinstance(pattern_str_or_str_list, str):
|
|
151
|
+
topic_str_list = [pattern_str_or_str_list]
|
|
152
|
+
elif isinstance(pattern_str_or_str_list, list):
|
|
153
|
+
topic_str_list = pattern_str_or_str_list
|
|
154
|
+
#
|
|
155
|
+
for topic_str in topic_str_list:
|
|
156
|
+
if partitions is None:
|
|
157
|
+
partitions_int = 1
|
|
158
|
+
else:
|
|
159
|
+
partitions_int = partitions
|
|
160
|
+
#
|
|
161
|
+
self.create(topic_str, partitions=partitions_int, config=config, **kwargs)
|
|
162
|
+
#
|
|
163
|
+
return topic_str_list
|
|
164
|
+
|
|
165
|
+
retouch = recreate
|
|
166
|
+
|
|
167
|
+
#
|
|
168
|
+
|
|
169
|
+
def cp_group_offsets(self, pattern, source_group, target_storage, target_group):
|
|
170
|
+
source_group_str = source_group
|
|
171
|
+
target_group_str = target_group
|
|
172
|
+
#
|
|
173
|
+
topic_str_list = self.admin.list_topics(pattern)
|
|
174
|
+
#
|
|
175
|
+
# Get the offsets of the source consumer group.
|
|
176
|
+
topic_str_offsets_dict_dict = {topic_str: offsets_dict for topic_str, offsets_dict in self.group_offsets(source_group_str)[source_group_str].items() if topic_str in topic_str_list}
|
|
177
|
+
#
|
|
178
|
+
# Consume one message from eacg topic with the target consumer group to bring it to life.
|
|
179
|
+
for topic_str in topic_str_list:
|
|
180
|
+
co = target_storage.consumer(topic_str, group=target_group_str, type="bytes")
|
|
181
|
+
co.consume(n=1)
|
|
182
|
+
co.close()
|
|
183
|
+
#
|
|
184
|
+
target_group_offsets = target_storage.group_offsets(target_group, topic_str_offsets_dict_dict)
|
|
185
|
+
#
|
|
186
|
+
return target_group_offsets
|
|
187
|
+
|
|
188
|
+
#
|
|
189
|
+
|
|
190
|
+
def offsets_diff(self, pattern, ts, end_ts, **kwargs):
|
|
191
|
+
ts_int = ts
|
|
192
|
+
end_ts_int = end_ts
|
|
193
|
+
#
|
|
194
|
+
if end_ts_int < ts_int:
|
|
195
|
+
raise Exception(f"End timestamp ({end_ts_int}) before start timestamp ({ts_int}).")
|
|
196
|
+
#
|
|
197
|
+
topic_str_partitions_int_dict = self.partitions(pattern, **kwargs)
|
|
198
|
+
#
|
|
199
|
+
topic_str_messages_int_dict = {}
|
|
200
|
+
for topic_str, partitions_int in topic_str_partitions_int_dict.items():
|
|
201
|
+
start_offsets_dict = self.offsets_for_times(topic_str, {partition_int: ts_int for partition_int in range(partitions_int)}, replace_not_found=True, **kwargs)[topic_str]
|
|
202
|
+
end_offsets_dict = self.offsets_for_times(topic_str, {partition_int: end_ts_int for partition_int in range(partitions_int)}, replace_not_found=True, **kwargs)[topic_str]
|
|
203
|
+
#
|
|
204
|
+
# print(start_offsets_dict)
|
|
205
|
+
# print(end_offsets_dict)
|
|
206
|
+
#
|
|
207
|
+
messages_int = sum([(end_offset_int - start_offset_int) + 1 for start_offset_int, end_offset_int in zip(start_offsets_dict.values(), end_offsets_dict.values())])
|
|
208
|
+
#
|
|
209
|
+
topic_str_messages_int_dict[topic_str] = messages_int
|
|
210
|
+
#
|
|
211
|
+
return topic_str_messages_int_dict
|
|
212
|
+
|
|
213
|
+
#
|
|
214
|
+
|
|
215
|
+
def message_size(self, topic_str, **kwargs):
|
|
216
|
+
def agg(partition_int_offset_int_size_int_tuple_dict_dict, message_dict):
|
|
217
|
+
partition_int = message_dict["partition"]
|
|
218
|
+
offset_int = message_dict["offset"]
|
|
219
|
+
key_bytes = message_dict["key"]
|
|
220
|
+
key_size_int = 0 if key_bytes is None else len(key_bytes)
|
|
221
|
+
value_bytes = message_dict["value"]
|
|
222
|
+
value_size_int = 0 if value_bytes is None else len(value_bytes)
|
|
223
|
+
#
|
|
224
|
+
if partition_int not in partition_int_offset_int_size_int_tuple_dict_dict:
|
|
225
|
+
partition_int_offset_int_size_int_tuple_dict_dict[partition_int] = {offset_int: None}
|
|
226
|
+
partition_int_offset_int_size_int_tuple_dict_dict[partition_int][offset_int] = (key_size_int, value_size_int)
|
|
227
|
+
return partition_int_offset_int_size_int_tuple_dict_dict
|
|
228
|
+
#
|
|
229
|
+
(partition_int_offset_int_size_int_tuple_dict_dict, n_int) = self.foldl(topic_str, agg, {}, type="bytes", **kwargs)
|
|
230
|
+
#
|
|
231
|
+
return partition_int_offset_int_size_int_tuple_dict_dict, n_int
|
|
232
|
+
|
|
233
|
+
def message_size_stats(self, topic_str, **kwargs):
|
|
234
|
+
partition_int_offset_int_size_int_tuple_dict_dict, n_int = self.message_size(topic_str, **kwargs)
|
|
235
|
+
#
|
|
236
|
+
total_size_int = 0
|
|
237
|
+
max_dict = {}
|
|
238
|
+
min_dict = {}
|
|
239
|
+
for partition_int, offset_int_size_int_tuple_dict in partition_int_offset_int_size_int_tuple_dict_dict.items():
|
|
240
|
+
for offset_int, (key_size_int, value_size_int) in offset_int_size_int_tuple_dict.items():
|
|
241
|
+
size_int = key_size_int + value_size_int
|
|
242
|
+
#
|
|
243
|
+
total_size_int += size_int
|
|
244
|
+
#
|
|
245
|
+
if max_dict == {}:
|
|
246
|
+
max_dict = {"size": size_int, "partition": partition_int, "offset": offset_int}
|
|
247
|
+
else:
|
|
248
|
+
old_max_int = max_dict["size"]
|
|
249
|
+
new_max_int = max(size_int, old_max_int)
|
|
250
|
+
if new_max_int != old_max_int:
|
|
251
|
+
max_dict = {"size": new_max_int, "partition": partition_int, "offset": offset_int}
|
|
252
|
+
#
|
|
253
|
+
if min_dict == {}:
|
|
254
|
+
min_dict = {"size": size_int, "partition": partition_int, "offset": offset_int}
|
|
255
|
+
else:
|
|
256
|
+
old_min_int = min_dict["size"]
|
|
257
|
+
new_min_int = min(size_int, old_min_int)
|
|
258
|
+
if new_min_int != old_min_int:
|
|
259
|
+
min_dict = {"size": new_min_int, "partition": partition_int, "offset": offset_int}
|
|
260
|
+
#
|
|
261
|
+
#
|
|
262
|
+
stats_dict = {"messages": n_int, "total_size": total_size_int, "average_size": total_size_int/n_int, "max_size": max_dict, "min_size": min_dict}
|
|
263
|
+
#
|
|
264
|
+
return stats_dict
|
|
265
|
+
|
|
266
|
+
|
|
267
|
+
def collect_value_set(self, topic_str, **kwargs):
|
|
268
|
+
value_json_str_set = set()
|
|
269
|
+
#
|
|
270
|
+
def collect(message_dict):
|
|
271
|
+
value_json_str = json.dumps(message_dict["value"])
|
|
272
|
+
value_json_str_set.add(value_json_str)
|
|
273
|
+
#
|
|
274
|
+
self.foreach(topic_str, foreach_function=collect, **kwargs)
|
|
275
|
+
#
|
|
276
|
+
return value_json_str_set
|
kafi/chunker.py
ADDED
|
@@ -0,0 +1,63 @@
|
|
|
1
|
+
import uuid
|
|
2
|
+
|
|
3
|
+
from kafi.serializer import Serializer
|
|
4
|
+
from kafi.helpers import message_dict_chunk_key_to_key, default_partitioner, key_to_chunk_key, split_bytes
|
|
5
|
+
|
|
6
|
+
#
|
|
7
|
+
|
|
8
|
+
class Chunker(Serializer):
|
|
9
|
+
def __init__(self, schema_registry_config_dict, **kwargs):
|
|
10
|
+
self.chunk_size_bytes_int = kwargs["chunk_size_bytes"] if "chunk_size_bytes" in kwargs else -1
|
|
11
|
+
if self.chunk_size_bytes_int == 0:
|
|
12
|
+
raise Exception("Chunk size is zero.")
|
|
13
|
+
if self.chunk_size_bytes_int > 0 and self.__class__.__name__ == "RestProxyProducer":
|
|
14
|
+
raise Exception("Chunking not supported for RestProxy storage.")
|
|
15
|
+
#
|
|
16
|
+
if self.chunk_size_bytes_int > 0:
|
|
17
|
+
self.partitioner_function = default_partitioner
|
|
18
|
+
self.projection_function = message_dict_chunk_key_to_key
|
|
19
|
+
#
|
|
20
|
+
super().__init__(schema_registry_config_dict, **kwargs)
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
#
|
|
24
|
+
|
|
25
|
+
def chunk(self, message_dict_list):
|
|
26
|
+
message_dict_list1 = []
|
|
27
|
+
if self.chunk_size_bytes_int > 0:
|
|
28
|
+
for message_dict in message_dict_list:
|
|
29
|
+
value_bytes = message_dict["value"]
|
|
30
|
+
#
|
|
31
|
+
if len(value_bytes) > self.chunk_size_bytes_int:
|
|
32
|
+
chunk_value_bytes_list = split_bytes(value_bytes, self.chunk_size_bytes_int)
|
|
33
|
+
#
|
|
34
|
+
headers_str_bytes_tuple_list = message_dict["headers"]
|
|
35
|
+
if headers_str_bytes_tuple_list is None:
|
|
36
|
+
headers_str_bytes_tuple_list = []
|
|
37
|
+
headers_str_bytes_dict = dict(headers_str_bytes_tuple_list)
|
|
38
|
+
headers_str_bytes_dict["kafi_chunked_message_id"] = bytes(str(uuid.uuid4()), "UTF-8")
|
|
39
|
+
headers_str_bytes_dict["kafi_number_of_chunks"] = len(chunk_value_bytes_list).to_bytes(32, byteorder="big")
|
|
40
|
+
#
|
|
41
|
+
for chunk_int, chunk_value_bytes in zip(range(len(chunk_value_bytes_list)), chunk_value_bytes_list):
|
|
42
|
+
# If the first byte of the value starts with 0 we assume this is a message serialized using Schema Registry. In that case, add the five bytes from the beginning of the message to each chunk (to avoid confluent.value.schema.validation == true blocking the individual chunks).
|
|
43
|
+
if self.value_type_str in ["avro", "jsonschema", "json_sr", "pb", "protobuf"] and chunk_int > 0:
|
|
44
|
+
chunk_value_bytes = value_bytes[0:5] + chunk_value_bytes
|
|
45
|
+
#
|
|
46
|
+
chunk_key_bytes = key_to_chunk_key(message_dict["key"], chunk_int)
|
|
47
|
+
#
|
|
48
|
+
headers_str_bytes_dict["kafi_chunk_number"] = chunk_int.to_bytes(32, byteorder="big")
|
|
49
|
+
#
|
|
50
|
+
chunk_headers_str_bytes_tuple_list = list(headers_str_bytes_dict.items())
|
|
51
|
+
#
|
|
52
|
+
message_dict1 = {"value": chunk_value_bytes,
|
|
53
|
+
"key": chunk_key_bytes,
|
|
54
|
+
"partition": message_dict["partition"],
|
|
55
|
+
"timestamp": message_dict["timestamp"],
|
|
56
|
+
"headers": chunk_headers_str_bytes_tuple_list}
|
|
57
|
+
message_dict_list1.append(message_dict1)
|
|
58
|
+
else:
|
|
59
|
+
message_dict_list1 = message_dict_list
|
|
60
|
+
else:
|
|
61
|
+
message_dict_list1 = message_dict_list
|
|
62
|
+
#
|
|
63
|
+
return message_dict_list1
|
kafi/dechunker.py
ADDED
|
@@ -0,0 +1,75 @@
|
|
|
1
|
+
from kafi.deserializer import Deserializer
|
|
2
|
+
from kafi.helpers import chunk_key_to_key
|
|
3
|
+
|
|
4
|
+
#
|
|
5
|
+
|
|
6
|
+
class Dechunker(Deserializer):
|
|
7
|
+
def __init__(self, schema_registry_config_dict, **kwargs):
|
|
8
|
+
# Dictionary mapping chunked message IDs to chunk numbers to chunk value_bytes (to reconstruct chunked messages).
|
|
9
|
+
self.chunks_dict = {}
|
|
10
|
+
#
|
|
11
|
+
super().__init__(schema_registry_config_dict, **kwargs)
|
|
12
|
+
|
|
13
|
+
#
|
|
14
|
+
|
|
15
|
+
def dechunk(self, message_dict_list):
|
|
16
|
+
message_dict_list1 = []
|
|
17
|
+
for message_dict in message_dict_list:
|
|
18
|
+
topic_str = message_dict["topic"]
|
|
19
|
+
# Get dictionary of headers
|
|
20
|
+
headers_str_bytes_tuple_list = message_dict["headers"]
|
|
21
|
+
if headers_str_bytes_tuple_list is None:
|
|
22
|
+
headers_str_bytes_tuple_list = []
|
|
23
|
+
headers_str_bytes_dict = dict(headers_str_bytes_tuple_list)
|
|
24
|
+
#
|
|
25
|
+
if "kafi_chunked_message_id" in headers_str_bytes_dict:
|
|
26
|
+
#
|
|
27
|
+
chunked_message_id_str = str(headers_str_bytes_dict["kafi_chunked_message_id"])
|
|
28
|
+
#
|
|
29
|
+
number_of_chunks_int = int.from_bytes(headers_str_bytes_dict["kafi_number_of_chunks"])
|
|
30
|
+
#
|
|
31
|
+
chunk_number_int = int.from_bytes(headers_str_bytes_dict["kafi_chunk_number"])
|
|
32
|
+
#
|
|
33
|
+
if chunked_message_id_str not in self.chunks_dict:
|
|
34
|
+
self.chunks_dict[chunked_message_id_str] = {chunk_number_int1: None for chunk_number_int1 in range(number_of_chunks_int)}
|
|
35
|
+
#
|
|
36
|
+
self.chunks_dict[chunked_message_id_str][chunk_number_int] = message_dict["value"]
|
|
37
|
+
#
|
|
38
|
+
if all(value_bytes is not None for value_bytes in self.chunks_dict[chunked_message_id_str].values()):
|
|
39
|
+
dechunked_value_bytes = b""
|
|
40
|
+
#
|
|
41
|
+
for chunk_number_int1, value_bytes in self.chunks_dict[chunked_message_id_str].items():
|
|
42
|
+
# Special handling if the values were serialized in conjunction with Schema Registry.
|
|
43
|
+
if self.topic_str_value_type_str_dict[topic_str] in ["avro", "jsonschema", "json_sr", "pb", "protobuf"]:
|
|
44
|
+
# If so, skip the first five bytes from all but the first chunk (upon produce, we add the first five bytes to all messages to avoid confluent.value.schema.validation == true blocking them).
|
|
45
|
+
if chunk_number_int1 == 0:
|
|
46
|
+
dechunked_value_bytes += value_bytes
|
|
47
|
+
else:
|
|
48
|
+
dechunked_value_bytes += value_bytes[5:]
|
|
49
|
+
# Else just dechunk.
|
|
50
|
+
else:
|
|
51
|
+
dechunked_value_bytes += value_bytes
|
|
52
|
+
#
|
|
53
|
+
key_bytes = chunk_key_to_key(message_dict["key"])
|
|
54
|
+
#
|
|
55
|
+
# Delete the header fields for chunking.
|
|
56
|
+
del headers_str_bytes_dict["kafi_chunked_message_id"]
|
|
57
|
+
del headers_str_bytes_dict["kafi_number_of_chunks"]
|
|
58
|
+
del headers_str_bytes_dict["kafi_chunk_number"]
|
|
59
|
+
headers_str_bytes_tuple_list = list(headers_str_bytes_dict.items())
|
|
60
|
+
#
|
|
61
|
+
message_dict2 = {"value": dechunked_value_bytes,
|
|
62
|
+
"key": key_bytes,
|
|
63
|
+
"headers": headers_str_bytes_tuple_list,
|
|
64
|
+
"timestamp": message_dict["timestamp"],
|
|
65
|
+
"partition": message_dict["partition"],
|
|
66
|
+
"offset": message_dict["offset"],
|
|
67
|
+
"topic": message_dict["topic"]}
|
|
68
|
+
message_dict_list1.append(message_dict2)
|
|
69
|
+
#
|
|
70
|
+
# Clean up the chunks dictionary.
|
|
71
|
+
del self.chunks_dict[chunked_message_id_str]
|
|
72
|
+
else:
|
|
73
|
+
message_dict_list1.append(message_dict)
|
|
74
|
+
#
|
|
75
|
+
return message_dict_list1
|
kafi/deserializer.py
ADDED
|
@@ -0,0 +1,148 @@
|
|
|
1
|
+
import os
|
|
2
|
+
import importlib
|
|
3
|
+
import json
|
|
4
|
+
import sys
|
|
5
|
+
import tempfile
|
|
6
|
+
import uuid
|
|
7
|
+
|
|
8
|
+
from confluent_kafka.schema_registry.avro import AvroDeserializer
|
|
9
|
+
from confluent_kafka.schema_registry.json_schema import JSONDeserializer
|
|
10
|
+
from confluent_kafka.schema_registry.protobuf import ProtobufDeserializer
|
|
11
|
+
from confluent_kafka.serialization import MessageField, SerializationContext
|
|
12
|
+
from google.protobuf.json_format import MessageToDict
|
|
13
|
+
|
|
14
|
+
from kafi.schemaregistry import SchemaRegistry
|
|
15
|
+
|
|
16
|
+
class Deserializer(SchemaRegistry):
|
|
17
|
+
def __init__(self, schema_registry_config_dict, **kwargs):
|
|
18
|
+
self.deser_from_dict = kwargs["deser_from_dict"] if "deser_from_dict" in kwargs else None
|
|
19
|
+
self.deser_conf = kwargs["deser_conf"] if "deser_conf" in kwargs else None
|
|
20
|
+
self.deser_rule_conf = kwargs["deser_rule_conf"] if "deser_rule_conf" in kwargs else None
|
|
21
|
+
self.deser_rule_registry = kwargs["deser_rule_registry"] if "deser_rule_registry" in kwargs else None
|
|
22
|
+
self.deser_json_decode = kwargs["deser_json_decode"] if "deser_json_decode" in kwargs else None
|
|
23
|
+
self.deser_return_record_name = kwargs["deser_return_record_name"] if "deser_return_record_name" in kwargs else False
|
|
24
|
+
#
|
|
25
|
+
super().__init__(schema_registry_config_dict)
|
|
26
|
+
|
|
27
|
+
def deserialize(self, payload_bytes, type_str, topic_str, headers_dict, key_bool):
|
|
28
|
+
if type_str.lower() == "bytes":
|
|
29
|
+
deserialized_payload = self.bytes_to_bytes(payload_bytes)
|
|
30
|
+
elif type_str.lower() in ["str", "string"]:
|
|
31
|
+
deserialized_payload = self.bytes_to_str(payload_bytes)
|
|
32
|
+
elif type_str.lower() == "json":
|
|
33
|
+
deserialized_payload = self.bytes_to_dict(payload_bytes)
|
|
34
|
+
elif type_str.lower() == "avro":
|
|
35
|
+
deserialized_payload = self.bytes_avro_to_dict(payload_bytes, topic_str, headers_dict, key_bool)
|
|
36
|
+
elif type_str.lower() in ["jsonschema", "json_sr"]:
|
|
37
|
+
deserialized_payload = self.bytes_jsonschema_to_dict(payload_bytes, topic_str, headers_dict, key_bool)
|
|
38
|
+
elif type_str.lower() in ["protobuf", "pb"]:
|
|
39
|
+
deserialized_payload = self.bytes_protobuf_to_dict(payload_bytes, topic_str, headers_dict, key_bool)
|
|
40
|
+
else:
|
|
41
|
+
raise Exception("Only \"str\", \"bytes\", \"json\", \"protobuf\" (\"pb\"), \"avro\" and \"jsonschema\" (\"json_sr\") supported.")
|
|
42
|
+
#
|
|
43
|
+
return deserialized_payload
|
|
44
|
+
|
|
45
|
+
def bytes_to_str(self, bytes):
|
|
46
|
+
if bytes:
|
|
47
|
+
return bytes.decode("utf-8")
|
|
48
|
+
else:
|
|
49
|
+
return bytes
|
|
50
|
+
|
|
51
|
+
def bytes_to_bytes(self, bytes):
|
|
52
|
+
return bytes
|
|
53
|
+
|
|
54
|
+
def bytes_to_dict(self, bytes):
|
|
55
|
+
if bytes is None:
|
|
56
|
+
return None
|
|
57
|
+
#
|
|
58
|
+
return json.loads(bytes)
|
|
59
|
+
|
|
60
|
+
def bytes_avro_to_dict(self, bytes, topic_str, headers_dict, key_bool):
|
|
61
|
+
if bytes is None:
|
|
62
|
+
return None
|
|
63
|
+
#
|
|
64
|
+
schema_str = self.get_schema_str(bytes, headers_dict, key_bool)
|
|
65
|
+
#
|
|
66
|
+
avroDeserializer = AvroDeserializer(self.schemaRegistryClient, schema_str, self.deser_from_dict, self.deser_return_record_name, self.deser_conf, self.deser_rule_conf, self.deser_rule_registry)
|
|
67
|
+
serializationContext = SerializationContext(topic_str, MessageField.KEY if key_bool else MessageField.VALUE)
|
|
68
|
+
dict = avroDeserializer(bytes, serializationContext)
|
|
69
|
+
return dict
|
|
70
|
+
|
|
71
|
+
def bytes_jsonschema_to_dict(self, bytes, topic_str, headers_dict, key_bool):
|
|
72
|
+
if bytes is None:
|
|
73
|
+
return None
|
|
74
|
+
#
|
|
75
|
+
schema_str = self.get_schema_str(bytes, headers_dict, key_bool)
|
|
76
|
+
#
|
|
77
|
+
jsonDeserializer = JSONDeserializer(schema_str, self.deser_from_dict, None, self.deser_conf, self.deser_rule_conf, self.deser_rule_registry, self.deser_json_decode)
|
|
78
|
+
serializationContext = SerializationContext(topic_str, MessageField.KEY if key_bool else MessageField.VALUE)
|
|
79
|
+
dict = jsonDeserializer(bytes, serializationContext)
|
|
80
|
+
return dict
|
|
81
|
+
|
|
82
|
+
def bytes_protobuf_to_dict(self, bytes, topic_str, headers_dict, key_bool):
|
|
83
|
+
if bytes is None:
|
|
84
|
+
return None
|
|
85
|
+
#
|
|
86
|
+
schema_id_int = int.from_bytes(bytes[1:5], "big")
|
|
87
|
+
if schema_id_int in self.schema_id_int_generalizedProtocolMessageType_protobuf_schema_str_tuple_dict:
|
|
88
|
+
generalizedProtocolMessageType, protobuf_schema_str = self.schema_id_int_generalizedProtocolMessageType_protobuf_schema_str_tuple_dict[schema_id_int]
|
|
89
|
+
else:
|
|
90
|
+
generalizedProtocolMessageType, protobuf_schema_str = self.schema_id_int_to_generalizedProtocolMessageType_protobuf_schema_str_tuple(schema_id_int)
|
|
91
|
+
self.schema_id_int_generalizedProtocolMessageType_protobuf_schema_str_tuple_dict[schema_id_int] = (generalizedProtocolMessageType, protobuf_schema_str)
|
|
92
|
+
#
|
|
93
|
+
# Prevent: RuntimeError: ProtobufSerializer: the 'use.deprecated.format' configuration property must be explicitly set due to backward incompatibility with older confluent-kafka-python Protobuf producers and consumers. See the release notes for more details
|
|
94
|
+
if self.deser_conf is None:
|
|
95
|
+
self.deser_conf = {"use.deprecated.format": False}
|
|
96
|
+
protobufDeserializer = ProtobufDeserializer(generalizedProtocolMessageType, self.deser_conf, None, self.deser_rule_conf, self.deser_rule_registry)
|
|
97
|
+
serializationContext = SerializationContext(topic_str, MessageField.KEY if key_bool else MessageField.VALUE)
|
|
98
|
+
protobuf_message = protobufDeserializer(bytes, serializationContext)
|
|
99
|
+
dict = MessageToDict(protobuf_message)
|
|
100
|
+
return dict
|
|
101
|
+
|
|
102
|
+
# Helpers
|
|
103
|
+
|
|
104
|
+
def get_schema_str(self, bytes, headers_dict, key_bool):
|
|
105
|
+
schema_id_key_str = "__key_schema_id" if key_bool else "__value_schema_id"
|
|
106
|
+
#
|
|
107
|
+
if headers_dict is not None and schema_id_key_str in headers_dict:
|
|
108
|
+
# Get the Schema ID from headers_dict if available.
|
|
109
|
+
schema_guid_bytes = headers_dict[schema_id_key_str]
|
|
110
|
+
# Skip the version byte (\x01).
|
|
111
|
+
schema_guid_bytes1 = schema_guid_bytes[1:]
|
|
112
|
+
# Convert to UUID.
|
|
113
|
+
schema_guid_str = str(uuid.UUID(bytes=schema_guid_bytes1))
|
|
114
|
+
# Get the schema at last.
|
|
115
|
+
schema_dict = self.get_schema_by_guid(schema_guid_str)
|
|
116
|
+
else:
|
|
117
|
+
# Else get the Schema ID from the payload.
|
|
118
|
+
schema_id_int = int.from_bytes(bytes[1:5], "big")
|
|
119
|
+
schema_dict = self.get_schema(schema_id_int)
|
|
120
|
+
#
|
|
121
|
+
schema_str = schema_dict["schema_str"]
|
|
122
|
+
#
|
|
123
|
+
return schema_str
|
|
124
|
+
|
|
125
|
+
def schema_id_int_to_generalizedProtocolMessageType_protobuf_schema_str_tuple(self, schema_id_int):
|
|
126
|
+
schema_dict = self.get_schema(schema_id_int)
|
|
127
|
+
schema_str = schema_dict["schema_str"]
|
|
128
|
+
#
|
|
129
|
+
generalizedProtocolMessageType = self.schema_id_int_and_schema_str_to_generalizedProtocolMessageType(schema_id_int, schema_str)
|
|
130
|
+
#
|
|
131
|
+
return generalizedProtocolMessageType, schema_str
|
|
132
|
+
|
|
133
|
+
def schema_id_int_and_schema_str_to_generalizedProtocolMessageType(self, schema_id_int, schema_str):
|
|
134
|
+
path_str = f"/{tempfile.gettempdir()}/kafi/protobuf/{self.storage_obj.config_str}"
|
|
135
|
+
os.makedirs(path_str, exist_ok=True)
|
|
136
|
+
file_str = f"schema_{schema_id_int}.proto"
|
|
137
|
+
file_path_str = f"{path_str}/{file_str}"
|
|
138
|
+
with open(file_path_str, "w") as textIOWrapper:
|
|
139
|
+
textIOWrapper.write(schema_str)
|
|
140
|
+
#
|
|
141
|
+
import grpc_tools.protoc
|
|
142
|
+
grpc_tools.protoc.main(["protoc", f"-I{path_str}", f"--python_out={path_str}", f"{file_str}"])
|
|
143
|
+
#
|
|
144
|
+
sys.path.insert(1, path_str)
|
|
145
|
+
schema_module = importlib.import_module(f"schema_{schema_id_int}_pb2")
|
|
146
|
+
schema_name_str = list(schema_module.DESCRIPTOR.message_types_by_name.keys())[0]
|
|
147
|
+
generalizedProtocolMessageType = getattr(schema_module, schema_name_str)
|
|
148
|
+
return generalizedProtocolMessageType
|
kafi/files.py
ADDED
|
@@ -0,0 +1,85 @@
|
|
|
1
|
+
import io
|
|
2
|
+
import pathlib
|
|
3
|
+
|
|
4
|
+
from kafi.pandas import Pandas
|
|
5
|
+
|
|
6
|
+
# Constants
|
|
7
|
+
|
|
8
|
+
ALL_MESSAGES = -1
|
|
9
|
+
|
|
10
|
+
#
|
|
11
|
+
|
|
12
|
+
#x = [{"name": "cookie", "calories": 500.0, "colour": "brown"}, {"name": "cake", "calories": 260.0, "colour": "white"}, {"name": "timtam", "calories": 80.0, "colour": "chocolate"}]
|
|
13
|
+
|
|
14
|
+
class Files(Pandas):
|
|
15
|
+
def topic_to_file(self, topic, fs_obj, file, n=ALL_MESSAGES, **kwargs):
|
|
16
|
+
if not fs_obj.__class__.__bases__[0].__name__== "FS":
|
|
17
|
+
raise Exception("The target must be a file system.")
|
|
18
|
+
#
|
|
19
|
+
file_str = file
|
|
20
|
+
#
|
|
21
|
+
suffix_str = pathlib.Path(file_str).suffix
|
|
22
|
+
if suffix_str not in [".csv", ".json", ".parquet", ".xlsx", ".xml", ".bytes"]:
|
|
23
|
+
raise Exception("Only \".csv\", \".json\", \".parquet\", \".xlsx\", \".xml\" and \".bytes\" supported.")
|
|
24
|
+
#
|
|
25
|
+
if suffix_str == ".bytes":
|
|
26
|
+
message_dict_list = self.cat(topic, n, type="bytes", **kwargs)
|
|
27
|
+
data_bytes = b""
|
|
28
|
+
for message_dict in message_dict_list:
|
|
29
|
+
value_bytes = message_dict["value"]
|
|
30
|
+
data_bytes += value_bytes + b"\n"
|
|
31
|
+
else:
|
|
32
|
+
df = self.topic_to_df(topic, n, **kwargs)
|
|
33
|
+
data_bytesIO = io.BytesIO()
|
|
34
|
+
#
|
|
35
|
+
if suffix_str == ".csv":
|
|
36
|
+
index_bool = kwargs["index"] if "index" in kwargs else False
|
|
37
|
+
df.to_csv(data_bytesIO, index=index_bool)
|
|
38
|
+
elif suffix_str == ".json":
|
|
39
|
+
index_bool = kwargs["index"] if "index" in kwargs else None
|
|
40
|
+
df.to_json(data_bytesIO, orient="records")
|
|
41
|
+
elif suffix_str == ".parquet":
|
|
42
|
+
index_bool = kwargs["index"] if "index" in kwargs else None
|
|
43
|
+
df.to_parquet(data_bytesIO, index=index_bool, engine="fastparquet")
|
|
44
|
+
elif suffix_str == ".xlsx":
|
|
45
|
+
index_bool = kwargs["index"] if "index" in kwargs else False
|
|
46
|
+
df.to_excel(data_bytesIO, index=index_bool)
|
|
47
|
+
elif suffix_str == ".xml":
|
|
48
|
+
index_bool = kwargs["index"] if "index" in kwargs else False
|
|
49
|
+
df.to_xml(data_bytesIO, index=index_bool)
|
|
50
|
+
#
|
|
51
|
+
data_bytes = data_bytesIO.getvalue()
|
|
52
|
+
#
|
|
53
|
+
file_abs_path_str = fs_obj.admin.get_file_abs_path_str(file_str)
|
|
54
|
+
fs_obj.admin.write_bytes(file_abs_path_str, data_bytes)
|
|
55
|
+
#
|
|
56
|
+
return len(data_bytes)
|
|
57
|
+
|
|
58
|
+
def file_to_topic(self, file, storage_obj, topic, n=ALL_MESSAGES, **kwargs):
|
|
59
|
+
if not self.__class__.__bases__[0].__name__== "FS":
|
|
60
|
+
raise Exception("The source must be a file system.")
|
|
61
|
+
#
|
|
62
|
+
import pandas as pd
|
|
63
|
+
#
|
|
64
|
+
file_str = file
|
|
65
|
+
#
|
|
66
|
+
suffix_str = pathlib.Path(file_str).suffix
|
|
67
|
+
if suffix_str not in [".csv", ".json", ".parquet", ".xlsx", ".xml"]:
|
|
68
|
+
raise Exception("Only \".csv\", \".json\", \".parquet\", \".xlsx\" and \".xml\" supported.")
|
|
69
|
+
#
|
|
70
|
+
file_abs_path_str = self.admin.get_file_abs_path_str(file_str)
|
|
71
|
+
data_bytes = self.admin.read_bytes(file_abs_path_str)
|
|
72
|
+
data_bytesIO = io.BytesIO(data_bytes)
|
|
73
|
+
#
|
|
74
|
+
if suffix_str == ".csv":
|
|
75
|
+
df = pd.read_csv(data_bytesIO)
|
|
76
|
+
elif suffix_str == ".json":
|
|
77
|
+
df = pd.read_json(data_bytesIO)
|
|
78
|
+
elif suffix_str == ".parquet":
|
|
79
|
+
df = pd.read_parquet(data_bytesIO)
|
|
80
|
+
elif suffix_str == ".xlsx":
|
|
81
|
+
df = pd.read_excel(data_bytesIO)
|
|
82
|
+
elif suffix_str == ".xml":
|
|
83
|
+
df = pd.read_xml(data_bytesIO)
|
|
84
|
+
#
|
|
85
|
+
return storage_obj.df_to_topic(df, topic, n, **kwargs)
|
kafi/fs/__init__.py
ADDED
|
File without changes
|
|
File without changes
|