kafi 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (59) hide show
  1. kafi/__init__.py +0 -0
  2. kafi/addons.py +276 -0
  3. kafi/chunker.py +63 -0
  4. kafi/dechunker.py +75 -0
  5. kafi/deserializer.py +148 -0
  6. kafi/files.py +85 -0
  7. kafi/fs/__init__.py +0 -0
  8. kafi/fs/azureblob/__init__.py +0 -0
  9. kafi/fs/azureblob/azureblob.py +31 -0
  10. kafi/fs/azureblob/azureblob_admin.py +96 -0
  11. kafi/fs/azureblob/azureblob_consumer.py +7 -0
  12. kafi/fs/azureblob/azureblob_producer.py +11 -0
  13. kafi/fs/fs.py +68 -0
  14. kafi/fs/fs_admin.py +415 -0
  15. kafi/fs/fs_consumer.py +181 -0
  16. kafi/fs/fs_producer.py +70 -0
  17. kafi/fs/local/__init__.py +0 -0
  18. kafi/fs/local/local.py +32 -0
  19. kafi/fs/local/local_admin.py +73 -0
  20. kafi/fs/local/local_consumer.py +11 -0
  21. kafi/fs/local/local_producer.py +12 -0
  22. kafi/fs/s3/__init__.py +0 -0
  23. kafi/fs/s3/s3.py +31 -0
  24. kafi/fs/s3/s3_admin.py +87 -0
  25. kafi/fs/s3/s3_consumer.py +7 -0
  26. kafi/fs/s3/s3_producer.py +11 -0
  27. kafi/functional.py +461 -0
  28. kafi/helpers.py +441 -0
  29. kafi/kafi.py +8 -0
  30. kafi/kafka/__init__.py +0 -0
  31. kafi/kafka/cluster/__init__.py +0 -0
  32. kafi/kafka/cluster/cluster.py +44 -0
  33. kafi/kafka/cluster/cluster_admin.py +656 -0
  34. kafi/kafka/cluster/cluster_consumer.py +166 -0
  35. kafi/kafka/cluster/cluster_producer.py +77 -0
  36. kafi/kafka/kafka.py +120 -0
  37. kafi/kafka/kafka_admin.py +5 -0
  38. kafi/kafka/kafka_consumer.py +11 -0
  39. kafi/kafka/kafka_producer.py +10 -0
  40. kafi/kafka/restproxy/__init__.py +0 -0
  41. kafi/kafka/restproxy/restproxy.py +62 -0
  42. kafi/kafka/restproxy/restproxy_admin.py +421 -0
  43. kafi/kafka/restproxy/restproxy_consumer.py +212 -0
  44. kafi/kafka/restproxy/restproxy_producer.py +134 -0
  45. kafi/pandas.py +46 -0
  46. kafi/schemaregistry.py +253 -0
  47. kafi/serializer.py +121 -0
  48. kafi/shell.py +125 -0
  49. kafi/storage.py +327 -0
  50. kafi/storage_admin.py +83 -0
  51. kafi/storage_consumer.py +213 -0
  52. kafi/storage_producer.py +101 -0
  53. kafi/streams/__init__.py +0 -0
  54. kafi/streams/streams.py +174 -0
  55. kafi/streams/topologynode.py +731 -0
  56. kafi-0.1.0.dist-info/METADATA +992 -0
  57. kafi-0.1.0.dist-info/RECORD +59 -0
  58. kafi-0.1.0.dist-info/WHEEL +4 -0
  59. kafi-0.1.0.dist-info/licenses/LICENSE +201 -0
kafi/__init__.py ADDED
File without changes
kafi/addons.py ADDED
@@ -0,0 +1,276 @@
1
+ import json
2
+
3
+ from kafi.functional import Functional
4
+ from kafi.helpers import copy_kwargs
5
+
6
+ # Constants
7
+
8
+ ALL_MESSAGES = -1
9
+
10
+ #
11
+
12
+ def default_projection_function(message_dict1, message_dict2):
13
+ message_dict = dict(message_dict1)
14
+ message_dict["value"] = message_dict1["value"] | message_dict2["value"]
15
+ return message_dict
16
+ #
17
+
18
+ class AddOns(Functional):
19
+ def compact(self, topic, n=ALL_MESSAGES, **kwargs):
20
+ def foldl_function(acc, message_dict):
21
+ key_hash_int_message_dict_dict = acc
22
+ #
23
+ key = message_dict["key"]
24
+ value = message_dict["value"]
25
+ #
26
+ if key is not None:
27
+ key_hash_int = hash(str(key))
28
+ if value is None:
29
+ if key_hash_int in key_hash_int_message_dict_dict:
30
+ del key_hash_int_message_dict_dict[key_hash_int]
31
+ else:
32
+ key_hash_int_message_dict_dict[key_hash_int] = message_dict
33
+ #
34
+ return key_hash_int_message_dict_dict
35
+ #
36
+
37
+ (key_hash_int_message_dict_dict, _) = self.foldl(topic, foldl_function, {}, n, **kwargs)
38
+ #
39
+ message_dict_list = list(key_hash_int_message_dict_dict.values())
40
+ #
41
+ return message_dict_list
42
+
43
+ def compact_to(self, topic, target_storage, target_topic, n=ALL_MESSAGES, **kwargs):
44
+ source_kwargs = copy_kwargs("source", **kwargs)
45
+ target_kwargs = copy_kwargs("target", **kwargs)
46
+ #
47
+ message_dict_list = self.compact(topic, n, **source_kwargs)
48
+ #
49
+ target_producer = target_storage.producer(target_topic, **target_kwargs)
50
+ key_bytes_list_value_bytes_list_tuple = target_producer.produce_list(message_dict_list, **target_kwargs)
51
+ target_producer.close()
52
+ #
53
+ return key_bytes_list_value_bytes_list_tuple
54
+
55
+ #
56
+
57
+ def join_to(self, source_topic1, source_storage2, source_topic2, target_storage, target_topic, get_key_function1=lambda x: x["key"], get_key_function2=lambda x: x["key"], projection_function=default_projection_function, join="left", n=ALL_MESSAGES, **kwargs):
58
+ join_str = join
59
+ #
60
+ if join_str not in ["inner", "left", "right"]:
61
+ raise Exception("Only \"inner\", \"left\" and \"right\" supported.")
62
+ #
63
+ def zip_foldl_to_function(acc, message_dict1, message_dict2):
64
+ # print(message_dict1["value"])
65
+ # print(message_dict2["value"])
66
+ # print("===")
67
+ (index_dict1, index_dict2) = acc
68
+ #
69
+ key1 = get_key_function1(message_dict1)
70
+ key2 = get_key_function2(message_dict2)
71
+ # DBSP: L join R = deltaL join deltaR + deltaL join R + L join deltaR
72
+ out_message_dict_list = []
73
+ # 1. deltaL join deltaR
74
+ if key1 == key2:
75
+ # Match in deltaL join deltaR.
76
+ out_message_dict_list.append(projection_function(message_dict1, message_dict2))
77
+ else:
78
+ # 2. deltaL join R
79
+ if key1 in index_dict2:
80
+ # Match in deltaL join R.
81
+ out_message_dict_list.append(projection_function(message_dict1, index_dict2[key1]))
82
+ else:
83
+ # Could not find key1 in index_dict2.
84
+ # Only append to the output if the join type is "left"
85
+ if join_str == "left":
86
+ out_message_dict_list.append(message_dict1)
87
+ # 3. L join deltaR
88
+ if key2 in index_dict1:
89
+ # Match in L join deltaR
90
+ out_message_dict_list.append(projection_function(index_dict1[key2], message_dict2))
91
+ else:
92
+ # Could not find key2 in index_dict1.
93
+ # Only append to the output if the join type is "right"
94
+ if join_str == "right":
95
+ out_message_dict_list.append(message_dict2)
96
+ # Depending on the join type, persist:
97
+ # * both sides (inner)
98
+ # * the left side (left)
99
+ # * the right side (right)
100
+ if join_str == "inner":
101
+ index_dict1[key1] = message_dict1
102
+ index_dict2[key2] = message_dict2
103
+ elif join_str == "left":
104
+ index_dict1[key1] = message_dict1
105
+ elif join_str == "right":
106
+ index_dict1[key2] = message_dict2
107
+ #
108
+ return ((index_dict1, index_dict2), list(out_message_dict_list))
109
+
110
+ return self.zip_foldl_to(source_topic1, source_storage2, source_topic2, target_storage, target_topic, zip_foldl_to_function, ({}, {}), n=n, **kwargs)
111
+
112
+ #
113
+
114
+ def repeat(self, topic_str, n=1, **kwargs):
115
+ n_int = n
116
+ #
117
+ message_dict_list = self.tail(topic_str, type="bytes", n=n_int, **kwargs)
118
+ pr = self.producer(topic_str, type="bytes", **kwargs)
119
+ pr.produce_list(message_dict_list, **kwargs)
120
+ pr.close()
121
+ #
122
+ return message_dict_list
123
+
124
+ #
125
+
126
+ def recreate(self, pattern, partitions=None, config={}, **kwargs):
127
+ pattern_str_or_str_list = pattern
128
+ #
129
+ topic_str_list = self.admin.list_topics(pattern_str_or_str_list)
130
+ #
131
+ if topic_str_list:
132
+ for topic_str in topic_str_list:
133
+ if partitions is None:
134
+ partitions_int = self.partitions(topic_str)[topic_str]
135
+ else:
136
+ partitions_int = partitions
137
+ #
138
+ old_config_dict = self.config(topic_str)[topic_str]
139
+ config_dict = {}
140
+ for key_str, value_str in old_config_dict.items():
141
+ if key_str in config:
142
+ config_dict[key_str] = config[key_str]
143
+ else:
144
+ config_dict[key_str] = value_str
145
+ #
146
+ self.delete(topic_str)
147
+ #
148
+ self.create(topic_str, partitions=partitions_int, config=config_dict, **kwargs)
149
+ else:
150
+ if isinstance(pattern_str_or_str_list, str):
151
+ topic_str_list = [pattern_str_or_str_list]
152
+ elif isinstance(pattern_str_or_str_list, list):
153
+ topic_str_list = pattern_str_or_str_list
154
+ #
155
+ for topic_str in topic_str_list:
156
+ if partitions is None:
157
+ partitions_int = 1
158
+ else:
159
+ partitions_int = partitions
160
+ #
161
+ self.create(topic_str, partitions=partitions_int, config=config, **kwargs)
162
+ #
163
+ return topic_str_list
164
+
165
+ retouch = recreate
166
+
167
+ #
168
+
169
+ def cp_group_offsets(self, pattern, source_group, target_storage, target_group):
170
+ source_group_str = source_group
171
+ target_group_str = target_group
172
+ #
173
+ topic_str_list = self.admin.list_topics(pattern)
174
+ #
175
+ # Get the offsets of the source consumer group.
176
+ topic_str_offsets_dict_dict = {topic_str: offsets_dict for topic_str, offsets_dict in self.group_offsets(source_group_str)[source_group_str].items() if topic_str in topic_str_list}
177
+ #
178
+ # Consume one message from eacg topic with the target consumer group to bring it to life.
179
+ for topic_str in topic_str_list:
180
+ co = target_storage.consumer(topic_str, group=target_group_str, type="bytes")
181
+ co.consume(n=1)
182
+ co.close()
183
+ #
184
+ target_group_offsets = target_storage.group_offsets(target_group, topic_str_offsets_dict_dict)
185
+ #
186
+ return target_group_offsets
187
+
188
+ #
189
+
190
+ def offsets_diff(self, pattern, ts, end_ts, **kwargs):
191
+ ts_int = ts
192
+ end_ts_int = end_ts
193
+ #
194
+ if end_ts_int < ts_int:
195
+ raise Exception(f"End timestamp ({end_ts_int}) before start timestamp ({ts_int}).")
196
+ #
197
+ topic_str_partitions_int_dict = self.partitions(pattern, **kwargs)
198
+ #
199
+ topic_str_messages_int_dict = {}
200
+ for topic_str, partitions_int in topic_str_partitions_int_dict.items():
201
+ start_offsets_dict = self.offsets_for_times(topic_str, {partition_int: ts_int for partition_int in range(partitions_int)}, replace_not_found=True, **kwargs)[topic_str]
202
+ end_offsets_dict = self.offsets_for_times(topic_str, {partition_int: end_ts_int for partition_int in range(partitions_int)}, replace_not_found=True, **kwargs)[topic_str]
203
+ #
204
+ # print(start_offsets_dict)
205
+ # print(end_offsets_dict)
206
+ #
207
+ messages_int = sum([(end_offset_int - start_offset_int) + 1 for start_offset_int, end_offset_int in zip(start_offsets_dict.values(), end_offsets_dict.values())])
208
+ #
209
+ topic_str_messages_int_dict[topic_str] = messages_int
210
+ #
211
+ return topic_str_messages_int_dict
212
+
213
+ #
214
+
215
+ def message_size(self, topic_str, **kwargs):
216
+ def agg(partition_int_offset_int_size_int_tuple_dict_dict, message_dict):
217
+ partition_int = message_dict["partition"]
218
+ offset_int = message_dict["offset"]
219
+ key_bytes = message_dict["key"]
220
+ key_size_int = 0 if key_bytes is None else len(key_bytes)
221
+ value_bytes = message_dict["value"]
222
+ value_size_int = 0 if value_bytes is None else len(value_bytes)
223
+ #
224
+ if partition_int not in partition_int_offset_int_size_int_tuple_dict_dict:
225
+ partition_int_offset_int_size_int_tuple_dict_dict[partition_int] = {offset_int: None}
226
+ partition_int_offset_int_size_int_tuple_dict_dict[partition_int][offset_int] = (key_size_int, value_size_int)
227
+ return partition_int_offset_int_size_int_tuple_dict_dict
228
+ #
229
+ (partition_int_offset_int_size_int_tuple_dict_dict, n_int) = self.foldl(topic_str, agg, {}, type="bytes", **kwargs)
230
+ #
231
+ return partition_int_offset_int_size_int_tuple_dict_dict, n_int
232
+
233
+ def message_size_stats(self, topic_str, **kwargs):
234
+ partition_int_offset_int_size_int_tuple_dict_dict, n_int = self.message_size(topic_str, **kwargs)
235
+ #
236
+ total_size_int = 0
237
+ max_dict = {}
238
+ min_dict = {}
239
+ for partition_int, offset_int_size_int_tuple_dict in partition_int_offset_int_size_int_tuple_dict_dict.items():
240
+ for offset_int, (key_size_int, value_size_int) in offset_int_size_int_tuple_dict.items():
241
+ size_int = key_size_int + value_size_int
242
+ #
243
+ total_size_int += size_int
244
+ #
245
+ if max_dict == {}:
246
+ max_dict = {"size": size_int, "partition": partition_int, "offset": offset_int}
247
+ else:
248
+ old_max_int = max_dict["size"]
249
+ new_max_int = max(size_int, old_max_int)
250
+ if new_max_int != old_max_int:
251
+ max_dict = {"size": new_max_int, "partition": partition_int, "offset": offset_int}
252
+ #
253
+ if min_dict == {}:
254
+ min_dict = {"size": size_int, "partition": partition_int, "offset": offset_int}
255
+ else:
256
+ old_min_int = min_dict["size"]
257
+ new_min_int = min(size_int, old_min_int)
258
+ if new_min_int != old_min_int:
259
+ min_dict = {"size": new_min_int, "partition": partition_int, "offset": offset_int}
260
+ #
261
+ #
262
+ stats_dict = {"messages": n_int, "total_size": total_size_int, "average_size": total_size_int/n_int, "max_size": max_dict, "min_size": min_dict}
263
+ #
264
+ return stats_dict
265
+
266
+
267
+ def collect_value_set(self, topic_str, **kwargs):
268
+ value_json_str_set = set()
269
+ #
270
+ def collect(message_dict):
271
+ value_json_str = json.dumps(message_dict["value"])
272
+ value_json_str_set.add(value_json_str)
273
+ #
274
+ self.foreach(topic_str, foreach_function=collect, **kwargs)
275
+ #
276
+ return value_json_str_set
kafi/chunker.py ADDED
@@ -0,0 +1,63 @@
1
+ import uuid
2
+
3
+ from kafi.serializer import Serializer
4
+ from kafi.helpers import message_dict_chunk_key_to_key, default_partitioner, key_to_chunk_key, split_bytes
5
+
6
+ #
7
+
8
+ class Chunker(Serializer):
9
+ def __init__(self, schema_registry_config_dict, **kwargs):
10
+ self.chunk_size_bytes_int = kwargs["chunk_size_bytes"] if "chunk_size_bytes" in kwargs else -1
11
+ if self.chunk_size_bytes_int == 0:
12
+ raise Exception("Chunk size is zero.")
13
+ if self.chunk_size_bytes_int > 0 and self.__class__.__name__ == "RestProxyProducer":
14
+ raise Exception("Chunking not supported for RestProxy storage.")
15
+ #
16
+ if self.chunk_size_bytes_int > 0:
17
+ self.partitioner_function = default_partitioner
18
+ self.projection_function = message_dict_chunk_key_to_key
19
+ #
20
+ super().__init__(schema_registry_config_dict, **kwargs)
21
+
22
+
23
+ #
24
+
25
+ def chunk(self, message_dict_list):
26
+ message_dict_list1 = []
27
+ if self.chunk_size_bytes_int > 0:
28
+ for message_dict in message_dict_list:
29
+ value_bytes = message_dict["value"]
30
+ #
31
+ if len(value_bytes) > self.chunk_size_bytes_int:
32
+ chunk_value_bytes_list = split_bytes(value_bytes, self.chunk_size_bytes_int)
33
+ #
34
+ headers_str_bytes_tuple_list = message_dict["headers"]
35
+ if headers_str_bytes_tuple_list is None:
36
+ headers_str_bytes_tuple_list = []
37
+ headers_str_bytes_dict = dict(headers_str_bytes_tuple_list)
38
+ headers_str_bytes_dict["kafi_chunked_message_id"] = bytes(str(uuid.uuid4()), "UTF-8")
39
+ headers_str_bytes_dict["kafi_number_of_chunks"] = len(chunk_value_bytes_list).to_bytes(32, byteorder="big")
40
+ #
41
+ for chunk_int, chunk_value_bytes in zip(range(len(chunk_value_bytes_list)), chunk_value_bytes_list):
42
+ # If the first byte of the value starts with 0 we assume this is a message serialized using Schema Registry. In that case, add the five bytes from the beginning of the message to each chunk (to avoid confluent.value.schema.validation == true blocking the individual chunks).
43
+ if self.value_type_str in ["avro", "jsonschema", "json_sr", "pb", "protobuf"] and chunk_int > 0:
44
+ chunk_value_bytes = value_bytes[0:5] + chunk_value_bytes
45
+ #
46
+ chunk_key_bytes = key_to_chunk_key(message_dict["key"], chunk_int)
47
+ #
48
+ headers_str_bytes_dict["kafi_chunk_number"] = chunk_int.to_bytes(32, byteorder="big")
49
+ #
50
+ chunk_headers_str_bytes_tuple_list = list(headers_str_bytes_dict.items())
51
+ #
52
+ message_dict1 = {"value": chunk_value_bytes,
53
+ "key": chunk_key_bytes,
54
+ "partition": message_dict["partition"],
55
+ "timestamp": message_dict["timestamp"],
56
+ "headers": chunk_headers_str_bytes_tuple_list}
57
+ message_dict_list1.append(message_dict1)
58
+ else:
59
+ message_dict_list1 = message_dict_list
60
+ else:
61
+ message_dict_list1 = message_dict_list
62
+ #
63
+ return message_dict_list1
kafi/dechunker.py ADDED
@@ -0,0 +1,75 @@
1
+ from kafi.deserializer import Deserializer
2
+ from kafi.helpers import chunk_key_to_key
3
+
4
+ #
5
+
6
+ class Dechunker(Deserializer):
7
+ def __init__(self, schema_registry_config_dict, **kwargs):
8
+ # Dictionary mapping chunked message IDs to chunk numbers to chunk value_bytes (to reconstruct chunked messages).
9
+ self.chunks_dict = {}
10
+ #
11
+ super().__init__(schema_registry_config_dict, **kwargs)
12
+
13
+ #
14
+
15
+ def dechunk(self, message_dict_list):
16
+ message_dict_list1 = []
17
+ for message_dict in message_dict_list:
18
+ topic_str = message_dict["topic"]
19
+ # Get dictionary of headers
20
+ headers_str_bytes_tuple_list = message_dict["headers"]
21
+ if headers_str_bytes_tuple_list is None:
22
+ headers_str_bytes_tuple_list = []
23
+ headers_str_bytes_dict = dict(headers_str_bytes_tuple_list)
24
+ #
25
+ if "kafi_chunked_message_id" in headers_str_bytes_dict:
26
+ #
27
+ chunked_message_id_str = str(headers_str_bytes_dict["kafi_chunked_message_id"])
28
+ #
29
+ number_of_chunks_int = int.from_bytes(headers_str_bytes_dict["kafi_number_of_chunks"])
30
+ #
31
+ chunk_number_int = int.from_bytes(headers_str_bytes_dict["kafi_chunk_number"])
32
+ #
33
+ if chunked_message_id_str not in self.chunks_dict:
34
+ self.chunks_dict[chunked_message_id_str] = {chunk_number_int1: None for chunk_number_int1 in range(number_of_chunks_int)}
35
+ #
36
+ self.chunks_dict[chunked_message_id_str][chunk_number_int] = message_dict["value"]
37
+ #
38
+ if all(value_bytes is not None for value_bytes in self.chunks_dict[chunked_message_id_str].values()):
39
+ dechunked_value_bytes = b""
40
+ #
41
+ for chunk_number_int1, value_bytes in self.chunks_dict[chunked_message_id_str].items():
42
+ # Special handling if the values were serialized in conjunction with Schema Registry.
43
+ if self.topic_str_value_type_str_dict[topic_str] in ["avro", "jsonschema", "json_sr", "pb", "protobuf"]:
44
+ # If so, skip the first five bytes from all but the first chunk (upon produce, we add the first five bytes to all messages to avoid confluent.value.schema.validation == true blocking them).
45
+ if chunk_number_int1 == 0:
46
+ dechunked_value_bytes += value_bytes
47
+ else:
48
+ dechunked_value_bytes += value_bytes[5:]
49
+ # Else just dechunk.
50
+ else:
51
+ dechunked_value_bytes += value_bytes
52
+ #
53
+ key_bytes = chunk_key_to_key(message_dict["key"])
54
+ #
55
+ # Delete the header fields for chunking.
56
+ del headers_str_bytes_dict["kafi_chunked_message_id"]
57
+ del headers_str_bytes_dict["kafi_number_of_chunks"]
58
+ del headers_str_bytes_dict["kafi_chunk_number"]
59
+ headers_str_bytes_tuple_list = list(headers_str_bytes_dict.items())
60
+ #
61
+ message_dict2 = {"value": dechunked_value_bytes,
62
+ "key": key_bytes,
63
+ "headers": headers_str_bytes_tuple_list,
64
+ "timestamp": message_dict["timestamp"],
65
+ "partition": message_dict["partition"],
66
+ "offset": message_dict["offset"],
67
+ "topic": message_dict["topic"]}
68
+ message_dict_list1.append(message_dict2)
69
+ #
70
+ # Clean up the chunks dictionary.
71
+ del self.chunks_dict[chunked_message_id_str]
72
+ else:
73
+ message_dict_list1.append(message_dict)
74
+ #
75
+ return message_dict_list1
kafi/deserializer.py ADDED
@@ -0,0 +1,148 @@
1
+ import os
2
+ import importlib
3
+ import json
4
+ import sys
5
+ import tempfile
6
+ import uuid
7
+
8
+ from confluent_kafka.schema_registry.avro import AvroDeserializer
9
+ from confluent_kafka.schema_registry.json_schema import JSONDeserializer
10
+ from confluent_kafka.schema_registry.protobuf import ProtobufDeserializer
11
+ from confluent_kafka.serialization import MessageField, SerializationContext
12
+ from google.protobuf.json_format import MessageToDict
13
+
14
+ from kafi.schemaregistry import SchemaRegistry
15
+
16
+ class Deserializer(SchemaRegistry):
17
+ def __init__(self, schema_registry_config_dict, **kwargs):
18
+ self.deser_from_dict = kwargs["deser_from_dict"] if "deser_from_dict" in kwargs else None
19
+ self.deser_conf = kwargs["deser_conf"] if "deser_conf" in kwargs else None
20
+ self.deser_rule_conf = kwargs["deser_rule_conf"] if "deser_rule_conf" in kwargs else None
21
+ self.deser_rule_registry = kwargs["deser_rule_registry"] if "deser_rule_registry" in kwargs else None
22
+ self.deser_json_decode = kwargs["deser_json_decode"] if "deser_json_decode" in kwargs else None
23
+ self.deser_return_record_name = kwargs["deser_return_record_name"] if "deser_return_record_name" in kwargs else False
24
+ #
25
+ super().__init__(schema_registry_config_dict)
26
+
27
+ def deserialize(self, payload_bytes, type_str, topic_str, headers_dict, key_bool):
28
+ if type_str.lower() == "bytes":
29
+ deserialized_payload = self.bytes_to_bytes(payload_bytes)
30
+ elif type_str.lower() in ["str", "string"]:
31
+ deserialized_payload = self.bytes_to_str(payload_bytes)
32
+ elif type_str.lower() == "json":
33
+ deserialized_payload = self.bytes_to_dict(payload_bytes)
34
+ elif type_str.lower() == "avro":
35
+ deserialized_payload = self.bytes_avro_to_dict(payload_bytes, topic_str, headers_dict, key_bool)
36
+ elif type_str.lower() in ["jsonschema", "json_sr"]:
37
+ deserialized_payload = self.bytes_jsonschema_to_dict(payload_bytes, topic_str, headers_dict, key_bool)
38
+ elif type_str.lower() in ["protobuf", "pb"]:
39
+ deserialized_payload = self.bytes_protobuf_to_dict(payload_bytes, topic_str, headers_dict, key_bool)
40
+ else:
41
+ raise Exception("Only \"str\", \"bytes\", \"json\", \"protobuf\" (\"pb\"), \"avro\" and \"jsonschema\" (\"json_sr\") supported.")
42
+ #
43
+ return deserialized_payload
44
+
45
+ def bytes_to_str(self, bytes):
46
+ if bytes:
47
+ return bytes.decode("utf-8")
48
+ else:
49
+ return bytes
50
+
51
+ def bytes_to_bytes(self, bytes):
52
+ return bytes
53
+
54
+ def bytes_to_dict(self, bytes):
55
+ if bytes is None:
56
+ return None
57
+ #
58
+ return json.loads(bytes)
59
+
60
+ def bytes_avro_to_dict(self, bytes, topic_str, headers_dict, key_bool):
61
+ if bytes is None:
62
+ return None
63
+ #
64
+ schema_str = self.get_schema_str(bytes, headers_dict, key_bool)
65
+ #
66
+ avroDeserializer = AvroDeserializer(self.schemaRegistryClient, schema_str, self.deser_from_dict, self.deser_return_record_name, self.deser_conf, self.deser_rule_conf, self.deser_rule_registry)
67
+ serializationContext = SerializationContext(topic_str, MessageField.KEY if key_bool else MessageField.VALUE)
68
+ dict = avroDeserializer(bytes, serializationContext)
69
+ return dict
70
+
71
+ def bytes_jsonschema_to_dict(self, bytes, topic_str, headers_dict, key_bool):
72
+ if bytes is None:
73
+ return None
74
+ #
75
+ schema_str = self.get_schema_str(bytes, headers_dict, key_bool)
76
+ #
77
+ jsonDeserializer = JSONDeserializer(schema_str, self.deser_from_dict, None, self.deser_conf, self.deser_rule_conf, self.deser_rule_registry, self.deser_json_decode)
78
+ serializationContext = SerializationContext(topic_str, MessageField.KEY if key_bool else MessageField.VALUE)
79
+ dict = jsonDeserializer(bytes, serializationContext)
80
+ return dict
81
+
82
+ def bytes_protobuf_to_dict(self, bytes, topic_str, headers_dict, key_bool):
83
+ if bytes is None:
84
+ return None
85
+ #
86
+ schema_id_int = int.from_bytes(bytes[1:5], "big")
87
+ if schema_id_int in self.schema_id_int_generalizedProtocolMessageType_protobuf_schema_str_tuple_dict:
88
+ generalizedProtocolMessageType, protobuf_schema_str = self.schema_id_int_generalizedProtocolMessageType_protobuf_schema_str_tuple_dict[schema_id_int]
89
+ else:
90
+ generalizedProtocolMessageType, protobuf_schema_str = self.schema_id_int_to_generalizedProtocolMessageType_protobuf_schema_str_tuple(schema_id_int)
91
+ self.schema_id_int_generalizedProtocolMessageType_protobuf_schema_str_tuple_dict[schema_id_int] = (generalizedProtocolMessageType, protobuf_schema_str)
92
+ #
93
+ # Prevent: RuntimeError: ProtobufSerializer: the 'use.deprecated.format' configuration property must be explicitly set due to backward incompatibility with older confluent-kafka-python Protobuf producers and consumers. See the release notes for more details
94
+ if self.deser_conf is None:
95
+ self.deser_conf = {"use.deprecated.format": False}
96
+ protobufDeserializer = ProtobufDeserializer(generalizedProtocolMessageType, self.deser_conf, None, self.deser_rule_conf, self.deser_rule_registry)
97
+ serializationContext = SerializationContext(topic_str, MessageField.KEY if key_bool else MessageField.VALUE)
98
+ protobuf_message = protobufDeserializer(bytes, serializationContext)
99
+ dict = MessageToDict(protobuf_message)
100
+ return dict
101
+
102
+ # Helpers
103
+
104
+ def get_schema_str(self, bytes, headers_dict, key_bool):
105
+ schema_id_key_str = "__key_schema_id" if key_bool else "__value_schema_id"
106
+ #
107
+ if headers_dict is not None and schema_id_key_str in headers_dict:
108
+ # Get the Schema ID from headers_dict if available.
109
+ schema_guid_bytes = headers_dict[schema_id_key_str]
110
+ # Skip the version byte (\x01).
111
+ schema_guid_bytes1 = schema_guid_bytes[1:]
112
+ # Convert to UUID.
113
+ schema_guid_str = str(uuid.UUID(bytes=schema_guid_bytes1))
114
+ # Get the schema at last.
115
+ schema_dict = self.get_schema_by_guid(schema_guid_str)
116
+ else:
117
+ # Else get the Schema ID from the payload.
118
+ schema_id_int = int.from_bytes(bytes[1:5], "big")
119
+ schema_dict = self.get_schema(schema_id_int)
120
+ #
121
+ schema_str = schema_dict["schema_str"]
122
+ #
123
+ return schema_str
124
+
125
+ def schema_id_int_to_generalizedProtocolMessageType_protobuf_schema_str_tuple(self, schema_id_int):
126
+ schema_dict = self.get_schema(schema_id_int)
127
+ schema_str = schema_dict["schema_str"]
128
+ #
129
+ generalizedProtocolMessageType = self.schema_id_int_and_schema_str_to_generalizedProtocolMessageType(schema_id_int, schema_str)
130
+ #
131
+ return generalizedProtocolMessageType, schema_str
132
+
133
+ def schema_id_int_and_schema_str_to_generalizedProtocolMessageType(self, schema_id_int, schema_str):
134
+ path_str = f"/{tempfile.gettempdir()}/kafi/protobuf/{self.storage_obj.config_str}"
135
+ os.makedirs(path_str, exist_ok=True)
136
+ file_str = f"schema_{schema_id_int}.proto"
137
+ file_path_str = f"{path_str}/{file_str}"
138
+ with open(file_path_str, "w") as textIOWrapper:
139
+ textIOWrapper.write(schema_str)
140
+ #
141
+ import grpc_tools.protoc
142
+ grpc_tools.protoc.main(["protoc", f"-I{path_str}", f"--python_out={path_str}", f"{file_str}"])
143
+ #
144
+ sys.path.insert(1, path_str)
145
+ schema_module = importlib.import_module(f"schema_{schema_id_int}_pb2")
146
+ schema_name_str = list(schema_module.DESCRIPTOR.message_types_by_name.keys())[0]
147
+ generalizedProtocolMessageType = getattr(schema_module, schema_name_str)
148
+ return generalizedProtocolMessageType
kafi/files.py ADDED
@@ -0,0 +1,85 @@
1
+ import io
2
+ import pathlib
3
+
4
+ from kafi.pandas import Pandas
5
+
6
+ # Constants
7
+
8
+ ALL_MESSAGES = -1
9
+
10
+ #
11
+
12
+ #x = [{"name": "cookie", "calories": 500.0, "colour": "brown"}, {"name": "cake", "calories": 260.0, "colour": "white"}, {"name": "timtam", "calories": 80.0, "colour": "chocolate"}]
13
+
14
+ class Files(Pandas):
15
+ def topic_to_file(self, topic, fs_obj, file, n=ALL_MESSAGES, **kwargs):
16
+ if not fs_obj.__class__.__bases__[0].__name__== "FS":
17
+ raise Exception("The target must be a file system.")
18
+ #
19
+ file_str = file
20
+ #
21
+ suffix_str = pathlib.Path(file_str).suffix
22
+ if suffix_str not in [".csv", ".json", ".parquet", ".xlsx", ".xml", ".bytes"]:
23
+ raise Exception("Only \".csv\", \".json\", \".parquet\", \".xlsx\", \".xml\" and \".bytes\" supported.")
24
+ #
25
+ if suffix_str == ".bytes":
26
+ message_dict_list = self.cat(topic, n, type="bytes", **kwargs)
27
+ data_bytes = b""
28
+ for message_dict in message_dict_list:
29
+ value_bytes = message_dict["value"]
30
+ data_bytes += value_bytes + b"\n"
31
+ else:
32
+ df = self.topic_to_df(topic, n, **kwargs)
33
+ data_bytesIO = io.BytesIO()
34
+ #
35
+ if suffix_str == ".csv":
36
+ index_bool = kwargs["index"] if "index" in kwargs else False
37
+ df.to_csv(data_bytesIO, index=index_bool)
38
+ elif suffix_str == ".json":
39
+ index_bool = kwargs["index"] if "index" in kwargs else None
40
+ df.to_json(data_bytesIO, orient="records")
41
+ elif suffix_str == ".parquet":
42
+ index_bool = kwargs["index"] if "index" in kwargs else None
43
+ df.to_parquet(data_bytesIO, index=index_bool, engine="fastparquet")
44
+ elif suffix_str == ".xlsx":
45
+ index_bool = kwargs["index"] if "index" in kwargs else False
46
+ df.to_excel(data_bytesIO, index=index_bool)
47
+ elif suffix_str == ".xml":
48
+ index_bool = kwargs["index"] if "index" in kwargs else False
49
+ df.to_xml(data_bytesIO, index=index_bool)
50
+ #
51
+ data_bytes = data_bytesIO.getvalue()
52
+ #
53
+ file_abs_path_str = fs_obj.admin.get_file_abs_path_str(file_str)
54
+ fs_obj.admin.write_bytes(file_abs_path_str, data_bytes)
55
+ #
56
+ return len(data_bytes)
57
+
58
+ def file_to_topic(self, file, storage_obj, topic, n=ALL_MESSAGES, **kwargs):
59
+ if not self.__class__.__bases__[0].__name__== "FS":
60
+ raise Exception("The source must be a file system.")
61
+ #
62
+ import pandas as pd
63
+ #
64
+ file_str = file
65
+ #
66
+ suffix_str = pathlib.Path(file_str).suffix
67
+ if suffix_str not in [".csv", ".json", ".parquet", ".xlsx", ".xml"]:
68
+ raise Exception("Only \".csv\", \".json\", \".parquet\", \".xlsx\" and \".xml\" supported.")
69
+ #
70
+ file_abs_path_str = self.admin.get_file_abs_path_str(file_str)
71
+ data_bytes = self.admin.read_bytes(file_abs_path_str)
72
+ data_bytesIO = io.BytesIO(data_bytes)
73
+ #
74
+ if suffix_str == ".csv":
75
+ df = pd.read_csv(data_bytesIO)
76
+ elif suffix_str == ".json":
77
+ df = pd.read_json(data_bytesIO)
78
+ elif suffix_str == ".parquet":
79
+ df = pd.read_parquet(data_bytesIO)
80
+ elif suffix_str == ".xlsx":
81
+ df = pd.read_excel(data_bytesIO)
82
+ elif suffix_str == ".xml":
83
+ df = pd.read_xml(data_bytesIO)
84
+ #
85
+ return storage_obj.df_to_topic(df, topic, n, **kwargs)
kafi/fs/__init__.py ADDED
File without changes
File without changes