adc-streaming 0.0.0__py2-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
adc/__init__.py ADDED
@@ -0,0 +1,11 @@
1
+ try:
2
+ from importlib.metadata import PackageNotFoundError, version
3
+ except ImportError:
4
+ # NOTE: remove after dropping support for Python < 3.8
5
+ from importlib_metadata import PackageNotFoundError, version
6
+
7
+ try:
8
+ __version__ = version("adc-streaming")
9
+ except PackageNotFoundError:
10
+ # package is not installed
11
+ pass
adc/auth.py ADDED
@@ -0,0 +1,87 @@
1
+ #!/usr/bin/env python
2
+
3
+ from enum import Enum
4
+
5
+ import certifi
6
+
7
+
8
+ class SASLMethod(Enum):
9
+ """SASL method to use for authentication.
10
+ """
11
+
12
+ PLAIN = 1
13
+ SCRAM_SHA_256 = 2
14
+ SCRAM_SHA_512 = 3
15
+ OAUTHBEARER = 4
16
+
17
+ def __str__(self):
18
+ return self.name.replace("_", "-")
19
+
20
+
21
+ class SASLAuth(object):
22
+ """Attach SASL-based authentication to a client.
23
+
24
+ Returns client-based auth options when called.
25
+
26
+ Parameters
27
+ ----------
28
+ user : `str`
29
+ Username to authenticate with.
30
+ password : `str`
31
+ Password to authenticate with.
32
+ ssl : `bool`, optional
33
+ Whether to enable SSL (enabled by default).
34
+ method : `SASLMethod`, optional
35
+ The SASL method to authenticate. The default is SASLMethod.OAUTHBEARER
36
+ if token_endpoint is provided, or SASLMethod.PLAIN otherwise.
37
+ See valid SASL methods in SASLMethod.
38
+ ssl_ca_location : `str`, optional
39
+ If using SSL via a self-signed cert, a path/location
40
+ to the certificate.
41
+ ssl_endpoint_identification_algorithm : `str`, optional
42
+ If using SSL, the algorithm used to verify that certificate is valid for the endpoint.
43
+ token_endpoint : `str`, optional
44
+ The OpenID Connect token endpoint URL.
45
+ Required for OAUTHBEARER / OpenID Connect, otherwise ignored.
46
+
47
+ """
48
+
49
+ def __init__(self, user, password, ssl=True, method=None, token_endpoint=None, **kwargs):
50
+ if method is None:
51
+ if token_endpoint is None:
52
+ method = SASLMethod.PLAIN
53
+ else:
54
+ method = SASLMethod.OAUTHBEARER
55
+ self._method = method
56
+
57
+ # set up SSL options
58
+ if ssl:
59
+ if "ssl_ca_location" in kwargs:
60
+ ssl_cert = kwargs["ssl_ca_location"]
61
+ else:
62
+ ssl_cert = certifi.where()
63
+
64
+ self._config = {
65
+ "security.protocol": "SASL_SSL",
66
+ "ssl.ca.location": ssl_cert,
67
+ "https.ca.location": ssl_cert,
68
+ }
69
+ if "ssl_endpoint_identification_algorithm" in kwargs:
70
+ self._config["ssl.endpoint.identification.algorithm"] = \
71
+ kwargs["ssl_endpoint_identification_algorithm"]
72
+ else:
73
+ self._config = {"security.protocol": "SASL_PLAINTEXT"}
74
+
75
+ # set up SASL options
76
+ self._config["sasl.mechanism"] = str(self._method)
77
+ if token_endpoint:
78
+ self._config["sasl.oauthbearer.client.id"] = user
79
+ self._config["sasl.oauthbearer.client.secret"] = password
80
+ self._config["sasl.oauthbearer.method"] = "oidc"
81
+ self._config["sasl.oauthbearer.token.endpoint.url"] = token_endpoint
82
+ else:
83
+ self._config["sasl.username"] = user
84
+ self._config["sasl.password"] = password
85
+
86
+ def __call__(self):
87
+ return self._config
adc/consumer.py ADDED
@@ -0,0 +1,362 @@
1
+ import dataclasses
2
+ import enum
3
+ import logging
4
+ import threading
5
+ from collections import defaultdict
6
+ from datetime import datetime, timedelta
7
+ # Imports from typing are deprecated as of Python 3.9 but required for
8
+ # compatibility with earlier versions
9
+ from typing import (Collection, Dict, Iterable, Iterator, List, Optional, Set,
10
+ Union)
11
+
12
+ import confluent_kafka # type: ignore
13
+ import confluent_kafka.admin # type: ignore
14
+
15
+ from .auth import SASLAuth
16
+ from .errors import ErrorCallback, log_client_errors
17
+
18
+
19
+ class LogicalOffset(enum.IntEnum):
20
+ BEGINNING = confluent_kafka.OFFSET_BEGINNING
21
+ EARLIEST = confluent_kafka.OFFSET_BEGINNING
22
+
23
+ END = confluent_kafka.OFFSET_END
24
+ LATEST = confluent_kafka.OFFSET_END
25
+
26
+ STORED = confluent_kafka.OFFSET_STORED
27
+
28
+ INVALID = confluent_kafka.OFFSET_INVALID
29
+
30
+
31
+ class Consumer:
32
+ conf: 'ConsumerConfig'
33
+ _consumer: confluent_kafka.Consumer
34
+ logger: logging.Logger
35
+
36
+ def __init__(self, conf: 'ConsumerConfig') -> None:
37
+ self.logger = logging.getLogger("adc-streaming.consumer")
38
+ self.conf = conf
39
+ self._consumer = confluent_kafka.Consumer(conf._to_confluent_kafka())
40
+ # Workaround for
41
+ # https://github.com/confluentinc/librdkafka/issues/3753#issuecomment-1058272987.
42
+ # FIXME: Remove once fixed upstream, or on removal of oauth_cb.
43
+ self._consumer.poll(0)
44
+ self._stop_event = threading.Event()
45
+
46
+ def subscribe(self,
47
+ topics: Union[str, Iterable],
48
+ timeout: timedelta = timedelta(seconds=10)):
49
+ """Subscribes to topics for consuming. This method doesn't use Kafka's
50
+ Consumer Groups; it assigns all partitions manually to this
51
+ process.
52
+
53
+ The topics must already exist for the subscription to succeed.
54
+ """
55
+ if isinstance(topics, str):
56
+ topics = [topics]
57
+
58
+ assignment = []
59
+ for topic in topics:
60
+ self.logger.debug(f"subscribing to topic {topic}")
61
+
62
+ try:
63
+ topic_meta = self.describe_topic(topic, timeout)
64
+ except KeyError:
65
+ raise ValueError(f"topic {topic} does not exist on the broker, so can't subscribe")
66
+
67
+ for partition_id in topic_meta.partitions.keys():
68
+ self.logger.debug(f"adding subscription to topic partition={partition_id}")
69
+ tp = confluent_kafka.TopicPartition(
70
+ topic=topic,
71
+ partition=partition_id,
72
+ )
73
+ assignment.append(tp)
74
+
75
+ self.logger.debug("registering topic assignment")
76
+ self._consumer.assign(assignment)
77
+
78
+ def describe_topic(
79
+ self,
80
+ topic: str,
81
+ timeout: timedelta = timedelta(seconds=5.0)) -> confluent_kafka.admin.TopicMetadata:
82
+ """Fetch confluent_kafka.admin.TopicMetadata describing a topic.
83
+ """
84
+ self.logger.debug(f"fetching cluster metadata to describe topic name={topic}")
85
+ cluster_meta = self._consumer.list_topics(timeout=timeout.total_seconds())
86
+ self.logger.debug(f"cluster metadata: {cluster_meta.topics}")
87
+ return cluster_meta.topics[topic]
88
+
89
+ def mark_done(self, msg: confluent_kafka.Message, asynchronous: bool = True):
90
+ """
91
+ Mark a message as fully-processed. In the background, the client will
92
+ continuously synchronize this information with Kafka so that the stream can be
93
+ resumed from this point in the future.
94
+
95
+ If asynchronous is set to False, however, the information will be sent
96
+ to Kafka immediately. This option allows fine-grained reliablity, but
97
+ can seriously reduce throughput if used on every message.
98
+ """
99
+ if asynchronous:
100
+ self._consumer.store_offsets(msg)
101
+ else:
102
+ self._consumer.commit(msg, asynchronous=False)
103
+
104
+ def _offsets_for_position(self, partitions: Collection[confluent_kafka.TopicPartition],
105
+ position: Union[datetime, LogicalOffset]) \
106
+ -> List[confluent_kafka.TopicPartition]:
107
+ if isinstance(position, datetime):
108
+ offset = int(position.timestamp() * 1000)
109
+ elif isinstance(position, LogicalOffset):
110
+ offset = position
111
+ else:
112
+ raise TypeError("Only datetime objects and logical offsets supported")
113
+
114
+ _partitions = [
115
+ confluent_kafka.TopicPartition(topic=tp.topic, partition=tp.partition, offset=offset)
116
+ for tp in partitions
117
+ ]
118
+
119
+ if isinstance(position, datetime):
120
+ self.logger.debug("looking up offsets for time")
121
+ return self._consumer.offsets_for_times(_partitions)
122
+ else:
123
+ return _partitions
124
+
125
+ def stop(self):
126
+ """Stops the runloop of the consumer. Useful when running the
127
+ consumer in a different thread.
128
+ """
129
+ self._stop_event.set()
130
+
131
+ def stream(self,
132
+ autocommit: bool = True,
133
+ batch_size: int = 100,
134
+ batch_timeout: timedelta = timedelta(seconds=1.0),
135
+ start_at: Union[datetime, LogicalOffset, None] = None
136
+ ) -> Iterator[confluent_kafka.Message]:
137
+ """Returns a stream which iterates over the messages in the topics
138
+ to which the client is subscribed.
139
+
140
+ If autocommit is true, then messages are automatically marked as handled
141
+ when they are yielded. This removes the need to call
142
+ 'mark_done' on each message. Callers using asynchronous message
143
+ processing or with complex processing needs should disable this.
144
+
145
+ batch_size controls the number of messages to request from Kafka per
146
+ batch. Higher values may be more efficient, but may add latency.
147
+
148
+ batch_timeout controls how long the client should wait for Kafka to
149
+ provide a full batch of batch_size messages. Higher values may be more
150
+ efficient, but may add latency.
151
+
152
+ start_at controls the location in every partition the client reads
153
+ from, overriding the configured default position or the stored offsets.
154
+ Either a special logical offset value (END, BEGINNING, STORE, INVALID)
155
+ or a datetime. Using a datetime will cause reading to start from the
156
+ first message *after* the specified datetime (or END if none exists).
157
+ Using this parameter will override any previously assigned but not
158
+ committed offsets if they are managed from outside adc. Passing None
159
+ (or leaving it unspecified) avoids reassignment.
160
+
161
+ If the consumer's configuration has read_forever set to False, then the
162
+ stream stops when the client has hit the last message in all partitions.
163
+ This set of partitions is calculated just once when iterate() is first
164
+ called; calling subscribe() after iterate() may cause inconsistent
165
+ behavior in this case.
166
+
167
+ """
168
+
169
+ if start_at is not None:
170
+ assignment = self._consumer.assignment()
171
+ self._consumer.assign(self._offsets_for_position(assignment, start_at))
172
+
173
+ if self.conf.read_forever:
174
+ return self._stream_forever(autocommit, batch_size, batch_timeout)
175
+ else:
176
+ return self._stream_until_eof(autocommit, batch_size, batch_timeout)
177
+
178
+ def _stream_forever(self,
179
+ autocommit: bool = True,
180
+ batch_size: int = 100,
181
+ batch_timeout: timedelta = timedelta(seconds=1.0),
182
+ ) -> Iterator[confluent_kafka.Message]:
183
+ self._stop_event.clear()
184
+ while not self._stop_event.is_set():
185
+ try:
186
+ messages = self._consumer.consume(batch_size,
187
+ batch_timeout.total_seconds())
188
+ for m in messages:
189
+ if self._stop_event.is_set():
190
+ break
191
+ err = m.error()
192
+ if err is None:
193
+ self.logger.debug(f"read message from partition {m.partition()}")
194
+ # Automatically mark message as processed, if desired
195
+ if autocommit:
196
+ self.mark_done(m, asynchronous=True)
197
+ yield m
198
+ else:
199
+ raise (confluent_kafka.KafkaException(err))
200
+ finally:
201
+ if autocommit:
202
+ self._consumer.commit(asynchronous=True)
203
+
204
+ def _stream_until_eof(self,
205
+ autocommit: bool = True,
206
+ batch_size: int = 100,
207
+ batch_timeout: timedelta = timedelta(seconds=1.0),
208
+ ) -> Iterator[confluent_kafka.Message]:
209
+ assignment = self._consumer.assignment()
210
+
211
+ # Make a map of topic-name -> set of partition IDs we're assigned to.
212
+ # When we hit a partition EOF, remove that partition from the map.
213
+ active_partitions: defaultdict[str, Set[int]] = defaultdict(set)
214
+ for tp in assignment:
215
+ self.logger.debug(f"tracking until eof for topic={tp.topic} partition={tp.partition}")
216
+ active_partitions[tp.topic].add(tp.partition)
217
+
218
+ self._stop_event.clear()
219
+ while len(active_partitions) > 0 and not self._stop_event.is_set():
220
+ messages = self._consumer.consume(batch_size, batch_timeout.total_seconds())
221
+ try:
222
+ for m in messages:
223
+ if self._stop_event.is_set():
224
+ raise StopIteration
225
+ err = m.error()
226
+ # A new message may arrive from a previously removed topic/partition,
227
+ # in which case it must be re-added
228
+ partition_set = active_partitions[m.topic()]
229
+ partition_set.add(m.partition())
230
+
231
+ if err is None:
232
+ self.logger.debug(f"read message from partition {m.partition()}")
233
+ # Automatically mark message as processed, if desired
234
+ if autocommit:
235
+ self.mark_done(m, asynchronous=True)
236
+ yield m
237
+ elif err.code() == confluent_kafka.KafkaError._PARTITION_EOF:
238
+ self.logger.debug(f"eof for topic={m.topic()} partition={m.partition()}")
239
+ # Done with this partition, remove it
240
+ partition_set.remove(m.partition())
241
+ if len(partition_set) == 0:
242
+ # Done with all partitions for the topic, remove it
243
+ del active_partitions[m.topic()]
244
+ else:
245
+ raise (confluent_kafka.KafkaException(err))
246
+ finally:
247
+ if autocommit:
248
+ self._consumer.commit(asynchronous=True)
249
+ self._stop_event.set()
250
+
251
+ def close(self):
252
+ """ Close the consumer, ending its subscriptions. """
253
+ self._consumer.close()
254
+
255
+
256
+ # Used to be called ConsumerStartPosition, though this was confusing because
257
+ # it only affects "auto.offset.reset" not the start position for a call to
258
+ # consume.
259
+ class ConsumerDefaultPosition(enum.Enum):
260
+ EARLIEST = 1
261
+ LATEST = 2
262
+
263
+ def __str__(self):
264
+ return self.name.lower()
265
+
266
+
267
+ # Alias to the old name
268
+ # TODO: Remove alias on the next breaking release
269
+ ConsumerStartPosition = ConsumerDefaultPosition
270
+
271
+
272
+ @dataclasses.dataclass
273
+ class ConsumerConfig:
274
+ broker_urls: List[str]
275
+ group_id: str
276
+
277
+ # When we have reached the last message on a topic, should we hold the
278
+ # stream open to wait for more messages?
279
+ read_forever: bool = True
280
+
281
+ # When reading a topic for the first time, where should we start in the
282
+ # stream? Note that, if the topic has already been consumed under the
283
+ # provided group_id, then consumption will start after the last message that
284
+ # was marked done with consumer.mark_done, regardless of this setting. This
285
+ # is only used when the position in the stream is unknown.
286
+ #
287
+ # You can force reading at a logical offset or datetime with the "start_at"
288
+ # argument to Consumer.stream().
289
+ #
290
+ # This is specified via a ConsumerDefaultPosition value.
291
+ #
292
+ # TODO: rename on next breaking release
293
+ start_at: ConsumerDefaultPosition = ConsumerDefaultPosition.EARLIEST
294
+
295
+ # Authentication package to pass in to read from Kafka.
296
+ auth: Optional[SASLAuth] = None
297
+
298
+ # Callback to execute whenever an internal Kafka error occurs.
299
+ error_callback: Optional[ErrorCallback] = log_client_errors
300
+
301
+ # How often should we save our progress to Kafka?
302
+ offset_commit_interval: timedelta = timedelta(seconds=5)
303
+
304
+ # Whether ncoming message CRCs should be checked to detect corruption in
305
+ # transit. Enabling this option has a small CPU use/throughput cost.
306
+ check_crcs: bool = False
307
+
308
+ # reconnect_backoff_time is the time that the backend should initially wait
309
+ # before attempting to reconnect to Kafka if its connection fails.
310
+ # Repeated failures will cause the wait time to be increased exponentially,
311
+ # with a random variation, until reconnect_max_time is reached.
312
+ reconnect_backoff_time: timedelta = timedelta(milliseconds=100)
313
+
314
+ # reconnect_max_time is the longest time that the backend should wait
315
+ # between attempts to reconnect to Kafka.
316
+ reconnect_max_time: timedelta = timedelta(seconds=10)
317
+
318
+ def _to_confluent_kafka(self) -> Dict:
319
+ def as_ms(td: timedelta):
320
+ """Convert a timedelta object to a duration in milliseconds"""
321
+ return int(td.total_seconds() * 1000.0)
322
+
323
+ config = {
324
+ "bootstrap.servers": ",".join(self.broker_urls),
325
+ "check.crcs": self.check_crcs,
326
+ "error_cb": self.error_callback,
327
+ "group.id": self.group_id,
328
+ "enable.auto.commit": True,
329
+ "auto.commit.interval.ms": as_ms(self.offset_commit_interval),
330
+ "enable.auto.offset.store": False,
331
+ "queued.min.messages": 1000,
332
+ "enable.partition.eof": not self.read_forever,
333
+ "reconnect.backoff.max.ms": as_ms(self.reconnect_max_time),
334
+ "reconnect.backoff.ms": as_ms(self.reconnect_backoff_time),
335
+ }
336
+ if self.start_at is ConsumerDefaultPosition.EARLIEST:
337
+ default_topic_config = config.get("default.topic.config", {})
338
+ default_topic_config = {
339
+ "auto.offset.reset": "EARLIEST",
340
+ }
341
+ config["default.topic.config"] = default_topic_config
342
+ elif self.start_at is ConsumerDefaultPosition.LATEST:
343
+ # FIXME: librdkafka has a bug in offset handling - it caches
344
+ # "OFFSET_END", and will repeatedly move to the end of the
345
+ # topic. See https://github.com/edenhill/librdkafka/pull/2876 -
346
+ # it should get fixed in v1.5 of librdkafka.
347
+
348
+ librdkafka_version = confluent_kafka.libversion()[0]
349
+ if librdkafka_version < "1.5.0":
350
+ self.logger.warn(
351
+ "In librdkafka before v1.5, LATEST offsets have buggy behavior; you may "
352
+ f"not receive data (your librdkafka version is {librdkafka_version}). See "
353
+ "https://github.com/confluentinc/confluent-kafka-dotnet/issues/1254.")
354
+ default_topic_config = config.get("default.topic.config", {})
355
+ default_topic_config = {
356
+ "auto.offset.reset": "LATEST",
357
+ }
358
+ config["default.topic.config"] = default_topic_config
359
+
360
+ if self.auth is not None:
361
+ config.update(self.auth())
362
+ return config
adc/errors.py ADDED
@@ -0,0 +1,102 @@
1
+ import logging
2
+ from typing import Callable
3
+
4
+ import confluent_kafka # type: ignore
5
+
6
+ logger = logging.getLogger("adc-streaming")
7
+
8
+
9
+ ErrorCallback = Callable[[confluent_kafka.KafkaError], None]
10
+
11
+
12
+ def log_client_errors(kafka_error: confluent_kafka.KafkaError):
13
+ if kafka_error.code() == confluent_kafka.KafkaError._ALL_BROKERS_DOWN:
14
+ # This error occurs very frequently. It's not nearly as fatal as it
15
+ # sounds: it really indicates that the client's broker metadata has
16
+ # timed out. It appears to get triggered in races during client
17
+ # shutdown, too. See https://github.com/edenhill/librdkafka/issues/2543
18
+ # for more background.
19
+ logger.warn("client is currently disconnected from all brokers")
20
+ else:
21
+ logger.error(f"internal kafka error: {kafka_error}")
22
+
23
+
24
+ DeliveryCallback = Callable[[confluent_kafka.KafkaError, confluent_kafka.Message], None]
25
+
26
+
27
+ def log_delivery_errors(
28
+ kafka_error: confluent_kafka.KafkaError,
29
+ msg: confluent_kafka.Message) -> None:
30
+ if kafka_error is not None:
31
+ logger.error(f"delivery error: {kafka_error}")
32
+
33
+
34
+ def raise_delivery_errors(kafka_error: confluent_kafka.KafkaError,
35
+ msg: confluent_kafka.Message) -> None:
36
+ if kafka_error is not None:
37
+ raise KafkaException.from_kafka_error(kafka_error, msg)
38
+ elif msg.error() is not None:
39
+ raise KafkaException.from_kafka_error(msg.error(), msg)
40
+
41
+
42
+ def _get_topic_related_errors():
43
+ """Build a set of all Kafka error codes which seem to relate to a specific topic.
44
+
45
+ This uses a list extracted from all documented error codes up to confluent_kafka v2.4,
46
+ but some of these errors did not exist or were not exposed in earlier versions.
47
+ To maintain backward compatibility, this function checks whether each error exists before
48
+ attempting to otherwise refer to it.
49
+ """
50
+ err_names = [
51
+ "_UNKNOWN_TOPIC",
52
+ "_NO_OFFSET",
53
+ "_LOG_TRUNCATION",
54
+ "OFFSET_OUT_OF_RANGE",
55
+ "UNKNOWN_TOPIC_OR_PART",
56
+ "NOT_LEADER_FOR_PARTITION",
57
+ "TOPIC_EXCEPTION",
58
+ "NOT_ENOUGH_REPLICAS",
59
+ "NOT_ENOUGH_REPLICAS_AFTER_APPEND",
60
+ "INVALID_COMMIT_OFFSET_SIZE",
61
+ "TOPIC_AUTHORIZATION_FAILED",
62
+ "TOPIC_ALREADY_EXISTS",
63
+ "INVALID_PARTITIONS",
64
+ "INVALID_REPLICATION_FACTOR",
65
+ "INVALID_REPLICA_ASSIGNMENT",
66
+ "REASSIGNMENT_IN_PROGRESS",
67
+ "TOPIC_DELETION_DISABLED",
68
+ "OFFSET_NOT_AVAILABLE",
69
+ "PREFERRED_LEADER_NOT_AVAILABLE",
70
+ "NO_REASSIGNMENT_IN_PROGRESS",
71
+ "GROUP_SUBSCRIBED_TO_TOPIC",
72
+ "UNSTABLE_OFFSET_COMMIT",
73
+ "UNKNOWN_TOPIC_ID",
74
+ ]
75
+ errors = set()
76
+ for name in err_names:
77
+ if hasattr(confluent_kafka.KafkaError, name):
78
+ errors.add(getattr(confluent_kafka.KafkaError, name))
79
+ else:
80
+ logger.debug(f"{name} does not exist in confluent_kafka version "
81
+ f"{confluent_kafka.__version__} ({confluent_kafka.libversion()})")
82
+ return errors
83
+
84
+
85
+ class KafkaException(Exception):
86
+ @classmethod
87
+ def from_kafka_error(cls, error, msg=None):
88
+ return cls(error, msg)
89
+
90
+ topic_related_errors = _get_topic_related_errors()
91
+
92
+ def __init__(self, error, msg=None):
93
+ self.error = error
94
+ self.name = error.name()
95
+ self.reason = error.str()
96
+ self.retriable = error.retriable()
97
+ self.fatal = error.fatal()
98
+ self.message = msg
99
+ ex_msg = f"Error communicating with Kafka: code={self.name} {self.reason}"
100
+ if msg and error.code() in KafkaException.topic_related_errors:
101
+ ex_msg += f" on topic {msg.topic()}"
102
+ super(KafkaException, self).__init__(ex_msg)
adc/io.py ADDED
@@ -0,0 +1,74 @@
1
+ import logging
2
+ import warnings
3
+ from contextlib import contextmanager
4
+ from typing import Iterable, List, Optional, Union
5
+
6
+ import confluent_kafka # type: ignore
7
+
8
+ from adc import auth, consumer, kafka, producer
9
+
10
+ logger = logging.getLogger("adc-streaming")
11
+
12
+
13
+ def open(url: str,
14
+ mode: str = 'r',
15
+ auth: Optional[auth.SASLAuth] = None,
16
+ start_at: consumer.ConsumerStartPosition = consumer.ConsumerStartPosition.EARLIEST, # noqa: E501
17
+ read_forever: bool = True,
18
+ ) -> Union[producer.Producer, Iterable[confluent_kafka.Message]]:
19
+ group_id, broker_addresses, topics = kafka.parse_kafka_url(url)
20
+ logger.debug("connecting to addresses=%s group_id=%s topics=%s",
21
+ broker_addresses, group_id, topics)
22
+ if mode == "r":
23
+ if group_id is None:
24
+ raise ValueError("group ID must be set when in reader mode")
25
+ return _open_consumer(group_id, broker_addresses, topics, auth, start_at, read_forever)
26
+ elif mode == "w":
27
+ if len(topics) != 1:
28
+ raise ValueError("must specify exactly one topic in write mode")
29
+ if group_id is not None:
30
+ warnings.warn("group ID has no effect when opening a stream in write mode")
31
+ if start_at is not consumer.ConsumerStartPosition.EARLIEST:
32
+ warnings.warn("start_at has no effect when opening a stream in write mode")
33
+ if read_forever is not True:
34
+ warnings.warn("read_forever has no effect when opening a stream in write mode")
35
+ return _open_producer(broker_addresses, topics[0], auth)
36
+ else:
37
+ raise ValueError("mode must be either 'w' or 'r'")
38
+
39
+
40
+ @contextmanager
41
+ def _open_consumer(
42
+ group_id: str,
43
+ broker_addresses: List[str],
44
+ topics: List[str],
45
+ auth: Optional[auth.SASLAuth],
46
+ start_at: consumer.ConsumerStartPosition,
47
+ read_forever: bool,
48
+ ) -> Iterable[confluent_kafka.Message]:
49
+ client = consumer.Consumer(consumer.ConsumerConfig(
50
+ broker_urls=broker_addresses,
51
+ group_id=group_id,
52
+ auth=auth,
53
+ start_at=start_at,
54
+ read_forever=read_forever,
55
+ ))
56
+ for t in topics:
57
+ client.subscribe(t)
58
+
59
+ try:
60
+ yield client.stream()
61
+ finally:
62
+ client.close()
63
+
64
+
65
+ def _open_producer(
66
+ broker_addresses: List[str],
67
+ topic: str,
68
+ auth: Optional[auth.SASLAuth],
69
+ ) -> producer.Producer:
70
+ return producer.Producer(producer.ProducerConfig(
71
+ broker_urls=broker_addresses,
72
+ topic=topic,
73
+ auth=auth,
74
+ ))
adc/kafka.py ADDED
@@ -0,0 +1,30 @@
1
+ from urllib.parse import urlparse
2
+
3
+
4
+ def parse_kafka_url(val):
5
+ """Extracts the group ID, broker addresses, and topic names from a Kafka URL.
6
+
7
+ The URL should be in this form:
8
+ ``kafka://[groupid@]broker[,broker2[,...]]/topic[,topic2[,...]]``
9
+
10
+ The returned group ID and topic may be None if they aren't in the URL.
11
+
12
+ """
13
+ parsed = urlparse(val)
14
+ if parsed.scheme != "kafka":
15
+ raise ValueError("invalid kafka URL: must start with 'kafka://'")
16
+
17
+ split_netloc = parsed.netloc.split("@", maxsplit=1)
18
+ if len(split_netloc) == 2:
19
+ group_id = split_netloc[0]
20
+ broker_addresses = split_netloc[1].split(",")
21
+ else:
22
+ group_id = None
23
+ broker_addresses = split_netloc[0].split(",")
24
+
25
+ topics = parsed.path.lstrip("/")
26
+ if len(topics) == 0:
27
+ split_topics = None
28
+ else:
29
+ split_topics = topics.split(",")
30
+ return group_id, broker_addresses, split_topics
adc/oidc.py ADDED
@@ -0,0 +1,27 @@
1
+ def set_oauth_cb(config):
2
+ """Implement client support for KIP-768 OpenID Connect.
3
+
4
+ Apache Kafka 3.1.0 supports authentication using OpenID Client Credentials.
5
+ Native support for Python is still incomplete due to this issue:
6
+ https://github.com/confluentinc/librdkafka/issues/3751
7
+
8
+ Meanwhile, this is a pure Python implementation of the refresh token
9
+ callback.
10
+ """
11
+ if config.pop('sasl.oauthbearer.method', None) != 'oidc':
12
+ return
13
+
14
+ client_id = config.pop('sasl.oauthbearer.client.id')
15
+ client_secret = config.pop('sasl.oauthbearer.client.secret')
16
+ scope = config.pop('sasl.oauthbearer.scope', None)
17
+ token_endpoint = config.pop('sasl.oauthbearer.token.endpoint.url')
18
+
19
+ from authlib.integrations.requests_client import OAuth2Session
20
+ session = OAuth2Session(client_id, client_secret, scope=scope)
21
+
22
+ def oauth_cb(*_, **__):
23
+ token = session.fetch_token(
24
+ token_endpoint, grant_type='client_credentials')
25
+ return token['access_token'], token['expires_at']
26
+
27
+ config['oauth_cb'] = oauth_cb
adc/producer.py ADDED
@@ -0,0 +1,164 @@
1
+ import abc
2
+ import dataclasses
3
+ import logging
4
+ from datetime import timedelta
5
+ from typing import Dict, List, Optional, Union
6
+
7
+ try: # this will work only in python >= 3.8
8
+ from typing import Literal
9
+ except ImportError:
10
+ from typing_extensions import Literal
11
+
12
+ import confluent_kafka # type: ignore
13
+
14
+ from .auth import SASLAuth
15
+ from .errors import (DeliveryCallback, ErrorCallback, log_client_errors,
16
+ log_delivery_errors)
17
+
18
+
19
+ class Producer:
20
+ conf: 'ProducerConfig'
21
+ _producer: confluent_kafka.Producer
22
+ logger: logging.Logger
23
+
24
+ def __init__(self, conf: 'ProducerConfig') -> None:
25
+ self.logger = logging.getLogger("adc-streaming.producer")
26
+ self.conf = conf
27
+ self.logger.debug(f"connecting to producer with config {conf._to_confluent_kafka()}")
28
+ self._producer = confluent_kafka.Producer(conf._to_confluent_kafka())
29
+ # Workaround for
30
+ # https://github.com/confluentinc/librdkafka/issues/3753#issuecomment-1058272987.
31
+ # FIXME: Remove once fixed upstream, or on removal of oauth_cb.
32
+ self._producer.poll(0)
33
+
34
+ def write(self,
35
+ msg: Union[bytes, 'Serializable'],
36
+ headers: Optional[Union[dict, list]] = None,
37
+ delivery_callback: Optional[DeliveryCallback] = log_delivery_errors,
38
+ topic: Optional[str] = None,
39
+ key: Optional[Union[str, bytes]] = None) -> None:
40
+ if isinstance(msg, Serializable):
41
+ msg = msg.serialize()
42
+ if topic is None:
43
+ if self.conf.topic is not None:
44
+ topic = self.conf.topic
45
+ else:
46
+ raise Exception("No topic specified for write: "
47
+ "Either configure a topic when constructing the Producer, "
48
+ "or specify the topic argument to write()")
49
+ self.logger.debug("writing message to %s", topic)
50
+ produce_kwargs = {"headers": headers, "key": key}
51
+ if delivery_callback is not None:
52
+ produce_kwargs["on_delivery"] = delivery_callback
53
+ while True:
54
+ try:
55
+ self._producer.produce(topic, msg, **produce_kwargs)
56
+ break
57
+ except BufferError:
58
+ # It's hard to know what the size limit on the buffer is, so we will try to
59
+ # wait until the number of items in it decreases (or it is empty, in case all
60
+ # messages were successfully sent in the time it took to handle the error).
61
+ buffer_len = len(self._producer)
62
+ self.logger.debug(f"Blocking due to BufferError, buffer size: {buffer_len}")
63
+ while buffer_len > 0 and len(self._producer) == buffer_len:
64
+ self._producer.poll(0.01)
65
+
66
+ def queued_message_count(self):
67
+ """Get the number of messages waiting to be sent to the broker.
68
+ """
69
+ return len(self._producer)
70
+
71
+ def flush(self, timeout: timedelta = timedelta(seconds=10)) -> int:
72
+ """Attempt to flush enqueued messages. Return the number of messages still
73
+ enqueued after the attempt.
74
+
75
+ """
76
+ n = self._producer.flush(timeout.total_seconds())
77
+ if n > 0:
78
+ self.logger.debug("flushed messages, %d still enqueued", n)
79
+ else:
80
+ self.logger.debug("flushed all messages")
81
+ return n
82
+
83
+ def close(self, timeout: timedelta = timedelta(seconds=10)) -> int:
84
+ self.logger.debug("shutting down producer")
85
+ return self.flush(timeout)
86
+
87
+ def __enter__(self) -> 'Producer':
88
+ return self
89
+
90
+ def __exit__(self, type, value, traceback) -> bool:
91
+ if type == KeyboardInterrupt:
92
+ print("Aborted (CTRL-C).")
93
+ return True
94
+ if type is None and value is None and traceback is None:
95
+ n_unsent = self.close()
96
+ if n_unsent > 0:
97
+ raise Exception(f"{n_unsent} messages remain unsent, some data may have been lost!")
98
+ return False
99
+ return False
100
+
101
+
102
+ @dataclasses.dataclass
103
+ class ProducerConfig:
104
+ broker_urls: List[str]
105
+ topic: Optional[str]
106
+ auth: Optional[SASLAuth] = None
107
+ error_callback: Optional[ErrorCallback] = log_client_errors
108
+
109
+ # produce_timeout sets the maximum amount of time that the backend can take
110
+ # to send a message to Kafka. Use a value of 0 to never timeout.
111
+ produce_timeout: timedelta = timedelta(seconds=10)
112
+
113
+ # produce_backoff_time sets the time the backend will wait before retrying
114
+ # to send a message to Kafka. May not be less than one millisecond.
115
+ produce_backoff_time: timedelta = timedelta(milliseconds=100)
116
+
117
+ # use_idempotence instructs the backend whether to ensure that messages are
118
+ # recorded by the broker exactly once and in the order of production.
119
+ use_idempotence: bool = False
120
+
121
+ # reconnect_backoff_time is the time that the backend should initially wait
122
+ # before attempting to reconnect to Kafka if its connection fails.
123
+ # Repeated failures will cause the wait time to be increased exponentially,
124
+ # with a random variation, until reconnect_max_time is reached.
125
+ reconnect_backoff_time: timedelta = timedelta(milliseconds=100)
126
+
127
+ # reconnect_max_time is the longest time that the backend should wait
128
+ # between attempts to reconnect to Kafka.
129
+ reconnect_max_time: timedelta = timedelta(seconds=10)
130
+
131
+ compression_type: Optional[Union[Literal['gzip'], Literal['snappy'],
132
+ Literal['lz4'], Literal['zstd']]] = None
133
+
134
+ # maximum message size, before compression
135
+ message_max_bytes: Optional[int] = None
136
+
137
+ def _to_confluent_kafka(self) -> Dict:
138
+ def as_ms(td: timedelta):
139
+ """Convert a timedelta object to a duration in milliseconds"""
140
+ return int(td.total_seconds() * 1000.0)
141
+
142
+ if self.produce_backoff_time < timedelta(milliseconds=1):
143
+ raise ValueError("produce_backoff_time may not be less than one millisecond")
144
+ config = {
145
+ "bootstrap.servers": ",".join(self.broker_urls),
146
+ "enable.idempotence": self.use_idempotence,
147
+ "message.timeout.ms": as_ms(self.produce_timeout),
148
+ "reconnect.backoff.max.ms": as_ms(self.reconnect_max_time),
149
+ "reconnect.backoff.ms": as_ms(self.reconnect_backoff_time),
150
+ "retry.backoff.ms": as_ms(self.produce_backoff_time),
151
+ "compression.type": self.compression_type or 'none',
152
+ }
153
+ if self.message_max_bytes is not None:
154
+ config['message.max.bytes'] = self.message_max_bytes
155
+ if self.error_callback is not None:
156
+ config["error_cb"] = self.error_callback
157
+ if self.auth is not None:
158
+ config.update(self.auth())
159
+ return config
160
+
161
+
162
+ class Serializable(abc.ABC):
163
+ def serialize(self) -> bytes:
164
+ raise NotImplementedError()
@@ -0,0 +1,29 @@
1
+ BSD 3-Clause License
2
+
3
+ Copyright (c) 2019, Mario Juric
4
+ All rights reserved.
5
+
6
+ Redistribution and use in source and binary forms, with or without
7
+ modification, are permitted provided that the following conditions are met:
8
+
9
+ 1. Redistributions of source code must retain the above copyright notice, this
10
+ list of conditions and the following disclaimer.
11
+
12
+ 2. Redistributions in binary form must reproduce the above copyright notice,
13
+ this list of conditions and the following disclaimer in the documentation
14
+ and/or other materials provided with the distribution.
15
+
16
+ 3. Neither the name of the copyright holder nor the names of its
17
+ contributors may be used to endorse or promote products derived from
18
+ this software without specific prior written permission.
19
+
20
+ THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS"
21
+ AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
22
+ IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE
23
+ DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT HOLDER OR CONTRIBUTORS BE LIABLE
24
+ FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
25
+ DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR
26
+ SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER
27
+ CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY,
28
+ OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
29
+ OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
@@ -0,0 +1,85 @@
1
+ Metadata-Version: 2.1
2
+ Name: adc-streaming
3
+ Version: 0.0.0
4
+ Summary: Astronomy Data Commons streaming client libraries
5
+ Home-page: https://github.com/astronomy-commons/adc-streaming
6
+ Author: Astronomy Data Commons Team
7
+ Author-email: swnelson@uw.edu
8
+ License: BSD
9
+ Platform: UNKNOWN
10
+ Classifier: Programming Language :: Python :: 3
11
+ Classifier: License :: OSI Approved :: BSD License
12
+ Classifier: Development Status :: 3 - Alpha
13
+ Classifier: Operating System :: POSIX :: Linux
14
+ Classifier: Operating System :: MacOS :: MacOS X
15
+ Description-Content-Type: text/markdown
16
+ Requires-Dist: confluent-kafka (>=2.11.0)
17
+ Requires-Dist: tqdm
18
+ Requires-Dist: certifi (>=2020.04.05.1)
19
+ Requires-Dist: dataclasses ; python_version < "3.7"
20
+ Requires-Dist: importlib-metadata ; python_version < "3.8"
21
+ Requires-Dist: typing-extensions ; python_version < "3.8"
22
+ Provides-Extra: dev
23
+ Requires-Dist: autopep8 ; extra == 'dev'
24
+ Requires-Dist: docker ; extra == 'dev'
25
+ Requires-Dist: flake8 ; extra == 'dev'
26
+ Requires-Dist: isort ; extra == 'dev'
27
+ Requires-Dist: pytest ; extra == 'dev'
28
+ Requires-Dist: pytest-timeout ; extra == 'dev'
29
+ Requires-Dist: pytest-integration ; extra == 'dev'
30
+ Requires-Dist: sphinx ; extra == 'dev'
31
+ Requires-Dist: sphinx-rtd-theme ; extra == 'dev'
32
+ Requires-Dist: twine ; extra == 'dev'
33
+
34
+ # Astronomy Data Commons Streaming Client Libraries
35
+
36
+ Libraries making it easy to access astronomy data commons resources.
37
+
38
+ ## Developer notes
39
+
40
+ ### Setup
41
+
42
+ To prepare for development, run `pip install --editable ".[dev]"` from within
43
+ the repo directory. This will install all dependencies, including those using
44
+ during development workflows.
45
+
46
+ This project expects you to use a `pip`-centric workflow for development on the
47
+ project itself. If you're using conda, then use the conda environment's `pip` to
48
+ install development dependencies, as described above.
49
+
50
+ Integration tests require Docker to run a Kafka broker. The broker might have
51
+ network problems on OSX if you use Docker Desktop; run the tests in a Linux
52
+ virtual machine (like with VirtualBox) to get around this.
53
+
54
+ ### Code Workflow
55
+
56
+ Write code, making changes.
57
+
58
+ Use `make format` to reformat your code to comply with PEP8.
59
+
60
+ Use `make lint` to catch common mistakes.
61
+
62
+ Use `make test-quick` to run fast unit tests.
63
+
64
+ Use `make test` to run the full slow test suite, including integration tests.
65
+
66
+ Once satisfied with all four of those, push your changes and open a PR.
67
+
68
+ ### Tag, build, and upload to PyPI and Conda
69
+
70
+ Tag a new version:
71
+ ```
72
+ git tag -s -a v0.x.x
73
+ ```
74
+
75
+ Build and release:
76
+
77
+ ```
78
+ make pypi-dist
79
+ make pypi-dist-check
80
+ make pypi-upload
81
+ make conda-build
82
+ make conda-upload
83
+ ```
84
+
85
+
@@ -0,0 +1,13 @@
1
+ adc/__init__.py,sha256=IGK6EiDe-qcBjT56a39nxC8SAaGbnKNUvoMbyvfrsbQ,332
2
+ adc/auth.py,sha256=k1WX81lh56V7rFJU-BPtZAdnvftU_qVjqM-8lPnuVTs,2857
3
+ adc/consumer.py,sha256=sfwH3KDmRDeUdryEzX2XCzKBvt3w11vrl3UqRstahog,15723
4
+ adc/errors.py,sha256=3ummsAbA1fa23jLif0kdoYt-1GbtDo8gUqnVY5opfig,3731
5
+ adc/io.py,sha256=RdfQuhmPLL9wVvUWaggRbeGKUVq5gMFp-QMO4Du-z9o,2541
6
+ adc/kafka.py,sha256=jsFypqYbfJk_CMpznjgUK4lSL7QRuXzYKGObx5OQWtg,933
7
+ adc/oidc.py,sha256=WRb3Z8-psffSwfYtOEV6HsmtEfEMmmm90JaFwOUdPkA,1072
8
+ adc/producer.py,sha256=KwNlZBccFhHfTUsgEZhJ4HG8-ExelmwF0K_HzyUCwGo,6875
9
+ adc_streaming-0.0.0.dist-info/LICENSE,sha256=BzSloDstU84TGMF0A6S-daRS2sVHoDgo3e7yh666Hgg,1519
10
+ adc_streaming-0.0.0.dist-info/METADATA,sha256=gbYADPcYuwxHRjOCTYVYJyxoLksDPc0gMX12Embodc4,2602
11
+ adc_streaming-0.0.0.dist-info/WHEEL,sha256=pqI-DBMA-Z6OTNov1nVxs7mwm6Yj2kHZGNp_6krVn1E,92
12
+ adc_streaming-0.0.0.dist-info/top_level.txt,sha256=8quD7Fop9Wwsv5j59wJl1NsuB_dGLESpqC4Tk7V2Qew,4
13
+ adc_streaming-0.0.0.dist-info/RECORD,,
@@ -0,0 +1,5 @@
1
+ Wheel-Version: 1.0
2
+ Generator: bdist_wheel (0.33.1)
3
+ Root-Is-Purelib: true
4
+ Tag: py2-none-any
5
+
@@ -0,0 +1 @@
1
+ adc