adc-streaming 2.3.1__tar.gz → 2.4.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {adc-streaming-2.3.1 → adc-streaming-2.4.0}/PKG-INFO +1 -1
- {adc-streaming-2.3.1 → adc-streaming-2.4.0}/adc/__init__.py +2 -2
- {adc-streaming-2.3.1 → adc-streaming-2.4.0}/adc/consumer.py +73 -9
- adc-streaming-2.4.0/adc/errors.py +102 -0
- {adc-streaming-2.3.1 → adc-streaming-2.4.0}/adc/oidc.py +6 -4
- {adc-streaming-2.3.1 → adc-streaming-2.4.0}/adc/producer.py +17 -8
- {adc-streaming-2.3.1 → adc-streaming-2.4.0}/adc_streaming.egg-info/PKG-INFO +1 -1
- {adc-streaming-2.3.1 → adc-streaming-2.4.0}/adc_streaming.egg-info/requires.txt +1 -1
- {adc-streaming-2.3.1 → adc-streaming-2.4.0}/setup.cfg +1 -0
- {adc-streaming-2.3.1 → adc-streaming-2.4.0}/setup.py +1 -1
- {adc-streaming-2.3.1 → adc-streaming-2.4.0}/tests/test_kafka_integration.py +194 -18
- adc-streaming-2.3.1/adc/errors.py +0 -54
- {adc-streaming-2.3.1 → adc-streaming-2.4.0}/.github/workflows/build.yml +0 -0
- {adc-streaming-2.3.1 → adc-streaming-2.4.0}/.gitignore +0 -0
- {adc-streaming-2.3.1 → adc-streaming-2.4.0}/LICENSE +0 -0
- {adc-streaming-2.3.1 → adc-streaming-2.4.0}/Makefile +0 -0
- {adc-streaming-2.3.1 → adc-streaming-2.4.0}/README.md +0 -0
- {adc-streaming-2.3.1 → adc-streaming-2.4.0}/adc/auth.py +0 -0
- {adc-streaming-2.3.1 → adc-streaming-2.4.0}/adc/io.py +0 -0
- {adc-streaming-2.3.1 → adc-streaming-2.4.0}/adc/kafka.py +0 -0
- {adc-streaming-2.3.1 → adc-streaming-2.4.0}/adc_streaming.egg-info/SOURCES.txt +0 -0
- {adc-streaming-2.3.1 → adc-streaming-2.4.0}/adc_streaming.egg-info/dependency_links.txt +0 -0
- {adc-streaming-2.3.1 → adc-streaming-2.4.0}/adc_streaming.egg-info/not-zip-safe +0 -0
- {adc-streaming-2.3.1 → adc-streaming-2.4.0}/adc_streaming.egg-info/top_level.txt +0 -0
- {adc-streaming-2.3.1 → adc-streaming-2.4.0}/doc/Makefile +0 -0
- {adc-streaming-2.3.1 → adc-streaming-2.4.0}/doc/_static/css/my_theme.css +0 -0
- {adc-streaming-2.3.1 → adc-streaming-2.4.0}/doc/_templates/layout.html +0 -0
- {adc-streaming-2.3.1 → adc-streaming-2.4.0}/doc/api/api.rst +0 -0
- {adc-streaming-2.3.1 → adc-streaming-2.4.0}/doc/api/streaming.rst +0 -0
- {adc-streaming-2.3.1 → adc-streaming-2.4.0}/doc/conf.py +0 -0
- {adc-streaming-2.3.1 → adc-streaming-2.4.0}/doc/index.rst +0 -0
- {adc-streaming-2.3.1 → adc-streaming-2.4.0}/doc/user/installation.rst +0 -0
- {adc-streaming-2.3.1 → adc-streaming-2.4.0}/doc/user/quickstart.rst +0 -0
- {adc-streaming-2.3.1 → adc-streaming-2.4.0}/pyproject.toml +0 -0
- {adc-streaming-2.3.1 → adc-streaming-2.4.0}/recipe/meta.yaml +0 -0
- {adc-streaming-2.3.1 → adc-streaming-2.4.0}/tests/test_auth.py +0 -0
- {adc-streaming-2.3.1 → adc-streaming-2.4.0}/tests/test_kafka.py +0 -0
|
@@ -1,8 +1,8 @@
|
|
|
1
1
|
try:
|
|
2
|
-
from importlib.metadata import
|
|
2
|
+
from importlib.metadata import PackageNotFoundError, version
|
|
3
3
|
except ImportError:
|
|
4
4
|
# NOTE: remove after dropping support for Python < 3.8
|
|
5
|
-
from importlib_metadata import
|
|
5
|
+
from importlib_metadata import PackageNotFoundError, version
|
|
6
6
|
|
|
7
7
|
try:
|
|
8
8
|
__version__ = version("adc-streaming")
|
|
@@ -1,10 +1,13 @@
|
|
|
1
1
|
import dataclasses
|
|
2
2
|
import enum
|
|
3
3
|
import logging
|
|
4
|
-
from datetime import timedelta
|
|
5
4
|
import threading
|
|
6
|
-
from typing import Dict, Iterable, Iterator, List, Optional, Set, Union
|
|
7
5
|
from collections import defaultdict
|
|
6
|
+
from datetime import datetime, timedelta
|
|
7
|
+
# Imports from typing are deprecated as of Python 3.9 but required for
|
|
8
|
+
# compatibility with earlier versions
|
|
9
|
+
from typing import (Collection, Dict, Iterable, Iterator, List, Optional, Set,
|
|
10
|
+
Union)
|
|
8
11
|
|
|
9
12
|
import confluent_kafka # type: ignore
|
|
10
13
|
import confluent_kafka.admin # type: ignore
|
|
@@ -14,6 +17,18 @@ from .errors import ErrorCallback, log_client_errors
|
|
|
14
17
|
from .oidc import set_oauth_cb
|
|
15
18
|
|
|
16
19
|
|
|
20
|
+
class LogicalOffset(enum.IntEnum):
|
|
21
|
+
BEGINNING = confluent_kafka.OFFSET_BEGINNING
|
|
22
|
+
EARLIEST = confluent_kafka.OFFSET_BEGINNING
|
|
23
|
+
|
|
24
|
+
END = confluent_kafka.OFFSET_END
|
|
25
|
+
LATEST = confluent_kafka.OFFSET_END
|
|
26
|
+
|
|
27
|
+
STORED = confluent_kafka.OFFSET_STORED
|
|
28
|
+
|
|
29
|
+
INVALID = confluent_kafka.OFFSET_INVALID
|
|
30
|
+
|
|
31
|
+
|
|
17
32
|
class Consumer:
|
|
18
33
|
conf: 'ConsumerConfig'
|
|
19
34
|
_consumer: confluent_kafka.Consumer
|
|
@@ -23,7 +38,7 @@ class Consumer:
|
|
|
23
38
|
self.logger = logging.getLogger("adc-streaming.consumer")
|
|
24
39
|
self.conf = conf
|
|
25
40
|
self._consumer = confluent_kafka.Consumer(conf._to_confluent_kafka())
|
|
26
|
-
# Workaround for https://github.com/
|
|
41
|
+
# Workaround for https://github.com/confluentinc/librdkafka/issues/3753#issuecomment-1058272987.
|
|
27
42
|
# FIXME: Remove once fixed upstream, or on removal of oauth_cb.
|
|
28
43
|
self._consumer.poll(0)
|
|
29
44
|
self._stop_event = threading.Event()
|
|
@@ -86,6 +101,27 @@ class Consumer:
|
|
|
86
101
|
else:
|
|
87
102
|
self._consumer.commit(msg, asynchronous=False)
|
|
88
103
|
|
|
104
|
+
def _offsets_for_position(self, partitions: Collection[confluent_kafka.TopicPartition],
|
|
105
|
+
position: Union[datetime, LogicalOffset]) \
|
|
106
|
+
-> List[confluent_kafka.TopicPartition]:
|
|
107
|
+
if isinstance(position, datetime):
|
|
108
|
+
offset = int(position.timestamp() * 1000)
|
|
109
|
+
elif isinstance(position, LogicalOffset):
|
|
110
|
+
offset = position
|
|
111
|
+
else:
|
|
112
|
+
raise TypeError("Only datetime objects and logical offsets supported")
|
|
113
|
+
|
|
114
|
+
_partitions = [
|
|
115
|
+
confluent_kafka.TopicPartition(topic=tp.topic, partition=tp.partition, offset=offset)
|
|
116
|
+
for tp in partitions
|
|
117
|
+
]
|
|
118
|
+
|
|
119
|
+
if isinstance(position, datetime):
|
|
120
|
+
self.logger.debug("looking up offsets for time")
|
|
121
|
+
return self._consumer.offsets_for_times(_partitions)
|
|
122
|
+
else:
|
|
123
|
+
return _partitions
|
|
124
|
+
|
|
89
125
|
def stop(self):
|
|
90
126
|
"""Stops the runloop of the consumer. Useful when running the
|
|
91
127
|
consumer in a different thread.
|
|
@@ -95,7 +131,8 @@ class Consumer:
|
|
|
95
131
|
def stream(self,
|
|
96
132
|
autocommit: bool = True,
|
|
97
133
|
batch_size: int = 100,
|
|
98
|
-
batch_timeout: timedelta = timedelta(seconds=1.0)
|
|
134
|
+
batch_timeout: timedelta = timedelta(seconds=1.0),
|
|
135
|
+
start_at: Union[datetime, LogicalOffset, None] = None
|
|
99
136
|
) -> Iterator[confluent_kafka.Message]:
|
|
100
137
|
"""Returns a stream which iterates over the messages in the topics
|
|
101
138
|
to which the client is subscribed.
|
|
@@ -112,6 +149,15 @@ class Consumer:
|
|
|
112
149
|
provide a full batch of batch_size messages. Higher values may be more
|
|
113
150
|
efficient, but may add latency.
|
|
114
151
|
|
|
152
|
+
start_at controls the location in every partition the client reads
|
|
153
|
+
from, overriding the configured default position or the stored offsets.
|
|
154
|
+
Either a special logical offset value (END, BEGINNING, STORE, INVALID)
|
|
155
|
+
or a datetime. Using a datetime will cause reading to start from the
|
|
156
|
+
first message *after* the specified datetime (or END if none exists).
|
|
157
|
+
Using this parameter will override any previously assigned but not
|
|
158
|
+
committed offsets if they are managed from outside adc. Passing None
|
|
159
|
+
(or leaving it unspecified) avoids reassignment.
|
|
160
|
+
|
|
115
161
|
If the consumer's configuration has read_forever set to False, then the
|
|
116
162
|
stream stops when the client has hit the last message in all partitions.
|
|
117
163
|
This set of partitions is calculated just once when iterate() is first
|
|
@@ -119,6 +165,11 @@ class Consumer:
|
|
|
119
165
|
behavior in this case.
|
|
120
166
|
|
|
121
167
|
"""
|
|
168
|
+
|
|
169
|
+
if start_at is not None:
|
|
170
|
+
assignment = self._consumer.assignment()
|
|
171
|
+
self._consumer.assign(self._offsets_for_position(assignment, start_at))
|
|
172
|
+
|
|
122
173
|
if self.conf.read_forever:
|
|
123
174
|
return self._stream_forever(autocommit, batch_size, batch_timeout)
|
|
124
175
|
else:
|
|
@@ -202,7 +253,10 @@ class Consumer:
|
|
|
202
253
|
self._consumer.close()
|
|
203
254
|
|
|
204
255
|
|
|
205
|
-
|
|
256
|
+
# Used to be called ConsumerStartPosition, though this was confusing because
|
|
257
|
+
# it only affects "auto.offset.reset" not the start position for a call to
|
|
258
|
+
# consume.
|
|
259
|
+
class ConsumerDefaultPosition(enum.Enum):
|
|
206
260
|
EARLIEST = 1
|
|
207
261
|
LATEST = 2
|
|
208
262
|
|
|
@@ -210,6 +264,11 @@ class ConsumerStartPosition(enum.Enum):
|
|
|
210
264
|
return self.name.lower()
|
|
211
265
|
|
|
212
266
|
|
|
267
|
+
# Alias to the old name
|
|
268
|
+
# TODO: Remove alias on the next breaking release
|
|
269
|
+
ConsumerStartPosition = ConsumerDefaultPosition
|
|
270
|
+
|
|
271
|
+
|
|
213
272
|
@dataclasses.dataclass
|
|
214
273
|
class ConsumerConfig:
|
|
215
274
|
broker_urls: List[str]
|
|
@@ -225,8 +284,13 @@ class ConsumerConfig:
|
|
|
225
284
|
# was marked done with consumer.mark_done, regardless of this setting. This
|
|
226
285
|
# is only used when the position in the stream is unknown.
|
|
227
286
|
#
|
|
228
|
-
#
|
|
229
|
-
|
|
287
|
+
# You can force reading at a logical offset or datetime with the "start_at"
|
|
288
|
+
# argument to Consumer.stream().
|
|
289
|
+
#
|
|
290
|
+
# This is specified via a ConsumerDefaultPosition value.
|
|
291
|
+
#
|
|
292
|
+
# TODO: rename on next breaking release
|
|
293
|
+
start_at: ConsumerDefaultPosition = ConsumerDefaultPosition.EARLIEST
|
|
230
294
|
|
|
231
295
|
# Authentication package to pass in to read from Kafka.
|
|
232
296
|
auth: Optional[SASLAuth] = None
|
|
@@ -269,13 +333,13 @@ class ConsumerConfig:
|
|
|
269
333
|
"reconnect.backoff.max.ms": as_ms(self.reconnect_max_time),
|
|
270
334
|
"reconnect.backoff.ms": as_ms(self.reconnect_backoff_time),
|
|
271
335
|
}
|
|
272
|
-
if self.start_at is
|
|
336
|
+
if self.start_at is ConsumerDefaultPosition.EARLIEST:
|
|
273
337
|
default_topic_config = config.get("default.topic.config", {})
|
|
274
338
|
default_topic_config = {
|
|
275
339
|
"auto.offset.reset": "EARLIEST",
|
|
276
340
|
}
|
|
277
341
|
config["default.topic.config"] = default_topic_config
|
|
278
|
-
elif self.start_at is
|
|
342
|
+
elif self.start_at is ConsumerDefaultPosition.LATEST:
|
|
279
343
|
# FIXME: librdkafka has a bug in offset handling - it caches
|
|
280
344
|
# "OFFSET_END", and will repeatedly move to the end of the
|
|
281
345
|
# topic. See https://github.com/edenhill/librdkafka/pull/2876 -
|
|
@@ -0,0 +1,102 @@
|
|
|
1
|
+
import logging
|
|
2
|
+
from typing import Callable
|
|
3
|
+
|
|
4
|
+
import confluent_kafka # type: ignore
|
|
5
|
+
|
|
6
|
+
logger = logging.getLogger("adc-streaming")
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
ErrorCallback = Callable[[confluent_kafka.KafkaError], None]
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
def log_client_errors(kafka_error: confluent_kafka.KafkaError):
|
|
13
|
+
if kafka_error.code() == confluent_kafka.KafkaError._ALL_BROKERS_DOWN:
|
|
14
|
+
# This error occurs very frequently. It's not nearly as fatal as it
|
|
15
|
+
# sounds: it really indicates that the client's broker metadata has
|
|
16
|
+
# timed out. It appears to get triggered in races during client
|
|
17
|
+
# shutdown, too. See https://github.com/edenhill/librdkafka/issues/2543
|
|
18
|
+
# for more background.
|
|
19
|
+
logger.warn("client is currently disconnected from all brokers")
|
|
20
|
+
else:
|
|
21
|
+
logger.error(f"internal kafka error: {kafka_error}")
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
DeliveryCallback = Callable[[confluent_kafka.KafkaError, confluent_kafka.Message], None]
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def log_delivery_errors(
|
|
28
|
+
kafka_error: confluent_kafka.KafkaError,
|
|
29
|
+
msg: confluent_kafka.Message) -> None:
|
|
30
|
+
if kafka_error is not None:
|
|
31
|
+
logger.error(f"delivery error: {kafka_error}")
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def raise_delivery_errors(kafka_error: confluent_kafka.KafkaError,
|
|
35
|
+
msg: confluent_kafka.Message) -> None:
|
|
36
|
+
if kafka_error is not None:
|
|
37
|
+
raise KafkaException.from_kafka_error(kafka_error, msg)
|
|
38
|
+
elif msg.error() is not None:
|
|
39
|
+
raise KafkaException.from_kafka_error(msg.error(), msg)
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def _get_topic_related_errors():
|
|
43
|
+
"""Build a set of all Kafka error codes which seem to relate to a specific topic.
|
|
44
|
+
|
|
45
|
+
This uses a list extracted from all documented error codes up to confluent_kafka v2.4,
|
|
46
|
+
but some of these errors did not exist or were not exposed in earlier versions.
|
|
47
|
+
To maintain backward compatibility, this function checks whether each error exists before
|
|
48
|
+
attempting to otherwise refer to it.
|
|
49
|
+
"""
|
|
50
|
+
err_names = [
|
|
51
|
+
"_UNKNOWN_TOPIC",
|
|
52
|
+
"_NO_OFFSET",
|
|
53
|
+
"_LOG_TRUNCATION",
|
|
54
|
+
"OFFSET_OUT_OF_RANGE",
|
|
55
|
+
"UNKNOWN_TOPIC_OR_PART",
|
|
56
|
+
"NOT_LEADER_FOR_PARTITION",
|
|
57
|
+
"TOPIC_EXCEPTION",
|
|
58
|
+
"NOT_ENOUGH_REPLICAS",
|
|
59
|
+
"NOT_ENOUGH_REPLICAS_AFTER_APPEND",
|
|
60
|
+
"INVALID_COMMIT_OFFSET_SIZE",
|
|
61
|
+
"TOPIC_AUTHORIZATION_FAILED",
|
|
62
|
+
"TOPIC_ALREADY_EXISTS",
|
|
63
|
+
"INVALID_PARTITIONS",
|
|
64
|
+
"INVALID_REPLICATION_FACTOR",
|
|
65
|
+
"INVALID_REPLICA_ASSIGNMENT",
|
|
66
|
+
"REASSIGNMENT_IN_PROGRESS",
|
|
67
|
+
"TOPIC_DELETION_DISABLED",
|
|
68
|
+
"OFFSET_NOT_AVAILABLE",
|
|
69
|
+
"PREFERRED_LEADER_NOT_AVAILABLE",
|
|
70
|
+
"NO_REASSIGNMENT_IN_PROGRESS",
|
|
71
|
+
"GROUP_SUBSCRIBED_TO_TOPIC",
|
|
72
|
+
"UNSTABLE_OFFSET_COMMIT",
|
|
73
|
+
"UNKNOWN_TOPIC_ID",
|
|
74
|
+
]
|
|
75
|
+
errors = set()
|
|
76
|
+
for name in err_names:
|
|
77
|
+
if hasattr(confluent_kafka.KafkaError, name):
|
|
78
|
+
errors.add(getattr(confluent_kafka.KafkaError, name))
|
|
79
|
+
else:
|
|
80
|
+
logger.debug(f"{name} does not exist in confluent_kafka version "
|
|
81
|
+
f"{confluent_kafka.__version__} ({confluent_kafka.libversion()})")
|
|
82
|
+
return errors
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
class KafkaException(Exception):
|
|
86
|
+
@classmethod
|
|
87
|
+
def from_kafka_error(cls, error, msg=None):
|
|
88
|
+
return cls(error, msg)
|
|
89
|
+
|
|
90
|
+
topic_related_errors = _get_topic_related_errors()
|
|
91
|
+
|
|
92
|
+
def __init__(self, error, msg=None):
|
|
93
|
+
self.error = error
|
|
94
|
+
self.name = error.name()
|
|
95
|
+
self.reason = error.str()
|
|
96
|
+
self.retriable = error.retriable()
|
|
97
|
+
self.fatal = error.fatal()
|
|
98
|
+
self.message = msg
|
|
99
|
+
ex_msg = f"Error communicating with Kafka: code={self.name} {self.reason}"
|
|
100
|
+
if msg and error.code() in KafkaException.topic_related_errors:
|
|
101
|
+
ex_msg += f" on topic {msg.topic()}"
|
|
102
|
+
super(KafkaException, self).__init__(ex_msg)
|
|
@@ -2,9 +2,11 @@ def set_oauth_cb(config):
|
|
|
2
2
|
"""Implement client support for KIP-768 OpenID Connect.
|
|
3
3
|
|
|
4
4
|
Apache Kafka 3.1.0 supports authentication using OpenID Client Credentials.
|
|
5
|
-
Native support for Python is
|
|
6
|
-
|
|
7
|
-
|
|
5
|
+
Native support for Python is still incomplete due to this issue:
|
|
6
|
+
https://github.com/confluentinc/librdkafka/issues/3751
|
|
7
|
+
|
|
8
|
+
Meanwhile, this is a pure Python implementation of the refresh token
|
|
9
|
+
callback.
|
|
8
10
|
"""
|
|
9
11
|
if config.pop('sasl.oauthbearer.method', None) != 'oidc':
|
|
10
12
|
return
|
|
@@ -16,7 +18,7 @@ def set_oauth_cb(config):
|
|
|
16
18
|
|
|
17
19
|
from authlib.integrations.requests_client import OAuth2Session
|
|
18
20
|
session = OAuth2Session(client_id, client_secret, scope=scope)
|
|
19
|
-
|
|
21
|
+
|
|
20
22
|
def oauth_cb(*_, **__):
|
|
21
23
|
token = session.fetch_token(
|
|
22
24
|
token_endpoint, grant_type='client_credentials')
|
|
@@ -1,9 +1,9 @@
|
|
|
1
1
|
import abc
|
|
2
|
-
from ast import comprehension
|
|
3
2
|
import dataclasses
|
|
4
3
|
import logging
|
|
5
4
|
from datetime import timedelta
|
|
6
5
|
from typing import Dict, List, Optional, Union
|
|
6
|
+
|
|
7
7
|
try: # this will work only in python >= 3.8
|
|
8
8
|
from typing import Literal
|
|
9
9
|
except ImportError:
|
|
@@ -27,22 +27,30 @@ class Producer:
|
|
|
27
27
|
self.conf = conf
|
|
28
28
|
self.logger.debug(f"connecting to producer with config {conf._to_confluent_kafka()}")
|
|
29
29
|
self._producer = confluent_kafka.Producer(conf._to_confluent_kafka())
|
|
30
|
-
# Workaround for https://github.com/
|
|
30
|
+
# Workaround for https://github.com/confluentinc/librdkafka/issues/3753#issuecomment-1058272987.
|
|
31
31
|
# FIXME: Remove once fixed upstream, or on removal of oauth_cb.
|
|
32
32
|
self._producer.poll(0)
|
|
33
33
|
|
|
34
34
|
def write(self,
|
|
35
35
|
msg: Union[bytes, 'Serializable'],
|
|
36
36
|
headers: Optional[Union[dict, list]] = None,
|
|
37
|
-
delivery_callback: Optional[DeliveryCallback] = log_delivery_errors
|
|
37
|
+
delivery_callback: Optional[DeliveryCallback] = log_delivery_errors,
|
|
38
|
+
topic: Optional[str] = None) -> None:
|
|
38
39
|
if isinstance(msg, Serializable):
|
|
39
40
|
msg = msg.serialize()
|
|
40
|
-
|
|
41
|
+
if topic is None:
|
|
42
|
+
if self.conf.topic is not None:
|
|
43
|
+
topic = self.conf.topic
|
|
44
|
+
else:
|
|
45
|
+
raise Exception("No topic specified for write: "
|
|
46
|
+
"Either configure a topic when constructing the Producer, "
|
|
47
|
+
"or specify the topic argument to write()")
|
|
48
|
+
self.logger.debug("writing message to %s", topic)
|
|
41
49
|
if delivery_callback is not None:
|
|
42
|
-
self._producer.produce(
|
|
50
|
+
self._producer.produce(topic, msg, headers=headers,
|
|
43
51
|
on_delivery=delivery_callback)
|
|
44
52
|
else:
|
|
45
|
-
self._producer.produce(
|
|
53
|
+
self._producer.produce(topic, msg, headers=headers)
|
|
46
54
|
|
|
47
55
|
def flush(self, timeout: timedelta = timedelta(seconds=10)) -> int:
|
|
48
56
|
"""Attempt to flush enqueued messages. Return the number of messages still
|
|
@@ -78,7 +86,7 @@ class Producer:
|
|
|
78
86
|
@dataclasses.dataclass
|
|
79
87
|
class ProducerConfig:
|
|
80
88
|
broker_urls: List[str]
|
|
81
|
-
topic: str
|
|
89
|
+
topic: Optional[str]
|
|
82
90
|
auth: Optional[SASLAuth] = None
|
|
83
91
|
error_callback: Optional[ErrorCallback] = log_client_errors
|
|
84
92
|
|
|
@@ -104,7 +112,8 @@ class ProducerConfig:
|
|
|
104
112
|
# between attempts to reconnect to Kafka.
|
|
105
113
|
reconnect_max_time: timedelta = timedelta(seconds=10)
|
|
106
114
|
|
|
107
|
-
compression_type: Optional[Union[Literal['gzip'], Literal['snappy'],
|
|
115
|
+
compression_type: Optional[Union[Literal['gzip'], Literal['snappy'],
|
|
116
|
+
Literal['lz4'], Literal['zstd']]] = None
|
|
108
117
|
|
|
109
118
|
# maximum message size, before compression
|
|
110
119
|
message_max_bytes: Optional[int] = None
|
|
@@ -4,7 +4,7 @@ from setuptools import setup
|
|
|
4
4
|
# requirements
|
|
5
5
|
install_requires = [
|
|
6
6
|
"authlib", # FIXME: drop after next release of confluent-kafka with OIDC support
|
|
7
|
-
"confluent-kafka",
|
|
7
|
+
"confluent-kafka >= 1.6.1, != 2.1.0, != 2.1.1",
|
|
8
8
|
"dataclasses ; python_version < '3.7'",
|
|
9
9
|
"importlib-metadata ; python_version < '3.8'",
|
|
10
10
|
"requests", # FIXME: drop after next release of confluent-kafka with OIDC support
|
|
@@ -2,7 +2,7 @@ import logging
|
|
|
2
2
|
import tempfile
|
|
3
3
|
import time
|
|
4
4
|
import unittest
|
|
5
|
-
from datetime import timedelta
|
|
5
|
+
from datetime import datetime, timedelta, timezone
|
|
6
6
|
from typing import List
|
|
7
7
|
|
|
8
8
|
import docker
|
|
@@ -59,10 +59,9 @@ class KafkaIntegrationTestCase(unittest.TestCase):
|
|
|
59
59
|
self.assertEqual(msg.topic(), topic)
|
|
60
60
|
self.assertEqual(msg.value(), b"can you hear me?")
|
|
61
61
|
|
|
62
|
-
|
|
63
|
-
def test_consume_from_end(self):
|
|
62
|
+
def test_reset_to_end(self):
|
|
64
63
|
# Write a few messages.
|
|
65
|
-
topic = "
|
|
64
|
+
topic = "test_reset_to_end"
|
|
66
65
|
simple_write_msgs(self.kafka, topic, [
|
|
67
66
|
"message 1",
|
|
68
67
|
"message 2",
|
|
@@ -79,15 +78,16 @@ class KafkaIntegrationTestCase(unittest.TestCase):
|
|
|
79
78
|
stream = consumer.stream()
|
|
80
79
|
|
|
81
80
|
# Now add messages after the "end"
|
|
81
|
+
time.sleep(0.5)
|
|
82
82
|
simple_write_msg(self.kafka, topic, "message 4")
|
|
83
|
-
|
|
83
|
+
time.sleep(0.5)
|
|
84
84
|
msg = next(stream)
|
|
85
85
|
self.assertEqual(msg.topic(), topic)
|
|
86
86
|
self.assertEqual(msg.value(), b"message 4")
|
|
87
87
|
|
|
88
|
-
def
|
|
88
|
+
def test_reset_to_beginning(self):
|
|
89
89
|
# Write a few messages.
|
|
90
|
-
topic = "
|
|
90
|
+
topic = "test_reset_to_beginning"
|
|
91
91
|
batch = [
|
|
92
92
|
"message 1",
|
|
93
93
|
"message 2",
|
|
@@ -164,24 +164,153 @@ class KafkaIntegrationTestCase(unittest.TestCase):
|
|
|
164
164
|
self.assertEqual(actual.value().decode(), expected)
|
|
165
165
|
|
|
166
166
|
# Start second consumer, also reading from earliest offset.
|
|
167
|
-
|
|
167
|
+
consumer_2a = adc.consumer.Consumer(adc.consumer.ConsumerConfig(
|
|
168
|
+
broker_urls=[self.kafka.address],
|
|
169
|
+
group_id="test_consumer_2",
|
|
170
|
+
auth=self.kafka.auth,
|
|
171
|
+
read_forever=False,
|
|
172
|
+
start_at=adc.consumer.ConsumerStartPosition.EARLIEST,
|
|
173
|
+
))
|
|
174
|
+
consumer_2a.subscribe(topic)
|
|
175
|
+
|
|
176
|
+
# read the topic using consumer_2a
|
|
177
|
+
stream_2a = consumer_2a.stream()
|
|
178
|
+
msgs_2a = [pair for pair in zip(batch_1, stream_2a)]
|
|
179
|
+
# end iteration early after batch 1
|
|
180
|
+
stream_2a.close()
|
|
181
|
+
|
|
182
|
+
# check that messages from only batch 1 were consumed
|
|
183
|
+
self.assertEqual(len(batch_1), len(msgs_2a))
|
|
184
|
+
for expected, actual in msgs_2a:
|
|
185
|
+
self.assertEqual(actual.topic(), topic)
|
|
186
|
+
self.assertEqual(actual.value().decode(), expected)
|
|
187
|
+
|
|
188
|
+
# commit autocommited indices
|
|
189
|
+
consumer_2a.close()
|
|
190
|
+
|
|
191
|
+
# Start another consumer with the same groupid
|
|
192
|
+
consumer_2b = adc.consumer.Consumer(adc.consumer.ConsumerConfig(
|
|
168
193
|
broker_urls=[self.kafka.address],
|
|
169
194
|
group_id="test_consumer_2",
|
|
170
195
|
auth=self.kafka.auth,
|
|
171
196
|
read_forever=False,
|
|
172
197
|
start_at=adc.consumer.ConsumerStartPosition.EARLIEST,
|
|
173
198
|
))
|
|
174
|
-
|
|
175
|
-
|
|
176
|
-
|
|
177
|
-
|
|
178
|
-
|
|
179
|
-
|
|
180
|
-
|
|
181
|
-
|
|
199
|
+
consumer_2b.subscribe(topic)
|
|
200
|
+
stream_2b = consumer_2b.stream(autocommit=False, start_at=adc.consumer.LogicalOffset.STORED)
|
|
201
|
+
|
|
202
|
+
# read the rest using consumer_2b
|
|
203
|
+
msgs_2b = [msg for msg in stream_2b]
|
|
204
|
+
|
|
205
|
+
# Now check that messages from only batch_2 were read.
|
|
206
|
+
assert consumer_2b._stop_event.is_set()
|
|
207
|
+
self.assertEqual(len(batch_2), len(msgs_2b))
|
|
208
|
+
for expected, actual in zip(batch_2, msgs_2b):
|
|
209
|
+
self.assertEqual(actual.topic(), topic)
|
|
210
|
+
self.assertEqual(actual.value().decode(), expected)
|
|
211
|
+
|
|
212
|
+
def test_consume_from_beginning(self):
|
|
213
|
+
# Write a few messages.
|
|
214
|
+
topic = "test_consume_from_beginning"
|
|
215
|
+
batch = [
|
|
216
|
+
"message 1",
|
|
217
|
+
"message 2",
|
|
218
|
+
"message 3",
|
|
219
|
+
"message 4",
|
|
220
|
+
]
|
|
221
|
+
simple_write_msgs(self.kafka, topic, batch)
|
|
222
|
+
|
|
223
|
+
consumer = adc.consumer.Consumer(adc.consumer.ConsumerConfig(
|
|
224
|
+
broker_urls=[self.kafka.address],
|
|
225
|
+
group_id="test_consumer",
|
|
226
|
+
auth=self.kafka.auth,
|
|
227
|
+
read_forever=False,
|
|
228
|
+
# Make reading start at the end by default
|
|
229
|
+
start_at=adc.consumer.ConsumerStartPosition.LATEST,
|
|
230
|
+
))
|
|
231
|
+
consumer.subscribe(topic)
|
|
232
|
+
# Request reading from the beginning
|
|
233
|
+
stream = consumer.stream(start_at=adc.consumer.LogicalOffset.BEGINNING)
|
|
234
|
+
msgs = [msg for msg in stream]
|
|
235
|
+
|
|
236
|
+
assert consumer._stop_event.is_set()
|
|
237
|
+
self.assertEqual(len(batch), len(msgs))
|
|
238
|
+
for expected, actual in zip(batch, msgs):
|
|
239
|
+
self.assertEqual(actual.topic(), topic)
|
|
240
|
+
self.assertEqual(actual.value().decode(), expected)
|
|
241
|
+
|
|
242
|
+
# Read again from the beginning
|
|
243
|
+
stream = consumer.stream(start_at=adc.consumer.LogicalOffset.BEGINNING)
|
|
244
|
+
msgs = [msg for msg in stream]
|
|
245
|
+
|
|
246
|
+
assert consumer._stop_event.is_set()
|
|
247
|
+
self.assertEqual(len(batch), len(msgs))
|
|
248
|
+
for expected, actual in zip(batch, msgs):
|
|
182
249
|
self.assertEqual(actual.topic(), topic)
|
|
183
250
|
self.assertEqual(actual.value().decode(), expected)
|
|
184
251
|
|
|
252
|
+
def test_consume_from_end(self):
|
|
253
|
+
# Write a few messages.
|
|
254
|
+
topic = "test_consume_from_end"
|
|
255
|
+
simple_write_msgs(self.kafka, topic, [
|
|
256
|
+
"message 1",
|
|
257
|
+
"message 2",
|
|
258
|
+
"message 3",
|
|
259
|
+
])
|
|
260
|
+
consumer = adc.consumer.Consumer(adc.consumer.ConsumerConfig(
|
|
261
|
+
broker_urls=[self.kafka.address],
|
|
262
|
+
group_id="test_consumer",
|
|
263
|
+
auth=self.kafka.auth,
|
|
264
|
+
# Make reading start at the beginning by default
|
|
265
|
+
start_at=adc.consumer.ConsumerStartPosition.EARLIEST,
|
|
266
|
+
))
|
|
267
|
+
consumer.subscribe(topic)
|
|
268
|
+
# Request reading from the end
|
|
269
|
+
stream = consumer.stream(start_at=adc.consumer.LogicalOffset.END)
|
|
270
|
+
|
|
271
|
+
# Now add messages after the "end"
|
|
272
|
+
time.sleep(0.5)
|
|
273
|
+
simple_write_msg(self.kafka, topic, "message 4")
|
|
274
|
+
time.sleep(0.5)
|
|
275
|
+
msg = next(stream)
|
|
276
|
+
self.assertEqual(msg.topic(), topic)
|
|
277
|
+
self.assertEqual(msg.value(), b"message 4")
|
|
278
|
+
|
|
279
|
+
def test_consume_from_datetime(self):
|
|
280
|
+
# Write a few messages.
|
|
281
|
+
topic = "test_consume_from_datetime"
|
|
282
|
+
simple_write_msgs(self.kafka, topic, [
|
|
283
|
+
"message 1",
|
|
284
|
+
"message 2",
|
|
285
|
+
"message 3",
|
|
286
|
+
])
|
|
287
|
+
# Wait a while, write, and wait some more
|
|
288
|
+
time.sleep(2)
|
|
289
|
+
client_middle_time = datetime.now()
|
|
290
|
+
time.sleep(2)
|
|
291
|
+
simple_write_msg(self.kafka, topic, "message 4")
|
|
292
|
+
time.sleep(1)
|
|
293
|
+
|
|
294
|
+
consumer = adc.consumer.Consumer(adc.consumer.ConsumerConfig(
|
|
295
|
+
broker_urls=[self.kafka.address],
|
|
296
|
+
group_id="test_consumer",
|
|
297
|
+
auth=self.kafka.auth,
|
|
298
|
+
read_forever=False,
|
|
299
|
+
start_at=adc.consumer.ConsumerStartPosition.EARLIEST,
|
|
300
|
+
))
|
|
301
|
+
consumer.subscribe(topic)
|
|
302
|
+
stream = consumer.stream()
|
|
303
|
+
timestamps = [datetime.fromtimestamp(msg.timestamp()[1] / 1000.0) for msg in stream]
|
|
304
|
+
|
|
305
|
+
middle_time = timestamps[2] + (timestamps[3] - timestamps[2]) / 2
|
|
306
|
+
diff = middle_time - client_middle_time
|
|
307
|
+
logger.info(f"Difference between client and received timestamps: {diff!s}")
|
|
308
|
+
|
|
309
|
+
stream = consumer.stream(start_at=middle_time)
|
|
310
|
+
msg = next(stream)
|
|
311
|
+
self.assertEqual(msg.topic(), topic)
|
|
312
|
+
self.assertEqual(msg.value(), b"message 4")
|
|
313
|
+
|
|
185
314
|
def test_consume_not_forever(self):
|
|
186
315
|
topic = "test_consume_not_forever"
|
|
187
316
|
simple_write_msg(self.kafka, topic, "message 1")
|
|
@@ -248,6 +377,43 @@ class KafkaIntegrationTestCase(unittest.TestCase):
|
|
|
248
377
|
self.assertEqual(messages[1].value(), b"message 2")
|
|
249
378
|
self.assertEqual(messages[2].value(), b"message 3")
|
|
250
379
|
|
|
380
|
+
def test_multi_topic_handling(self):
|
|
381
|
+
"""Use a single producer object to write messages to multiple topics,
|
|
382
|
+
and check that a consumer can receive them all.
|
|
383
|
+
|
|
384
|
+
"""
|
|
385
|
+
topics = ["test_multi_1", "test_multi_2"]
|
|
386
|
+
|
|
387
|
+
# Push some messages in
|
|
388
|
+
producer = adc.producer.Producer(adc.producer.ProducerConfig(
|
|
389
|
+
broker_urls=[self.kafka.address],
|
|
390
|
+
topic=None,
|
|
391
|
+
auth=self.kafka.auth,
|
|
392
|
+
))
|
|
393
|
+
for i in range(0,8):
|
|
394
|
+
producer.write(str(i), topic=topics[i%2])
|
|
395
|
+
producer.flush()
|
|
396
|
+
logger.info("messages sent")
|
|
397
|
+
|
|
398
|
+
# check that we receive the messages from the right topics
|
|
399
|
+
consumer = adc.consumer.Consumer(adc.consumer.ConsumerConfig(
|
|
400
|
+
broker_urls=[self.kafka.address],
|
|
401
|
+
group_id="test_consumer",
|
|
402
|
+
auth=self.kafka.auth,
|
|
403
|
+
))
|
|
404
|
+
consumer.subscribe(topics)
|
|
405
|
+
stream = consumer.stream()
|
|
406
|
+
total_messages = 0;
|
|
407
|
+
for msg in stream:
|
|
408
|
+
if msg.error() is not None:
|
|
409
|
+
raise Exception(msg.error())
|
|
410
|
+
idx = int(msg.value())
|
|
411
|
+
self.assertEqual(msg.topic(), topics[idx%2])
|
|
412
|
+
total_messages += 1
|
|
413
|
+
if total_messages == 8:
|
|
414
|
+
break
|
|
415
|
+
self.assertEqual(total_messages, 8)
|
|
416
|
+
|
|
251
417
|
|
|
252
418
|
class KafkaDockerConnection:
|
|
253
419
|
"""Holds connection information for communicating with a Kafka broker running
|
|
@@ -308,6 +474,8 @@ class KafkaDockerConnection:
|
|
|
308
474
|
if not addrs:
|
|
309
475
|
return None
|
|
310
476
|
ip = addrs[0]['HostIp']
|
|
477
|
+
if len(ip) == 0:
|
|
478
|
+
ip = "localhost"
|
|
311
479
|
port = addrs[0]['HostPort']
|
|
312
480
|
return f"{ip}:{port}"
|
|
313
481
|
|
|
@@ -373,8 +541,16 @@ class KafkaDockerConnection:
|
|
|
373
541
|
detach=True,
|
|
374
542
|
auto_remove=True,
|
|
375
543
|
network=self.net.name,
|
|
376
|
-
#
|
|
377
|
-
|
|
544
|
+
# Kafka insists on redirecting consumers to one of its advertised listeners,
|
|
545
|
+
# which it will get wrong if it is running in a private container network.
|
|
546
|
+
# To fix this, we need to tell it what to advertise, which means we must
|
|
547
|
+
# know what port will be visible from the host system, and we cannot use an
|
|
548
|
+
# ephemeral port, which would be known to us only after the container is
|
|
549
|
+
# started. Since we have to pick something, pick 9092, which means that
|
|
550
|
+
# these tests cannot run if there is already an instance of Kafka running on
|
|
551
|
+
# the same host.
|
|
552
|
+
ports={"9092/tcp": 9092},
|
|
553
|
+
command=["/root/runServer","--advertisedListener","SASL_SSL://localhost:9092"],
|
|
378
554
|
)
|
|
379
555
|
|
|
380
556
|
def get_or_create_docker_network(self):
|
|
@@ -1,54 +0,0 @@
|
|
|
1
|
-
import logging
|
|
2
|
-
from typing import Callable
|
|
3
|
-
|
|
4
|
-
import confluent_kafka # type: ignore
|
|
5
|
-
|
|
6
|
-
logger = logging.getLogger("adc-streaming")
|
|
7
|
-
|
|
8
|
-
|
|
9
|
-
ErrorCallback = Callable[[confluent_kafka.KafkaError], None]
|
|
10
|
-
|
|
11
|
-
|
|
12
|
-
def log_client_errors(kafka_error: confluent_kafka.KafkaError):
|
|
13
|
-
if kafka_error.code() == confluent_kafka.KafkaError._ALL_BROKERS_DOWN:
|
|
14
|
-
# This error occurs very frequently. It's not nearly as fatal as it
|
|
15
|
-
# sounds: it really indicates that the client's broker metadata has
|
|
16
|
-
# timed out. It appears to get triggered in races during client
|
|
17
|
-
# shutdown, too. See https://github.com/edenhill/librdkafka/issues/2543
|
|
18
|
-
# for more background.
|
|
19
|
-
logger.warn("client is currently disconnected from all brokers")
|
|
20
|
-
else:
|
|
21
|
-
logger.error(f"internal kafka error: {kafka_error}")
|
|
22
|
-
|
|
23
|
-
|
|
24
|
-
DeliveryCallback = Callable[[confluent_kafka.KafkaError, confluent_kafka.Message], None]
|
|
25
|
-
|
|
26
|
-
|
|
27
|
-
def log_delivery_errors(
|
|
28
|
-
kafka_error: confluent_kafka.KafkaError,
|
|
29
|
-
msg: confluent_kafka.Message) -> None:
|
|
30
|
-
if kafka_error is not None:
|
|
31
|
-
logger.error(f"delivery error: {kafka_error}")
|
|
32
|
-
|
|
33
|
-
|
|
34
|
-
def raise_delivery_errors(kafka_error: confluent_kafka.KafkaError,
|
|
35
|
-
msg: confluent_kafka.Message) -> None:
|
|
36
|
-
if kafka_error is not None:
|
|
37
|
-
raise KafkaException.from_kafka_error(kafka_error)
|
|
38
|
-
elif msg.error() is not None:
|
|
39
|
-
raise KafkaException.from_kafka_error(msg.error())
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
class KafkaException(Exception):
|
|
43
|
-
@classmethod
|
|
44
|
-
def from_kafka_error(cls, error):
|
|
45
|
-
return cls(error)
|
|
46
|
-
|
|
47
|
-
def __init__(self, error):
|
|
48
|
-
self.error = error
|
|
49
|
-
self.name = error.name()
|
|
50
|
-
self.reason = error.str()
|
|
51
|
-
self.retriable = error.retriable()
|
|
52
|
-
self.fatal = error.fatal()
|
|
53
|
-
msg = f"Error communicating with Kafka: code={self.name} {self.reason}"
|
|
54
|
-
super(KafkaException, self).__init__(msg)
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|