adc-streaming 2.3.2__tar.gz → 2.5.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (37) hide show
  1. {adc-streaming-2.3.2 → adc-streaming-2.5.0}/.github/workflows/build.yml +1 -1
  2. {adc-streaming-2.3.2 → adc-streaming-2.5.0}/PKG-INFO +1 -1
  3. {adc-streaming-2.3.2 → adc-streaming-2.5.0}/adc/__init__.py +2 -2
  4. {adc-streaming-2.3.2 → adc-streaming-2.5.0}/adc/auth.py +5 -0
  5. {adc-streaming-2.3.2 → adc-streaming-2.5.0}/adc/consumer.py +76 -11
  6. adc-streaming-2.5.0/adc/errors.py +102 -0
  7. {adc-streaming-2.3.2 → adc-streaming-2.5.0}/adc/oidc.py +1 -1
  8. {adc-streaming-2.3.2 → adc-streaming-2.5.0}/adc/producer.py +19 -8
  9. {adc-streaming-2.3.2 → adc-streaming-2.5.0}/adc_streaming.egg-info/PKG-INFO +1 -1
  10. {adc-streaming-2.3.2 → adc-streaming-2.5.0}/setup.cfg +1 -0
  11. {adc-streaming-2.3.2 → adc-streaming-2.5.0}/tests/test_auth.py +1 -1
  12. {adc-streaming-2.3.2 → adc-streaming-2.5.0}/tests/test_kafka_integration.py +226 -21
  13. adc-streaming-2.3.2/adc/errors.py +0 -54
  14. {adc-streaming-2.3.2 → adc-streaming-2.5.0}/.gitignore +0 -0
  15. {adc-streaming-2.3.2 → adc-streaming-2.5.0}/LICENSE +0 -0
  16. {adc-streaming-2.3.2 → adc-streaming-2.5.0}/Makefile +0 -0
  17. {adc-streaming-2.3.2 → adc-streaming-2.5.0}/README.md +0 -0
  18. {adc-streaming-2.3.2 → adc-streaming-2.5.0}/adc/io.py +0 -0
  19. {adc-streaming-2.3.2 → adc-streaming-2.5.0}/adc/kafka.py +0 -0
  20. {adc-streaming-2.3.2 → adc-streaming-2.5.0}/adc_streaming.egg-info/SOURCES.txt +0 -0
  21. {adc-streaming-2.3.2 → adc-streaming-2.5.0}/adc_streaming.egg-info/dependency_links.txt +0 -0
  22. {adc-streaming-2.3.2 → adc-streaming-2.5.0}/adc_streaming.egg-info/not-zip-safe +0 -0
  23. {adc-streaming-2.3.2 → adc-streaming-2.5.0}/adc_streaming.egg-info/requires.txt +0 -0
  24. {adc-streaming-2.3.2 → adc-streaming-2.5.0}/adc_streaming.egg-info/top_level.txt +0 -0
  25. {adc-streaming-2.3.2 → adc-streaming-2.5.0}/doc/Makefile +0 -0
  26. {adc-streaming-2.3.2 → adc-streaming-2.5.0}/doc/_static/css/my_theme.css +0 -0
  27. {adc-streaming-2.3.2 → adc-streaming-2.5.0}/doc/_templates/layout.html +0 -0
  28. {adc-streaming-2.3.2 → adc-streaming-2.5.0}/doc/api/api.rst +0 -0
  29. {adc-streaming-2.3.2 → adc-streaming-2.5.0}/doc/api/streaming.rst +0 -0
  30. {adc-streaming-2.3.2 → adc-streaming-2.5.0}/doc/conf.py +0 -0
  31. {adc-streaming-2.3.2 → adc-streaming-2.5.0}/doc/index.rst +0 -0
  32. {adc-streaming-2.3.2 → adc-streaming-2.5.0}/doc/user/installation.rst +0 -0
  33. {adc-streaming-2.3.2 → adc-streaming-2.5.0}/doc/user/quickstart.rst +0 -0
  34. {adc-streaming-2.3.2 → adc-streaming-2.5.0}/pyproject.toml +0 -0
  35. {adc-streaming-2.3.2 → adc-streaming-2.5.0}/recipe/meta.yaml +0 -0
  36. {adc-streaming-2.3.2 → adc-streaming-2.5.0}/setup.py +0 -0
  37. {adc-streaming-2.3.2 → adc-streaming-2.5.0}/tests/test_kafka.py +0 -0
@@ -7,7 +7,7 @@ jobs:
7
7
  runs-on: ubuntu-latest
8
8
  strategy:
9
9
  matrix:
10
- python-version: [3.7, 3.8, 3.9]
10
+ python-version: ["3.9", "3.10", "3.11", "3.12", "3.13"]
11
11
 
12
12
  steps:
13
13
  - name: Check out the code
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.1
2
2
  Name: adc-streaming
3
- Version: 2.3.2
3
+ Version: 2.5.0
4
4
  Summary: Astronomy Data Commons streaming client libraries
5
5
  Home-page: https://github.com/astronomy-commons/adc-streaming
6
6
  Author: Astronomy Data Commons Team
@@ -1,8 +1,8 @@
1
1
  try:
2
- from importlib.metadata import version, PackageNotFoundError
2
+ from importlib.metadata import PackageNotFoundError, version
3
3
  except ImportError:
4
4
  # NOTE: remove after dropping support for Python < 3.8
5
- from importlib_metadata import version, PackageNotFoundError
5
+ from importlib_metadata import PackageNotFoundError, version
6
6
 
7
7
  try:
8
8
  __version__ = version("adc-streaming")
@@ -38,6 +38,8 @@ class SASLAuth(object):
38
38
  ssl_ca_location : `str`, optional
39
39
  If using SSL via a self-signed cert, a path/location
40
40
  to the certificate.
41
+ ssl_endpoint_identification_algorithm : `str`, optional
42
+ If using SSL, the algorithm used to verify that certificate is valid for the endpoint.
41
43
  token_endpoint : `str`, optional
42
44
  The OpenID Connect token endpoint URL.
43
45
  Required for OAUTHBEARER / OpenID Connect, otherwise ignored.
@@ -63,6 +65,9 @@ class SASLAuth(object):
63
65
  "security.protocol": "SASL_SSL",
64
66
  "ssl.ca.location": ssl_cert,
65
67
  }
68
+ if "ssl_endpoint_identification_algorithm" in kwargs:
69
+ self._config["ssl.endpoint.identification.algorithm"] = \
70
+ kwargs["ssl_endpoint_identification_algorithm"]
66
71
  else:
67
72
  self._config = {"security.protocol": "SASL_PLAINTEXT"}
68
73
 
@@ -1,10 +1,13 @@
1
1
  import dataclasses
2
2
  import enum
3
3
  import logging
4
- from datetime import timedelta
5
4
  import threading
6
- from typing import Dict, Iterable, Iterator, List, Optional, Set, Union
7
5
  from collections import defaultdict
6
+ from datetime import datetime, timedelta
7
+ # Imports from typing are deprecated as of Python 3.9 but required for
8
+ # compatibility with earlier versions
9
+ from typing import (Collection, Dict, Iterable, Iterator, List, Optional, Set,
10
+ Union)
8
11
 
9
12
  import confluent_kafka # type: ignore
10
13
  import confluent_kafka.admin # type: ignore
@@ -14,6 +17,18 @@ from .errors import ErrorCallback, log_client_errors
14
17
  from .oidc import set_oauth_cb
15
18
 
16
19
 
20
+ class LogicalOffset(enum.IntEnum):
21
+ BEGINNING = confluent_kafka.OFFSET_BEGINNING
22
+ EARLIEST = confluent_kafka.OFFSET_BEGINNING
23
+
24
+ END = confluent_kafka.OFFSET_END
25
+ LATEST = confluent_kafka.OFFSET_END
26
+
27
+ STORED = confluent_kafka.OFFSET_STORED
28
+
29
+ INVALID = confluent_kafka.OFFSET_INVALID
30
+
31
+
17
32
  class Consumer:
18
33
  conf: 'ConsumerConfig'
19
34
  _consumer: confluent_kafka.Consumer
@@ -23,7 +38,8 @@ class Consumer:
23
38
  self.logger = logging.getLogger("adc-streaming.consumer")
24
39
  self.conf = conf
25
40
  self._consumer = confluent_kafka.Consumer(conf._to_confluent_kafka())
26
- # Workaround for https://github.com/confluentinc/librdkafka/issues/3753#issuecomment-1058272987.
41
+ # Workaround for
42
+ # https://github.com/confluentinc/librdkafka/issues/3753#issuecomment-1058272987.
27
43
  # FIXME: Remove once fixed upstream, or on removal of oauth_cb.
28
44
  self._consumer.poll(0)
29
45
  self._stop_event = threading.Event()
@@ -86,6 +102,27 @@ class Consumer:
86
102
  else:
87
103
  self._consumer.commit(msg, asynchronous=False)
88
104
 
105
+ def _offsets_for_position(self, partitions: Collection[confluent_kafka.TopicPartition],
106
+ position: Union[datetime, LogicalOffset]) \
107
+ -> List[confluent_kafka.TopicPartition]:
108
+ if isinstance(position, datetime):
109
+ offset = int(position.timestamp() * 1000)
110
+ elif isinstance(position, LogicalOffset):
111
+ offset = position
112
+ else:
113
+ raise TypeError("Only datetime objects and logical offsets supported")
114
+
115
+ _partitions = [
116
+ confluent_kafka.TopicPartition(topic=tp.topic, partition=tp.partition, offset=offset)
117
+ for tp in partitions
118
+ ]
119
+
120
+ if isinstance(position, datetime):
121
+ self.logger.debug("looking up offsets for time")
122
+ return self._consumer.offsets_for_times(_partitions)
123
+ else:
124
+ return _partitions
125
+
89
126
  def stop(self):
90
127
  """Stops the runloop of the consumer. Useful when running the
91
128
  consumer in a different thread.
@@ -95,7 +132,8 @@ class Consumer:
95
132
  def stream(self,
96
133
  autocommit: bool = True,
97
134
  batch_size: int = 100,
98
- batch_timeout: timedelta = timedelta(seconds=1.0)
135
+ batch_timeout: timedelta = timedelta(seconds=1.0),
136
+ start_at: Union[datetime, LogicalOffset, None] = None
99
137
  ) -> Iterator[confluent_kafka.Message]:
100
138
  """Returns a stream which iterates over the messages in the topics
101
139
  to which the client is subscribed.
@@ -112,6 +150,15 @@ class Consumer:
112
150
  provide a full batch of batch_size messages. Higher values may be more
113
151
  efficient, but may add latency.
114
152
 
153
+ start_at controls the location in every partition the client reads
154
+ from, overriding the configured default position or the stored offsets.
155
+ Either a special logical offset value (END, BEGINNING, STORE, INVALID)
156
+ or a datetime. Using a datetime will cause reading to start from the
157
+ first message *after* the specified datetime (or END if none exists).
158
+ Using this parameter will override any previously assigned but not
159
+ committed offsets if they are managed from outside adc. Passing None
160
+ (or leaving it unspecified) avoids reassignment.
161
+
115
162
  If the consumer's configuration has read_forever set to False, then the
116
163
  stream stops when the client has hit the last message in all partitions.
117
164
  This set of partitions is calculated just once when iterate() is first
@@ -119,6 +166,11 @@ class Consumer:
119
166
  behavior in this case.
120
167
 
121
168
  """
169
+
170
+ if start_at is not None:
171
+ assignment = self._consumer.assignment()
172
+ self._consumer.assign(self._offsets_for_position(assignment, start_at))
173
+
122
174
  if self.conf.read_forever:
123
175
  return self._stream_forever(autocommit, batch_size, batch_timeout)
124
176
  else:
@@ -145,7 +197,7 @@ class Consumer:
145
197
  self.mark_done(m, asynchronous=True)
146
198
  yield m
147
199
  else:
148
- raise(confluent_kafka.KafkaException(err))
200
+ raise (confluent_kafka.KafkaException(err))
149
201
  finally:
150
202
  if autocommit:
151
203
  self._consumer.commit(asynchronous=True)
@@ -191,7 +243,7 @@ class Consumer:
191
243
  # Done with all partitions for the topic, remove it
192
244
  del active_partitions[m.topic()]
193
245
  else:
194
- raise(confluent_kafka.KafkaException(err))
246
+ raise (confluent_kafka.KafkaException(err))
195
247
  finally:
196
248
  if autocommit:
197
249
  self._consumer.commit(asynchronous=True)
@@ -202,7 +254,10 @@ class Consumer:
202
254
  self._consumer.close()
203
255
 
204
256
 
205
- class ConsumerStartPosition(enum.Enum):
257
+ # Used to be called ConsumerStartPosition, though this was confusing because
258
+ # it only affects "auto.offset.reset" not the start position for a call to
259
+ # consume.
260
+ class ConsumerDefaultPosition(enum.Enum):
206
261
  EARLIEST = 1
207
262
  LATEST = 2
208
263
 
@@ -210,6 +265,11 @@ class ConsumerStartPosition(enum.Enum):
210
265
  return self.name.lower()
211
266
 
212
267
 
268
+ # Alias to the old name
269
+ # TODO: Remove alias on the next breaking release
270
+ ConsumerStartPosition = ConsumerDefaultPosition
271
+
272
+
213
273
  @dataclasses.dataclass
214
274
  class ConsumerConfig:
215
275
  broker_urls: List[str]
@@ -225,8 +285,13 @@ class ConsumerConfig:
225
285
  # was marked done with consumer.mark_done, regardless of this setting. This
226
286
  # is only used when the position in the stream is unknown.
227
287
  #
228
- # This is specified as a logical offset via a ConsumerStartPosition value.
229
- start_at: ConsumerStartPosition = ConsumerStartPosition.EARLIEST
288
+ # You can force reading at a logical offset or datetime with the "start_at"
289
+ # argument to Consumer.stream().
290
+ #
291
+ # This is specified via a ConsumerDefaultPosition value.
292
+ #
293
+ # TODO: rename on next breaking release
294
+ start_at: ConsumerDefaultPosition = ConsumerDefaultPosition.EARLIEST
230
295
 
231
296
  # Authentication package to pass in to read from Kafka.
232
297
  auth: Optional[SASLAuth] = None
@@ -269,13 +334,13 @@ class ConsumerConfig:
269
334
  "reconnect.backoff.max.ms": as_ms(self.reconnect_max_time),
270
335
  "reconnect.backoff.ms": as_ms(self.reconnect_backoff_time),
271
336
  }
272
- if self.start_at is ConsumerStartPosition.EARLIEST:
337
+ if self.start_at is ConsumerDefaultPosition.EARLIEST:
273
338
  default_topic_config = config.get("default.topic.config", {})
274
339
  default_topic_config = {
275
340
  "auto.offset.reset": "EARLIEST",
276
341
  }
277
342
  config["default.topic.config"] = default_topic_config
278
- elif self.start_at is ConsumerStartPosition.LATEST:
343
+ elif self.start_at is ConsumerDefaultPosition.LATEST:
279
344
  # FIXME: librdkafka has a bug in offset handling - it caches
280
345
  # "OFFSET_END", and will repeatedly move to the end of the
281
346
  # topic. See https://github.com/edenhill/librdkafka/pull/2876 -
@@ -0,0 +1,102 @@
1
+ import logging
2
+ from typing import Callable
3
+
4
+ import confluent_kafka # type: ignore
5
+
6
+ logger = logging.getLogger("adc-streaming")
7
+
8
+
9
+ ErrorCallback = Callable[[confluent_kafka.KafkaError], None]
10
+
11
+
12
+ def log_client_errors(kafka_error: confluent_kafka.KafkaError):
13
+ if kafka_error.code() == confluent_kafka.KafkaError._ALL_BROKERS_DOWN:
14
+ # This error occurs very frequently. It's not nearly as fatal as it
15
+ # sounds: it really indicates that the client's broker metadata has
16
+ # timed out. It appears to get triggered in races during client
17
+ # shutdown, too. See https://github.com/edenhill/librdkafka/issues/2543
18
+ # for more background.
19
+ logger.warn("client is currently disconnected from all brokers")
20
+ else:
21
+ logger.error(f"internal kafka error: {kafka_error}")
22
+
23
+
24
+ DeliveryCallback = Callable[[confluent_kafka.KafkaError, confluent_kafka.Message], None]
25
+
26
+
27
+ def log_delivery_errors(
28
+ kafka_error: confluent_kafka.KafkaError,
29
+ msg: confluent_kafka.Message) -> None:
30
+ if kafka_error is not None:
31
+ logger.error(f"delivery error: {kafka_error}")
32
+
33
+
34
+ def raise_delivery_errors(kafka_error: confluent_kafka.KafkaError,
35
+ msg: confluent_kafka.Message) -> None:
36
+ if kafka_error is not None:
37
+ raise KafkaException.from_kafka_error(kafka_error, msg)
38
+ elif msg.error() is not None:
39
+ raise KafkaException.from_kafka_error(msg.error(), msg)
40
+
41
+
42
+ def _get_topic_related_errors():
43
+ """Build a set of all Kafka error codes which seem to relate to a specific topic.
44
+
45
+ This uses a list extracted from all documented error codes up to confluent_kafka v2.4,
46
+ but some of these errors did not exist or were not exposed in earlier versions.
47
+ To maintain backward compatibility, this function checks whether each error exists before
48
+ attempting to otherwise refer to it.
49
+ """
50
+ err_names = [
51
+ "_UNKNOWN_TOPIC",
52
+ "_NO_OFFSET",
53
+ "_LOG_TRUNCATION",
54
+ "OFFSET_OUT_OF_RANGE",
55
+ "UNKNOWN_TOPIC_OR_PART",
56
+ "NOT_LEADER_FOR_PARTITION",
57
+ "TOPIC_EXCEPTION",
58
+ "NOT_ENOUGH_REPLICAS",
59
+ "NOT_ENOUGH_REPLICAS_AFTER_APPEND",
60
+ "INVALID_COMMIT_OFFSET_SIZE",
61
+ "TOPIC_AUTHORIZATION_FAILED",
62
+ "TOPIC_ALREADY_EXISTS",
63
+ "INVALID_PARTITIONS",
64
+ "INVALID_REPLICATION_FACTOR",
65
+ "INVALID_REPLICA_ASSIGNMENT",
66
+ "REASSIGNMENT_IN_PROGRESS",
67
+ "TOPIC_DELETION_DISABLED",
68
+ "OFFSET_NOT_AVAILABLE",
69
+ "PREFERRED_LEADER_NOT_AVAILABLE",
70
+ "NO_REASSIGNMENT_IN_PROGRESS",
71
+ "GROUP_SUBSCRIBED_TO_TOPIC",
72
+ "UNSTABLE_OFFSET_COMMIT",
73
+ "UNKNOWN_TOPIC_ID",
74
+ ]
75
+ errors = set()
76
+ for name in err_names:
77
+ if hasattr(confluent_kafka.KafkaError, name):
78
+ errors.add(getattr(confluent_kafka.KafkaError, name))
79
+ else:
80
+ logger.debug(f"{name} does not exist in confluent_kafka version "
81
+ f"{confluent_kafka.__version__} ({confluent_kafka.libversion()})")
82
+ return errors
83
+
84
+
85
+ class KafkaException(Exception):
86
+ @classmethod
87
+ def from_kafka_error(cls, error, msg=None):
88
+ return cls(error, msg)
89
+
90
+ topic_related_errors = _get_topic_related_errors()
91
+
92
+ def __init__(self, error, msg=None):
93
+ self.error = error
94
+ self.name = error.name()
95
+ self.reason = error.str()
96
+ self.retriable = error.retriable()
97
+ self.fatal = error.fatal()
98
+ self.message = msg
99
+ ex_msg = f"Error communicating with Kafka: code={self.name} {self.reason}"
100
+ if msg and error.code() in KafkaException.topic_related_errors:
101
+ ex_msg += f" on topic {msg.topic()}"
102
+ super(KafkaException, self).__init__(ex_msg)
@@ -18,7 +18,7 @@ def set_oauth_cb(config):
18
18
 
19
19
  from authlib.integrations.requests_client import OAuth2Session
20
20
  session = OAuth2Session(client_id, client_secret, scope=scope)
21
-
21
+
22
22
  def oauth_cb(*_, **__):
23
23
  token = session.fetch_token(
24
24
  token_endpoint, grant_type='client_credentials')
@@ -1,9 +1,9 @@
1
1
  import abc
2
- from ast import comprehension
3
2
  import dataclasses
4
3
  import logging
5
4
  from datetime import timedelta
6
5
  from typing import Dict, List, Optional, Union
6
+
7
7
  try: # this will work only in python >= 3.8
8
8
  from typing import Literal
9
9
  except ImportError:
@@ -27,22 +27,32 @@ class Producer:
27
27
  self.conf = conf
28
28
  self.logger.debug(f"connecting to producer with config {conf._to_confluent_kafka()}")
29
29
  self._producer = confluent_kafka.Producer(conf._to_confluent_kafka())
30
- # Workaround for https://github.com/confluentinc/librdkafka/issues/3753#issuecomment-1058272987.
30
+ # Workaround for
31
+ # https://github.com/confluentinc/librdkafka/issues/3753#issuecomment-1058272987.
31
32
  # FIXME: Remove once fixed upstream, or on removal of oauth_cb.
32
33
  self._producer.poll(0)
33
34
 
34
35
  def write(self,
35
36
  msg: Union[bytes, 'Serializable'],
36
37
  headers: Optional[Union[dict, list]] = None,
37
- delivery_callback: Optional[DeliveryCallback] = log_delivery_errors) -> None:
38
+ delivery_callback: Optional[DeliveryCallback] = log_delivery_errors,
39
+ topic: Optional[str] = None,
40
+ key: Optional[Union[str, bytes]] = None) -> None:
38
41
  if isinstance(msg, Serializable):
39
42
  msg = msg.serialize()
40
- self.logger.debug("writing message to %s", self.conf.topic)
43
+ if topic is None:
44
+ if self.conf.topic is not None:
45
+ topic = self.conf.topic
46
+ else:
47
+ raise Exception("No topic specified for write: "
48
+ "Either configure a topic when constructing the Producer, "
49
+ "or specify the topic argument to write()")
50
+ self.logger.debug("writing message to %s", topic)
41
51
  if delivery_callback is not None:
42
- self._producer.produce(self.conf.topic, msg, headers=headers,
52
+ self._producer.produce(topic, msg, headers=headers, key=key,
43
53
  on_delivery=delivery_callback)
44
54
  else:
45
- self._producer.produce(self.conf.topic, msg, headers=headers)
55
+ self._producer.produce(topic, msg, headers=headers, key=key,)
46
56
 
47
57
  def flush(self, timeout: timedelta = timedelta(seconds=10)) -> int:
48
58
  """Attempt to flush enqueued messages. Return the number of messages still
@@ -78,7 +88,7 @@ class Producer:
78
88
  @dataclasses.dataclass
79
89
  class ProducerConfig:
80
90
  broker_urls: List[str]
81
- topic: str
91
+ topic: Optional[str]
82
92
  auth: Optional[SASLAuth] = None
83
93
  error_callback: Optional[ErrorCallback] = log_client_errors
84
94
 
@@ -104,7 +114,8 @@ class ProducerConfig:
104
114
  # between attempts to reconnect to Kafka.
105
115
  reconnect_max_time: timedelta = timedelta(seconds=10)
106
116
 
107
- compression_type: Optional[Union[Literal['gzip'], Literal['snappy'], Literal['lz4'], Literal['zstd']]] = None
117
+ compression_type: Optional[Union[Literal['gzip'], Literal['snappy'],
118
+ Literal['lz4'], Literal['zstd']]] = None
108
119
 
109
120
  # maximum message size, before compression
110
121
  message_max_bytes: Optional[int] = None
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.1
2
2
  Name: adc-streaming
3
- Version: 2.3.2
3
+ Version: 2.5.0
4
4
  Summary: Astronomy Data Commons streaming client libraries
5
5
  Home-page: https://github.com/astronomy-commons/adc-streaming
6
6
  Author: Astronomy Data Commons Team
@@ -13,6 +13,7 @@ exclude = setup.py,
13
13
  [tool:pytest]
14
14
  log_cli = True
15
15
  log_cli_level = INFO
16
+ testpaths = tests
16
17
 
17
18
  [egg_info]
18
19
  tag_build =
@@ -1,6 +1,6 @@
1
1
  import pytest
2
2
 
3
- from adc.auth import SASLAuth, SASLMethod
3
+ from adc.auth import SASLAuth
4
4
 
5
5
 
6
6
  @pytest.mark.parametrize('auth,expected_config', [
@@ -2,8 +2,8 @@ import logging
2
2
  import tempfile
3
3
  import time
4
4
  import unittest
5
- from datetime import timedelta
6
- from typing import List
5
+ from datetime import datetime, timedelta
6
+ from typing import List, Optional
7
7
 
8
8
  import docker
9
9
  import pytest
@@ -59,10 +59,34 @@ class KafkaIntegrationTestCase(unittest.TestCase):
59
59
  self.assertEqual(msg.topic(), topic)
60
60
  self.assertEqual(msg.value(), b"can you hear me?")
61
61
 
62
- @unittest.skip("skipping due to bug in librdkafka")
63
- def test_consume_from_end(self):
62
+ def test_message_with_key(self):
63
+ """Try writing a message into the Kafka broker, and try pulling the same
64
+ message back out.
65
+
66
+ """
67
+ topic = "test_message_with_key"
68
+ # Push one message in...
69
+ simple_write_msg(self.kafka, topic, "can you hear me?", key="test_msg")
70
+ # ... and pull it back out.
71
+ consumer = adc.consumer.Consumer(adc.consumer.ConsumerConfig(
72
+ broker_urls=[self.kafka.address],
73
+ group_id="test_consumer",
74
+ auth=self.kafka.auth,
75
+ ))
76
+ consumer.subscribe(topic)
77
+ stream = consumer.stream()
78
+
79
+ msg = next(stream)
80
+ if msg.error() is not None:
81
+ raise Exception(msg.error())
82
+
83
+ self.assertEqual(msg.topic(), topic)
84
+ self.assertEqual(msg.value(), b"can you hear me?")
85
+ self.assertEqual(msg.key(), b"test_msg")
86
+
87
+ def test_reset_to_end(self):
64
88
  # Write a few messages.
65
- topic = "test_consume_from_end"
89
+ topic = "test_reset_to_end"
66
90
  simple_write_msgs(self.kafka, topic, [
67
91
  "message 1",
68
92
  "message 2",
@@ -79,15 +103,16 @@ class KafkaIntegrationTestCase(unittest.TestCase):
79
103
  stream = consumer.stream()
80
104
 
81
105
  # Now add messages after the "end"
106
+ time.sleep(0.5)
82
107
  simple_write_msg(self.kafka, topic, "message 4")
83
-
108
+ time.sleep(0.5)
84
109
  msg = next(stream)
85
110
  self.assertEqual(msg.topic(), topic)
86
111
  self.assertEqual(msg.value(), b"message 4")
87
112
 
88
- def test_consume_from_beginning(self):
113
+ def test_reset_to_beginning(self):
89
114
  # Write a few messages.
90
- topic = "test_consume_from_beginning"
115
+ topic = "test_reset_to_beginning"
91
116
  batch = [
92
117
  "message 1",
93
118
  "message 2",
@@ -164,24 +189,153 @@ class KafkaIntegrationTestCase(unittest.TestCase):
164
189
  self.assertEqual(actual.value().decode(), expected)
165
190
 
166
191
  # Start second consumer, also reading from earliest offset.
167
- consumer_2 = adc.consumer.Consumer(adc.consumer.ConsumerConfig(
192
+ consumer_2a = adc.consumer.Consumer(adc.consumer.ConsumerConfig(
168
193
  broker_urls=[self.kafka.address],
169
194
  group_id="test_consumer_2",
170
195
  auth=self.kafka.auth,
171
196
  read_forever=False,
172
197
  start_at=adc.consumer.ConsumerStartPosition.EARLIEST,
173
198
  ))
174
- consumer_2.subscribe(topic)
175
- stream_2 = consumer_2.stream()
176
- msgs_2 = [msg for msg in stream_2]
177
-
178
- # Now check that messages from both batches are processed.
179
- assert consumer_2._stop_event.is_set()
180
- self.assertEqual(len(batch_1 + batch_2), len(msgs_2))
181
- for expected, actual in zip(batch_1 + batch_2, msgs_2):
199
+ consumer_2a.subscribe(topic)
200
+
201
+ # read the topic using consumer_2a
202
+ stream_2a = consumer_2a.stream()
203
+ msgs_2a = [pair for pair in zip(batch_1, stream_2a)]
204
+ # end iteration early after batch 1
205
+ stream_2a.close()
206
+
207
+ # check that messages from only batch 1 were consumed
208
+ self.assertEqual(len(batch_1), len(msgs_2a))
209
+ for expected, actual in msgs_2a:
210
+ self.assertEqual(actual.topic(), topic)
211
+ self.assertEqual(actual.value().decode(), expected)
212
+
213
+ # commit autocommited indices
214
+ consumer_2a.close()
215
+
216
+ # Start another consumer with the same groupid
217
+ consumer_2b = adc.consumer.Consumer(adc.consumer.ConsumerConfig(
218
+ broker_urls=[self.kafka.address],
219
+ group_id="test_consumer_2",
220
+ auth=self.kafka.auth,
221
+ read_forever=False,
222
+ start_at=adc.consumer.ConsumerStartPosition.EARLIEST,
223
+ ))
224
+ consumer_2b.subscribe(topic)
225
+ stream_2b = consumer_2b.stream(autocommit=False, start_at=adc.consumer.LogicalOffset.STORED)
226
+
227
+ # read the rest using consumer_2b
228
+ msgs_2b = [msg for msg in stream_2b]
229
+
230
+ # Now check that messages from only batch_2 were read.
231
+ assert consumer_2b._stop_event.is_set()
232
+ self.assertEqual(len(batch_2), len(msgs_2b))
233
+ for expected, actual in zip(batch_2, msgs_2b):
234
+ self.assertEqual(actual.topic(), topic)
235
+ self.assertEqual(actual.value().decode(), expected)
236
+
237
+ def test_consume_from_beginning(self):
238
+ # Write a few messages.
239
+ topic = "test_consume_from_beginning"
240
+ batch = [
241
+ "message 1",
242
+ "message 2",
243
+ "message 3",
244
+ "message 4",
245
+ ]
246
+ simple_write_msgs(self.kafka, topic, batch)
247
+
248
+ consumer = adc.consumer.Consumer(adc.consumer.ConsumerConfig(
249
+ broker_urls=[self.kafka.address],
250
+ group_id="test_consumer",
251
+ auth=self.kafka.auth,
252
+ read_forever=False,
253
+ # Make reading start at the end by default
254
+ start_at=adc.consumer.ConsumerStartPosition.LATEST,
255
+ ))
256
+ consumer.subscribe(topic)
257
+ # Request reading from the beginning
258
+ stream = consumer.stream(start_at=adc.consumer.LogicalOffset.BEGINNING)
259
+ msgs = [msg for msg in stream]
260
+
261
+ assert consumer._stop_event.is_set()
262
+ self.assertEqual(len(batch), len(msgs))
263
+ for expected, actual in zip(batch, msgs):
182
264
  self.assertEqual(actual.topic(), topic)
183
265
  self.assertEqual(actual.value().decode(), expected)
184
266
 
267
+ # Read again from the beginning
268
+ stream = consumer.stream(start_at=adc.consumer.LogicalOffset.BEGINNING)
269
+ msgs = [msg for msg in stream]
270
+
271
+ assert consumer._stop_event.is_set()
272
+ self.assertEqual(len(batch), len(msgs))
273
+ for expected, actual in zip(batch, msgs):
274
+ self.assertEqual(actual.topic(), topic)
275
+ self.assertEqual(actual.value().decode(), expected)
276
+
277
+ def test_consume_from_end(self):
278
+ # Write a few messages.
279
+ topic = "test_consume_from_end"
280
+ simple_write_msgs(self.kafka, topic, [
281
+ "message 1",
282
+ "message 2",
283
+ "message 3",
284
+ ])
285
+ consumer = adc.consumer.Consumer(adc.consumer.ConsumerConfig(
286
+ broker_urls=[self.kafka.address],
287
+ group_id="test_consumer",
288
+ auth=self.kafka.auth,
289
+ # Make reading start at the beginning by default
290
+ start_at=adc.consumer.ConsumerStartPosition.EARLIEST,
291
+ ))
292
+ consumer.subscribe(topic)
293
+ # Request reading from the end
294
+ stream = consumer.stream(start_at=adc.consumer.LogicalOffset.END)
295
+
296
+ # Now add messages after the "end"
297
+ time.sleep(0.5)
298
+ simple_write_msg(self.kafka, topic, "message 4")
299
+ time.sleep(0.5)
300
+ msg = next(stream)
301
+ self.assertEqual(msg.topic(), topic)
302
+ self.assertEqual(msg.value(), b"message 4")
303
+
304
+ def test_consume_from_datetime(self):
305
+ # Write a few messages.
306
+ topic = "test_consume_from_datetime"
307
+ simple_write_msgs(self.kafka, topic, [
308
+ "message 1",
309
+ "message 2",
310
+ "message 3",
311
+ ])
312
+ # Wait a while, write, and wait some more
313
+ time.sleep(2)
314
+ client_middle_time = datetime.now()
315
+ time.sleep(2)
316
+ simple_write_msg(self.kafka, topic, "message 4")
317
+ time.sleep(1)
318
+
319
+ consumer = adc.consumer.Consumer(adc.consumer.ConsumerConfig(
320
+ broker_urls=[self.kafka.address],
321
+ group_id="test_consumer",
322
+ auth=self.kafka.auth,
323
+ read_forever=False,
324
+ start_at=adc.consumer.ConsumerStartPosition.EARLIEST,
325
+ ))
326
+ consumer.subscribe(topic)
327
+ stream = consumer.stream()
328
+ timestamps = [datetime.fromtimestamp(msg.timestamp()[1] / 1000.0) for msg in stream]
329
+
330
+ middle_time = timestamps[2] + (timestamps[3] - timestamps[2]) / 2
331
+ diff = middle_time - client_middle_time
332
+ logger.info(f"Difference between client and received timestamps: {diff!s}")
333
+
334
+ stream = consumer.stream(start_at=middle_time)
335
+ msg = next(stream)
336
+ self.assertEqual(msg.topic(), topic)
337
+ self.assertEqual(msg.value(), b"message 4")
338
+
185
339
  def test_consume_not_forever(self):
186
340
  topic = "test_consume_not_forever"
187
341
  simple_write_msg(self.kafka, topic, "message 1")
@@ -248,6 +402,43 @@ class KafkaIntegrationTestCase(unittest.TestCase):
248
402
  self.assertEqual(messages[1].value(), b"message 2")
249
403
  self.assertEqual(messages[2].value(), b"message 3")
250
404
 
405
+ def test_multi_topic_handling(self):
406
+ """Use a single producer object to write messages to multiple topics,
407
+ and check that a consumer can receive them all.
408
+
409
+ """
410
+ topics = ["test_multi_1", "test_multi_2"]
411
+
412
+ # Push some messages in
413
+ producer = adc.producer.Producer(adc.producer.ProducerConfig(
414
+ broker_urls=[self.kafka.address],
415
+ topic=None,
416
+ auth=self.kafka.auth,
417
+ ))
418
+ for i in range(0, 8):
419
+ producer.write(str(i), topic=topics[i % 2])
420
+ producer.flush()
421
+ logger.info("messages sent")
422
+
423
+ # check that we receive the messages from the right topics
424
+ consumer = adc.consumer.Consumer(adc.consumer.ConsumerConfig(
425
+ broker_urls=[self.kafka.address],
426
+ group_id="test_consumer",
427
+ auth=self.kafka.auth,
428
+ ))
429
+ consumer.subscribe(topics)
430
+ stream = consumer.stream()
431
+ total_messages = 0
432
+ for msg in stream:
433
+ if msg.error() is not None:
434
+ raise Exception(msg.error())
435
+ idx = int(msg.value())
436
+ self.assertEqual(msg.topic(), topics[idx % 2])
437
+ total_messages += 1
438
+ if total_messages == 8:
439
+ break
440
+ self.assertEqual(total_messages, 8)
441
+
251
442
 
252
443
  class KafkaDockerConnection:
253
444
  """Holds connection information for communicating with a Kafka broker running
@@ -285,6 +476,10 @@ class KafkaDockerConnection:
285
476
  self.auth = adc.auth.SASLAuth(
286
477
  user="test", password="test-pass",
287
478
  ssl_ca_location=self.certfile.name,
479
+ # disable endpoint verification because the docker service generates a certificate with
480
+ # a useless subject (the container ID) which can never match the hostname used to
481
+ # connect (typically 0.0.0.0)
482
+ ssl_endpoint_identification_algorithm="none",
288
483
  )
289
484
 
290
485
  def poll_for_kafka_broker_address(self, maxiter=20, sleep=timedelta(milliseconds=500)):
@@ -308,6 +503,8 @@ class KafkaDockerConnection:
308
503
  if not addrs:
309
504
  return None
310
505
  ip = addrs[0]['HostIp']
506
+ if len(ip) == 0:
507
+ ip = "localhost"
311
508
  port = addrs[0]['HostPort']
312
509
  return f"{ip}:{port}"
313
510
 
@@ -373,8 +570,16 @@ class KafkaDockerConnection:
373
570
  detach=True,
374
571
  auto_remove=True,
375
572
  network=self.net.name,
376
- # Setting None below the OS pick an ephemeral port.
377
- ports={"9092/tcp": None},
573
+ # Kafka insists on redirecting consumers to one of its advertised listeners,
574
+ # which it will get wrong if it is running in a private container network.
575
+ # To fix this, we need to tell it what to advertise, which means we must
576
+ # know what port will be visible from the host system, and we cannot use an
577
+ # ephemeral port, which would be known to us only after the container is
578
+ # started. Since we have to pick something, pick 9092, which means that
579
+ # these tests cannot run if there is already an instance of Kafka running on
580
+ # the same host.
581
+ ports={"9092/tcp": 9092},
582
+ command=["/root/runServer", "--advertisedListener", "SASL_SSL://localhost:9092"],
378
583
  )
379
584
 
380
585
  def get_or_create_docker_network(self):
@@ -389,13 +594,13 @@ class KafkaDockerConnection:
389
594
  return self.docker_client.networks.create(name="adc-integration-test")
390
595
 
391
596
 
392
- def simple_write_msg(conn: KafkaDockerConnection, topic: str, msg: str):
597
+ def simple_write_msg(conn: KafkaDockerConnection, topic: str, msg: str, key: Optional[str] = None):
393
598
  producer = adc.producer.Producer(adc.producer.ProducerConfig(
394
599
  broker_urls=[conn.address],
395
600
  topic=topic,
396
601
  auth=conn.auth,
397
602
  ))
398
- producer.write(msg)
603
+ producer.write(msg, key=key)
399
604
  producer.flush()
400
605
 
401
606
 
@@ -1,54 +0,0 @@
1
- import logging
2
- from typing import Callable
3
-
4
- import confluent_kafka # type: ignore
5
-
6
- logger = logging.getLogger("adc-streaming")
7
-
8
-
9
- ErrorCallback = Callable[[confluent_kafka.KafkaError], None]
10
-
11
-
12
- def log_client_errors(kafka_error: confluent_kafka.KafkaError):
13
- if kafka_error.code() == confluent_kafka.KafkaError._ALL_BROKERS_DOWN:
14
- # This error occurs very frequently. It's not nearly as fatal as it
15
- # sounds: it really indicates that the client's broker metadata has
16
- # timed out. It appears to get triggered in races during client
17
- # shutdown, too. See https://github.com/edenhill/librdkafka/issues/2543
18
- # for more background.
19
- logger.warn("client is currently disconnected from all brokers")
20
- else:
21
- logger.error(f"internal kafka error: {kafka_error}")
22
-
23
-
24
- DeliveryCallback = Callable[[confluent_kafka.KafkaError, confluent_kafka.Message], None]
25
-
26
-
27
- def log_delivery_errors(
28
- kafka_error: confluent_kafka.KafkaError,
29
- msg: confluent_kafka.Message) -> None:
30
- if kafka_error is not None:
31
- logger.error(f"delivery error: {kafka_error}")
32
-
33
-
34
- def raise_delivery_errors(kafka_error: confluent_kafka.KafkaError,
35
- msg: confluent_kafka.Message) -> None:
36
- if kafka_error is not None:
37
- raise KafkaException.from_kafka_error(kafka_error)
38
- elif msg.error() is not None:
39
- raise KafkaException.from_kafka_error(msg.error())
40
-
41
-
42
- class KafkaException(Exception):
43
- @classmethod
44
- def from_kafka_error(cls, error):
45
- return cls(error)
46
-
47
- def __init__(self, error):
48
- self.error = error
49
- self.name = error.name()
50
- self.reason = error.str()
51
- self.retriable = error.retriable()
52
- self.fatal = error.fatal()
53
- msg = f"Error communicating with Kafka: code={self.name} {self.reason}"
54
- super(KafkaException, self).__init__(msg)
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes