adc-streaming 2.3.1__tar.gz → 2.4.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (37) hide show
  1. {adc-streaming-2.3.1 → adc-streaming-2.4.0}/PKG-INFO +1 -1
  2. {adc-streaming-2.3.1 → adc-streaming-2.4.0}/adc/__init__.py +2 -2
  3. {adc-streaming-2.3.1 → adc-streaming-2.4.0}/adc/consumer.py +73 -9
  4. adc-streaming-2.4.0/adc/errors.py +102 -0
  5. {adc-streaming-2.3.1 → adc-streaming-2.4.0}/adc/oidc.py +6 -4
  6. {adc-streaming-2.3.1 → adc-streaming-2.4.0}/adc/producer.py +17 -8
  7. {adc-streaming-2.3.1 → adc-streaming-2.4.0}/adc_streaming.egg-info/PKG-INFO +1 -1
  8. {adc-streaming-2.3.1 → adc-streaming-2.4.0}/adc_streaming.egg-info/requires.txt +1 -1
  9. {adc-streaming-2.3.1 → adc-streaming-2.4.0}/setup.cfg +1 -0
  10. {adc-streaming-2.3.1 → adc-streaming-2.4.0}/setup.py +1 -1
  11. {adc-streaming-2.3.1 → adc-streaming-2.4.0}/tests/test_kafka_integration.py +194 -18
  12. adc-streaming-2.3.1/adc/errors.py +0 -54
  13. {adc-streaming-2.3.1 → adc-streaming-2.4.0}/.github/workflows/build.yml +0 -0
  14. {adc-streaming-2.3.1 → adc-streaming-2.4.0}/.gitignore +0 -0
  15. {adc-streaming-2.3.1 → adc-streaming-2.4.0}/LICENSE +0 -0
  16. {adc-streaming-2.3.1 → adc-streaming-2.4.0}/Makefile +0 -0
  17. {adc-streaming-2.3.1 → adc-streaming-2.4.0}/README.md +0 -0
  18. {adc-streaming-2.3.1 → adc-streaming-2.4.0}/adc/auth.py +0 -0
  19. {adc-streaming-2.3.1 → adc-streaming-2.4.0}/adc/io.py +0 -0
  20. {adc-streaming-2.3.1 → adc-streaming-2.4.0}/adc/kafka.py +0 -0
  21. {adc-streaming-2.3.1 → adc-streaming-2.4.0}/adc_streaming.egg-info/SOURCES.txt +0 -0
  22. {adc-streaming-2.3.1 → adc-streaming-2.4.0}/adc_streaming.egg-info/dependency_links.txt +0 -0
  23. {adc-streaming-2.3.1 → adc-streaming-2.4.0}/adc_streaming.egg-info/not-zip-safe +0 -0
  24. {adc-streaming-2.3.1 → adc-streaming-2.4.0}/adc_streaming.egg-info/top_level.txt +0 -0
  25. {adc-streaming-2.3.1 → adc-streaming-2.4.0}/doc/Makefile +0 -0
  26. {adc-streaming-2.3.1 → adc-streaming-2.4.0}/doc/_static/css/my_theme.css +0 -0
  27. {adc-streaming-2.3.1 → adc-streaming-2.4.0}/doc/_templates/layout.html +0 -0
  28. {adc-streaming-2.3.1 → adc-streaming-2.4.0}/doc/api/api.rst +0 -0
  29. {adc-streaming-2.3.1 → adc-streaming-2.4.0}/doc/api/streaming.rst +0 -0
  30. {adc-streaming-2.3.1 → adc-streaming-2.4.0}/doc/conf.py +0 -0
  31. {adc-streaming-2.3.1 → adc-streaming-2.4.0}/doc/index.rst +0 -0
  32. {adc-streaming-2.3.1 → adc-streaming-2.4.0}/doc/user/installation.rst +0 -0
  33. {adc-streaming-2.3.1 → adc-streaming-2.4.0}/doc/user/quickstart.rst +0 -0
  34. {adc-streaming-2.3.1 → adc-streaming-2.4.0}/pyproject.toml +0 -0
  35. {adc-streaming-2.3.1 → adc-streaming-2.4.0}/recipe/meta.yaml +0 -0
  36. {adc-streaming-2.3.1 → adc-streaming-2.4.0}/tests/test_auth.py +0 -0
  37. {adc-streaming-2.3.1 → adc-streaming-2.4.0}/tests/test_kafka.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.1
2
2
  Name: adc-streaming
3
- Version: 2.3.1
3
+ Version: 2.4.0
4
4
  Summary: Astronomy Data Commons streaming client libraries
5
5
  Home-page: https://github.com/astronomy-commons/adc-streaming
6
6
  Author: Astronomy Data Commons Team
@@ -1,8 +1,8 @@
1
1
  try:
2
- from importlib.metadata import version, PackageNotFoundError
2
+ from importlib.metadata import PackageNotFoundError, version
3
3
  except ImportError:
4
4
  # NOTE: remove after dropping support for Python < 3.8
5
- from importlib_metadata import version, PackageNotFoundError
5
+ from importlib_metadata import PackageNotFoundError, version
6
6
 
7
7
  try:
8
8
  __version__ = version("adc-streaming")
@@ -1,10 +1,13 @@
1
1
  import dataclasses
2
2
  import enum
3
3
  import logging
4
- from datetime import timedelta
5
4
  import threading
6
- from typing import Dict, Iterable, Iterator, List, Optional, Set, Union
7
5
  from collections import defaultdict
6
+ from datetime import datetime, timedelta
7
+ # Imports from typing are deprecated as of Python 3.9 but required for
8
+ # compatibility with earlier versions
9
+ from typing import (Collection, Dict, Iterable, Iterator, List, Optional, Set,
10
+ Union)
8
11
 
9
12
  import confluent_kafka # type: ignore
10
13
  import confluent_kafka.admin # type: ignore
@@ -14,6 +17,18 @@ from .errors import ErrorCallback, log_client_errors
14
17
  from .oidc import set_oauth_cb
15
18
 
16
19
 
20
+ class LogicalOffset(enum.IntEnum):
21
+ BEGINNING = confluent_kafka.OFFSET_BEGINNING
22
+ EARLIEST = confluent_kafka.OFFSET_BEGINNING
23
+
24
+ END = confluent_kafka.OFFSET_END
25
+ LATEST = confluent_kafka.OFFSET_END
26
+
27
+ STORED = confluent_kafka.OFFSET_STORED
28
+
29
+ INVALID = confluent_kafka.OFFSET_INVALID
30
+
31
+
17
32
  class Consumer:
18
33
  conf: 'ConsumerConfig'
19
34
  _consumer: confluent_kafka.Consumer
@@ -23,7 +38,7 @@ class Consumer:
23
38
  self.logger = logging.getLogger("adc-streaming.consumer")
24
39
  self.conf = conf
25
40
  self._consumer = confluent_kafka.Consumer(conf._to_confluent_kafka())
26
- # Workaround for https://github.com/edenhill/librdkafka/issues/3871.
41
+ # Workaround for https://github.com/confluentinc/librdkafka/issues/3753#issuecomment-1058272987.
27
42
  # FIXME: Remove once fixed upstream, or on removal of oauth_cb.
28
43
  self._consumer.poll(0)
29
44
  self._stop_event = threading.Event()
@@ -86,6 +101,27 @@ class Consumer:
86
101
  else:
87
102
  self._consumer.commit(msg, asynchronous=False)
88
103
 
104
+ def _offsets_for_position(self, partitions: Collection[confluent_kafka.TopicPartition],
105
+ position: Union[datetime, LogicalOffset]) \
106
+ -> List[confluent_kafka.TopicPartition]:
107
+ if isinstance(position, datetime):
108
+ offset = int(position.timestamp() * 1000)
109
+ elif isinstance(position, LogicalOffset):
110
+ offset = position
111
+ else:
112
+ raise TypeError("Only datetime objects and logical offsets supported")
113
+
114
+ _partitions = [
115
+ confluent_kafka.TopicPartition(topic=tp.topic, partition=tp.partition, offset=offset)
116
+ for tp in partitions
117
+ ]
118
+
119
+ if isinstance(position, datetime):
120
+ self.logger.debug("looking up offsets for time")
121
+ return self._consumer.offsets_for_times(_partitions)
122
+ else:
123
+ return _partitions
124
+
89
125
  def stop(self):
90
126
  """Stops the runloop of the consumer. Useful when running the
91
127
  consumer in a different thread.
@@ -95,7 +131,8 @@ class Consumer:
95
131
  def stream(self,
96
132
  autocommit: bool = True,
97
133
  batch_size: int = 100,
98
- batch_timeout: timedelta = timedelta(seconds=1.0)
134
+ batch_timeout: timedelta = timedelta(seconds=1.0),
135
+ start_at: Union[datetime, LogicalOffset, None] = None
99
136
  ) -> Iterator[confluent_kafka.Message]:
100
137
  """Returns a stream which iterates over the messages in the topics
101
138
  to which the client is subscribed.
@@ -112,6 +149,15 @@ class Consumer:
112
149
  provide a full batch of batch_size messages. Higher values may be more
113
150
  efficient, but may add latency.
114
151
 
152
+ start_at controls the location in every partition the client reads
153
+ from, overriding the configured default position or the stored offsets.
154
+ Either a special logical offset value (END, BEGINNING, STORE, INVALID)
155
+ or a datetime. Using a datetime will cause reading to start from the
156
+ first message *after* the specified datetime (or END if none exists).
157
+ Using this parameter will override any previously assigned but not
158
+ committed offsets if they are managed from outside adc. Passing None
159
+ (or leaving it unspecified) avoids reassignment.
160
+
115
161
  If the consumer's configuration has read_forever set to False, then the
116
162
  stream stops when the client has hit the last message in all partitions.
117
163
  This set of partitions is calculated just once when iterate() is first
@@ -119,6 +165,11 @@ class Consumer:
119
165
  behavior in this case.
120
166
 
121
167
  """
168
+
169
+ if start_at is not None:
170
+ assignment = self._consumer.assignment()
171
+ self._consumer.assign(self._offsets_for_position(assignment, start_at))
172
+
122
173
  if self.conf.read_forever:
123
174
  return self._stream_forever(autocommit, batch_size, batch_timeout)
124
175
  else:
@@ -202,7 +253,10 @@ class Consumer:
202
253
  self._consumer.close()
203
254
 
204
255
 
205
- class ConsumerStartPosition(enum.Enum):
256
+ # Used to be called ConsumerStartPosition, though this was confusing because
257
+ # it only affects "auto.offset.reset" not the start position for a call to
258
+ # consume.
259
+ class ConsumerDefaultPosition(enum.Enum):
206
260
  EARLIEST = 1
207
261
  LATEST = 2
208
262
 
@@ -210,6 +264,11 @@ class ConsumerStartPosition(enum.Enum):
210
264
  return self.name.lower()
211
265
 
212
266
 
267
+ # Alias to the old name
268
+ # TODO: Remove alias on the next breaking release
269
+ ConsumerStartPosition = ConsumerDefaultPosition
270
+
271
+
213
272
  @dataclasses.dataclass
214
273
  class ConsumerConfig:
215
274
  broker_urls: List[str]
@@ -225,8 +284,13 @@ class ConsumerConfig:
225
284
  # was marked done with consumer.mark_done, regardless of this setting. This
226
285
  # is only used when the position in the stream is unknown.
227
286
  #
228
- # This is specified as a logical offset via a ConsumerStartPosition value.
229
- start_at: ConsumerStartPosition = ConsumerStartPosition.EARLIEST
287
+ # You can force reading at a logical offset or datetime with the "start_at"
288
+ # argument to Consumer.stream().
289
+ #
290
+ # This is specified via a ConsumerDefaultPosition value.
291
+ #
292
+ # TODO: rename on next breaking release
293
+ start_at: ConsumerDefaultPosition = ConsumerDefaultPosition.EARLIEST
230
294
 
231
295
  # Authentication package to pass in to read from Kafka.
232
296
  auth: Optional[SASLAuth] = None
@@ -269,13 +333,13 @@ class ConsumerConfig:
269
333
  "reconnect.backoff.max.ms": as_ms(self.reconnect_max_time),
270
334
  "reconnect.backoff.ms": as_ms(self.reconnect_backoff_time),
271
335
  }
272
- if self.start_at is ConsumerStartPosition.EARLIEST:
336
+ if self.start_at is ConsumerDefaultPosition.EARLIEST:
273
337
  default_topic_config = config.get("default.topic.config", {})
274
338
  default_topic_config = {
275
339
  "auto.offset.reset": "EARLIEST",
276
340
  }
277
341
  config["default.topic.config"] = default_topic_config
278
- elif self.start_at is ConsumerStartPosition.LATEST:
342
+ elif self.start_at is ConsumerDefaultPosition.LATEST:
279
343
  # FIXME: librdkafka has a bug in offset handling - it caches
280
344
  # "OFFSET_END", and will repeatedly move to the end of the
281
345
  # topic. See https://github.com/edenhill/librdkafka/pull/2876 -
@@ -0,0 +1,102 @@
1
+ import logging
2
+ from typing import Callable
3
+
4
+ import confluent_kafka # type: ignore
5
+
6
+ logger = logging.getLogger("adc-streaming")
7
+
8
+
9
+ ErrorCallback = Callable[[confluent_kafka.KafkaError], None]
10
+
11
+
12
+ def log_client_errors(kafka_error: confluent_kafka.KafkaError):
13
+ if kafka_error.code() == confluent_kafka.KafkaError._ALL_BROKERS_DOWN:
14
+ # This error occurs very frequently. It's not nearly as fatal as it
15
+ # sounds: it really indicates that the client's broker metadata has
16
+ # timed out. It appears to get triggered in races during client
17
+ # shutdown, too. See https://github.com/edenhill/librdkafka/issues/2543
18
+ # for more background.
19
+ logger.warn("client is currently disconnected from all brokers")
20
+ else:
21
+ logger.error(f"internal kafka error: {kafka_error}")
22
+
23
+
24
+ DeliveryCallback = Callable[[confluent_kafka.KafkaError, confluent_kafka.Message], None]
25
+
26
+
27
+ def log_delivery_errors(
28
+ kafka_error: confluent_kafka.KafkaError,
29
+ msg: confluent_kafka.Message) -> None:
30
+ if kafka_error is not None:
31
+ logger.error(f"delivery error: {kafka_error}")
32
+
33
+
34
+ def raise_delivery_errors(kafka_error: confluent_kafka.KafkaError,
35
+ msg: confluent_kafka.Message) -> None:
36
+ if kafka_error is not None:
37
+ raise KafkaException.from_kafka_error(kafka_error, msg)
38
+ elif msg.error() is not None:
39
+ raise KafkaException.from_kafka_error(msg.error(), msg)
40
+
41
+
42
+ def _get_topic_related_errors():
43
+ """Build a set of all Kafka error codes which seem to relate to a specific topic.
44
+
45
+ This uses a list extracted from all documented error codes up to confluent_kafka v2.4,
46
+ but some of these errors did not exist or were not exposed in earlier versions.
47
+ To maintain backward compatibility, this function checks whether each error exists before
48
+ attempting to otherwise refer to it.
49
+ """
50
+ err_names = [
51
+ "_UNKNOWN_TOPIC",
52
+ "_NO_OFFSET",
53
+ "_LOG_TRUNCATION",
54
+ "OFFSET_OUT_OF_RANGE",
55
+ "UNKNOWN_TOPIC_OR_PART",
56
+ "NOT_LEADER_FOR_PARTITION",
57
+ "TOPIC_EXCEPTION",
58
+ "NOT_ENOUGH_REPLICAS",
59
+ "NOT_ENOUGH_REPLICAS_AFTER_APPEND",
60
+ "INVALID_COMMIT_OFFSET_SIZE",
61
+ "TOPIC_AUTHORIZATION_FAILED",
62
+ "TOPIC_ALREADY_EXISTS",
63
+ "INVALID_PARTITIONS",
64
+ "INVALID_REPLICATION_FACTOR",
65
+ "INVALID_REPLICA_ASSIGNMENT",
66
+ "REASSIGNMENT_IN_PROGRESS",
67
+ "TOPIC_DELETION_DISABLED",
68
+ "OFFSET_NOT_AVAILABLE",
69
+ "PREFERRED_LEADER_NOT_AVAILABLE",
70
+ "NO_REASSIGNMENT_IN_PROGRESS",
71
+ "GROUP_SUBSCRIBED_TO_TOPIC",
72
+ "UNSTABLE_OFFSET_COMMIT",
73
+ "UNKNOWN_TOPIC_ID",
74
+ ]
75
+ errors = set()
76
+ for name in err_names:
77
+ if hasattr(confluent_kafka.KafkaError, name):
78
+ errors.add(getattr(confluent_kafka.KafkaError, name))
79
+ else:
80
+ logger.debug(f"{name} does not exist in confluent_kafka version "
81
+ f"{confluent_kafka.__version__} ({confluent_kafka.libversion()})")
82
+ return errors
83
+
84
+
85
+ class KafkaException(Exception):
86
+ @classmethod
87
+ def from_kafka_error(cls, error, msg=None):
88
+ return cls(error, msg)
89
+
90
+ topic_related_errors = _get_topic_related_errors()
91
+
92
+ def __init__(self, error, msg=None):
93
+ self.error = error
94
+ self.name = error.name()
95
+ self.reason = error.str()
96
+ self.retriable = error.retriable()
97
+ self.fatal = error.fatal()
98
+ self.message = msg
99
+ ex_msg = f"Error communicating with Kafka: code={self.name} {self.reason}"
100
+ if msg and error.code() in KafkaException.topic_related_errors:
101
+ ex_msg += f" on topic {msg.topic()}"
102
+ super(KafkaException, self).__init__(ex_msg)
@@ -2,9 +2,11 @@ def set_oauth_cb(config):
2
2
  """Implement client support for KIP-768 OpenID Connect.
3
3
 
4
4
  Apache Kafka 3.1.0 supports authentication using OpenID Client Credentials.
5
- Native support for Python is coming in the next release of librdkafka
6
- (version 1.9.0). Meanwhile, this is a pure Python implementation of the
7
- refresh token callback.
5
+ Native support for Python is still incomplete due to this issue:
6
+ https://github.com/confluentinc/librdkafka/issues/3751
7
+
8
+ Meanwhile, this is a pure Python implementation of the refresh token
9
+ callback.
8
10
  """
9
11
  if config.pop('sasl.oauthbearer.method', None) != 'oidc':
10
12
  return
@@ -16,7 +18,7 @@ def set_oauth_cb(config):
16
18
 
17
19
  from authlib.integrations.requests_client import OAuth2Session
18
20
  session = OAuth2Session(client_id, client_secret, scope=scope)
19
-
21
+
20
22
  def oauth_cb(*_, **__):
21
23
  token = session.fetch_token(
22
24
  token_endpoint, grant_type='client_credentials')
@@ -1,9 +1,9 @@
1
1
  import abc
2
- from ast import comprehension
3
2
  import dataclasses
4
3
  import logging
5
4
  from datetime import timedelta
6
5
  from typing import Dict, List, Optional, Union
6
+
7
7
  try: # this will work only in python >= 3.8
8
8
  from typing import Literal
9
9
  except ImportError:
@@ -27,22 +27,30 @@ class Producer:
27
27
  self.conf = conf
28
28
  self.logger.debug(f"connecting to producer with config {conf._to_confluent_kafka()}")
29
29
  self._producer = confluent_kafka.Producer(conf._to_confluent_kafka())
30
- # Workaround for https://github.com/edenhill/librdkafka/issues/3871.
30
+ # Workaround for https://github.com/confluentinc/librdkafka/issues/3753#issuecomment-1058272987.
31
31
  # FIXME: Remove once fixed upstream, or on removal of oauth_cb.
32
32
  self._producer.poll(0)
33
33
 
34
34
  def write(self,
35
35
  msg: Union[bytes, 'Serializable'],
36
36
  headers: Optional[Union[dict, list]] = None,
37
- delivery_callback: Optional[DeliveryCallback] = log_delivery_errors) -> None:
37
+ delivery_callback: Optional[DeliveryCallback] = log_delivery_errors,
38
+ topic: Optional[str] = None) -> None:
38
39
  if isinstance(msg, Serializable):
39
40
  msg = msg.serialize()
40
- self.logger.debug("writing message to %s", self.conf.topic)
41
+ if topic is None:
42
+ if self.conf.topic is not None:
43
+ topic = self.conf.topic
44
+ else:
45
+ raise Exception("No topic specified for write: "
46
+ "Either configure a topic when constructing the Producer, "
47
+ "or specify the topic argument to write()")
48
+ self.logger.debug("writing message to %s", topic)
41
49
  if delivery_callback is not None:
42
- self._producer.produce(self.conf.topic, msg, headers=headers,
50
+ self._producer.produce(topic, msg, headers=headers,
43
51
  on_delivery=delivery_callback)
44
52
  else:
45
- self._producer.produce(self.conf.topic, msg, headers=headers)
53
+ self._producer.produce(topic, msg, headers=headers)
46
54
 
47
55
  def flush(self, timeout: timedelta = timedelta(seconds=10)) -> int:
48
56
  """Attempt to flush enqueued messages. Return the number of messages still
@@ -78,7 +86,7 @@ class Producer:
78
86
  @dataclasses.dataclass
79
87
  class ProducerConfig:
80
88
  broker_urls: List[str]
81
- topic: str
89
+ topic: Optional[str]
82
90
  auth: Optional[SASLAuth] = None
83
91
  error_callback: Optional[ErrorCallback] = log_client_errors
84
92
 
@@ -104,7 +112,8 @@ class ProducerConfig:
104
112
  # between attempts to reconnect to Kafka.
105
113
  reconnect_max_time: timedelta = timedelta(seconds=10)
106
114
 
107
- compression_type: Optional[Union[Literal['gzip'], Literal['snappy'], Literal['lz4'], Literal['zstd']]] = None
115
+ compression_type: Optional[Union[Literal['gzip'], Literal['snappy'],
116
+ Literal['lz4'], Literal['zstd']]] = None
108
117
 
109
118
  # maximum message size, before compression
110
119
  message_max_bytes: Optional[int] = None
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.1
2
2
  Name: adc-streaming
3
- Version: 2.3.1
3
+ Version: 2.4.0
4
4
  Summary: Astronomy Data Commons streaming client libraries
5
5
  Home-page: https://github.com/astronomy-commons/adc-streaming
6
6
  Author: Astronomy Data Commons Team
@@ -1,5 +1,5 @@
1
1
  authlib
2
- confluent-kafka
2
+ confluent-kafka!=2.1.0,!=2.1.1,>=1.6.1
3
3
  requests
4
4
  tqdm
5
5
  certifi>=2020.04.05.1
@@ -13,6 +13,7 @@ exclude = setup.py,
13
13
  [tool:pytest]
14
14
  log_cli = True
15
15
  log_cli_level = INFO
16
+ testpaths = tests
16
17
 
17
18
  [egg_info]
18
19
  tag_build =
@@ -4,7 +4,7 @@ from setuptools import setup
4
4
  # requirements
5
5
  install_requires = [
6
6
  "authlib", # FIXME: drop after next release of confluent-kafka with OIDC support
7
- "confluent-kafka",
7
+ "confluent-kafka >= 1.6.1, != 2.1.0, != 2.1.1",
8
8
  "dataclasses ; python_version < '3.7'",
9
9
  "importlib-metadata ; python_version < '3.8'",
10
10
  "requests", # FIXME: drop after next release of confluent-kafka with OIDC support
@@ -2,7 +2,7 @@ import logging
2
2
  import tempfile
3
3
  import time
4
4
  import unittest
5
- from datetime import timedelta
5
+ from datetime import datetime, timedelta, timezone
6
6
  from typing import List
7
7
 
8
8
  import docker
@@ -59,10 +59,9 @@ class KafkaIntegrationTestCase(unittest.TestCase):
59
59
  self.assertEqual(msg.topic(), topic)
60
60
  self.assertEqual(msg.value(), b"can you hear me?")
61
61
 
62
- @unittest.skip("skipping due to bug in librdkafka")
63
- def test_consume_from_end(self):
62
+ def test_reset_to_end(self):
64
63
  # Write a few messages.
65
- topic = "test_consume_from_end"
64
+ topic = "test_reset_to_end"
66
65
  simple_write_msgs(self.kafka, topic, [
67
66
  "message 1",
68
67
  "message 2",
@@ -79,15 +78,16 @@ class KafkaIntegrationTestCase(unittest.TestCase):
79
78
  stream = consumer.stream()
80
79
 
81
80
  # Now add messages after the "end"
81
+ time.sleep(0.5)
82
82
  simple_write_msg(self.kafka, topic, "message 4")
83
-
83
+ time.sleep(0.5)
84
84
  msg = next(stream)
85
85
  self.assertEqual(msg.topic(), topic)
86
86
  self.assertEqual(msg.value(), b"message 4")
87
87
 
88
- def test_consume_from_beginning(self):
88
+ def test_reset_to_beginning(self):
89
89
  # Write a few messages.
90
- topic = "test_consume_from_beginning"
90
+ topic = "test_reset_to_beginning"
91
91
  batch = [
92
92
  "message 1",
93
93
  "message 2",
@@ -164,24 +164,153 @@ class KafkaIntegrationTestCase(unittest.TestCase):
164
164
  self.assertEqual(actual.value().decode(), expected)
165
165
 
166
166
  # Start second consumer, also reading from earliest offset.
167
- consumer_2 = adc.consumer.Consumer(adc.consumer.ConsumerConfig(
167
+ consumer_2a = adc.consumer.Consumer(adc.consumer.ConsumerConfig(
168
+ broker_urls=[self.kafka.address],
169
+ group_id="test_consumer_2",
170
+ auth=self.kafka.auth,
171
+ read_forever=False,
172
+ start_at=adc.consumer.ConsumerStartPosition.EARLIEST,
173
+ ))
174
+ consumer_2a.subscribe(topic)
175
+
176
+ # read the topic using consumer_2a
177
+ stream_2a = consumer_2a.stream()
178
+ msgs_2a = [pair for pair in zip(batch_1, stream_2a)]
179
+ # end iteration early after batch 1
180
+ stream_2a.close()
181
+
182
+ # check that messages from only batch 1 were consumed
183
+ self.assertEqual(len(batch_1), len(msgs_2a))
184
+ for expected, actual in msgs_2a:
185
+ self.assertEqual(actual.topic(), topic)
186
+ self.assertEqual(actual.value().decode(), expected)
187
+
188
+ # commit autocommited indices
189
+ consumer_2a.close()
190
+
191
+ # Start another consumer with the same groupid
192
+ consumer_2b = adc.consumer.Consumer(adc.consumer.ConsumerConfig(
168
193
  broker_urls=[self.kafka.address],
169
194
  group_id="test_consumer_2",
170
195
  auth=self.kafka.auth,
171
196
  read_forever=False,
172
197
  start_at=adc.consumer.ConsumerStartPosition.EARLIEST,
173
198
  ))
174
- consumer_2.subscribe(topic)
175
- stream_2 = consumer_2.stream()
176
- msgs_2 = [msg for msg in stream_2]
177
-
178
- # Now check that messages from both batches are processed.
179
- assert consumer_2._stop_event.is_set()
180
- self.assertEqual(len(batch_1 + batch_2), len(msgs_2))
181
- for expected, actual in zip(batch_1 + batch_2, msgs_2):
199
+ consumer_2b.subscribe(topic)
200
+ stream_2b = consumer_2b.stream(autocommit=False, start_at=adc.consumer.LogicalOffset.STORED)
201
+
202
+ # read the rest using consumer_2b
203
+ msgs_2b = [msg for msg in stream_2b]
204
+
205
+ # Now check that messages from only batch_2 were read.
206
+ assert consumer_2b._stop_event.is_set()
207
+ self.assertEqual(len(batch_2), len(msgs_2b))
208
+ for expected, actual in zip(batch_2, msgs_2b):
209
+ self.assertEqual(actual.topic(), topic)
210
+ self.assertEqual(actual.value().decode(), expected)
211
+
212
+ def test_consume_from_beginning(self):
213
+ # Write a few messages.
214
+ topic = "test_consume_from_beginning"
215
+ batch = [
216
+ "message 1",
217
+ "message 2",
218
+ "message 3",
219
+ "message 4",
220
+ ]
221
+ simple_write_msgs(self.kafka, topic, batch)
222
+
223
+ consumer = adc.consumer.Consumer(adc.consumer.ConsumerConfig(
224
+ broker_urls=[self.kafka.address],
225
+ group_id="test_consumer",
226
+ auth=self.kafka.auth,
227
+ read_forever=False,
228
+ # Make reading start at the end by default
229
+ start_at=adc.consumer.ConsumerStartPosition.LATEST,
230
+ ))
231
+ consumer.subscribe(topic)
232
+ # Request reading from the beginning
233
+ stream = consumer.stream(start_at=adc.consumer.LogicalOffset.BEGINNING)
234
+ msgs = [msg for msg in stream]
235
+
236
+ assert consumer._stop_event.is_set()
237
+ self.assertEqual(len(batch), len(msgs))
238
+ for expected, actual in zip(batch, msgs):
239
+ self.assertEqual(actual.topic(), topic)
240
+ self.assertEqual(actual.value().decode(), expected)
241
+
242
+ # Read again from the beginning
243
+ stream = consumer.stream(start_at=adc.consumer.LogicalOffset.BEGINNING)
244
+ msgs = [msg for msg in stream]
245
+
246
+ assert consumer._stop_event.is_set()
247
+ self.assertEqual(len(batch), len(msgs))
248
+ for expected, actual in zip(batch, msgs):
182
249
  self.assertEqual(actual.topic(), topic)
183
250
  self.assertEqual(actual.value().decode(), expected)
184
251
 
252
+ def test_consume_from_end(self):
253
+ # Write a few messages.
254
+ topic = "test_consume_from_end"
255
+ simple_write_msgs(self.kafka, topic, [
256
+ "message 1",
257
+ "message 2",
258
+ "message 3",
259
+ ])
260
+ consumer = adc.consumer.Consumer(adc.consumer.ConsumerConfig(
261
+ broker_urls=[self.kafka.address],
262
+ group_id="test_consumer",
263
+ auth=self.kafka.auth,
264
+ # Make reading start at the beginning by default
265
+ start_at=adc.consumer.ConsumerStartPosition.EARLIEST,
266
+ ))
267
+ consumer.subscribe(topic)
268
+ # Request reading from the end
269
+ stream = consumer.stream(start_at=adc.consumer.LogicalOffset.END)
270
+
271
+ # Now add messages after the "end"
272
+ time.sleep(0.5)
273
+ simple_write_msg(self.kafka, topic, "message 4")
274
+ time.sleep(0.5)
275
+ msg = next(stream)
276
+ self.assertEqual(msg.topic(), topic)
277
+ self.assertEqual(msg.value(), b"message 4")
278
+
279
+ def test_consume_from_datetime(self):
280
+ # Write a few messages.
281
+ topic = "test_consume_from_datetime"
282
+ simple_write_msgs(self.kafka, topic, [
283
+ "message 1",
284
+ "message 2",
285
+ "message 3",
286
+ ])
287
+ # Wait a while, write, and wait some more
288
+ time.sleep(2)
289
+ client_middle_time = datetime.now()
290
+ time.sleep(2)
291
+ simple_write_msg(self.kafka, topic, "message 4")
292
+ time.sleep(1)
293
+
294
+ consumer = adc.consumer.Consumer(adc.consumer.ConsumerConfig(
295
+ broker_urls=[self.kafka.address],
296
+ group_id="test_consumer",
297
+ auth=self.kafka.auth,
298
+ read_forever=False,
299
+ start_at=adc.consumer.ConsumerStartPosition.EARLIEST,
300
+ ))
301
+ consumer.subscribe(topic)
302
+ stream = consumer.stream()
303
+ timestamps = [datetime.fromtimestamp(msg.timestamp()[1] / 1000.0) for msg in stream]
304
+
305
+ middle_time = timestamps[2] + (timestamps[3] - timestamps[2]) / 2
306
+ diff = middle_time - client_middle_time
307
+ logger.info(f"Difference between client and received timestamps: {diff!s}")
308
+
309
+ stream = consumer.stream(start_at=middle_time)
310
+ msg = next(stream)
311
+ self.assertEqual(msg.topic(), topic)
312
+ self.assertEqual(msg.value(), b"message 4")
313
+
185
314
  def test_consume_not_forever(self):
186
315
  topic = "test_consume_not_forever"
187
316
  simple_write_msg(self.kafka, topic, "message 1")
@@ -248,6 +377,43 @@ class KafkaIntegrationTestCase(unittest.TestCase):
248
377
  self.assertEqual(messages[1].value(), b"message 2")
249
378
  self.assertEqual(messages[2].value(), b"message 3")
250
379
 
380
+ def test_multi_topic_handling(self):
381
+ """Use a single producer object to write messages to multiple topics,
382
+ and check that a consumer can receive them all.
383
+
384
+ """
385
+ topics = ["test_multi_1", "test_multi_2"]
386
+
387
+ # Push some messages in
388
+ producer = adc.producer.Producer(adc.producer.ProducerConfig(
389
+ broker_urls=[self.kafka.address],
390
+ topic=None,
391
+ auth=self.kafka.auth,
392
+ ))
393
+ for i in range(0,8):
394
+ producer.write(str(i), topic=topics[i%2])
395
+ producer.flush()
396
+ logger.info("messages sent")
397
+
398
+ # check that we receive the messages from the right topics
399
+ consumer = adc.consumer.Consumer(adc.consumer.ConsumerConfig(
400
+ broker_urls=[self.kafka.address],
401
+ group_id="test_consumer",
402
+ auth=self.kafka.auth,
403
+ ))
404
+ consumer.subscribe(topics)
405
+ stream = consumer.stream()
406
+ total_messages = 0;
407
+ for msg in stream:
408
+ if msg.error() is not None:
409
+ raise Exception(msg.error())
410
+ idx = int(msg.value())
411
+ self.assertEqual(msg.topic(), topics[idx%2])
412
+ total_messages += 1
413
+ if total_messages == 8:
414
+ break
415
+ self.assertEqual(total_messages, 8)
416
+
251
417
 
252
418
  class KafkaDockerConnection:
253
419
  """Holds connection information for communicating with a Kafka broker running
@@ -308,6 +474,8 @@ class KafkaDockerConnection:
308
474
  if not addrs:
309
475
  return None
310
476
  ip = addrs[0]['HostIp']
477
+ if len(ip) == 0:
478
+ ip = "localhost"
311
479
  port = addrs[0]['HostPort']
312
480
  return f"{ip}:{port}"
313
481
 
@@ -373,8 +541,16 @@ class KafkaDockerConnection:
373
541
  detach=True,
374
542
  auto_remove=True,
375
543
  network=self.net.name,
376
- # Setting None below the OS pick an ephemeral port.
377
- ports={"9092/tcp": None},
544
+ # Kafka insists on redirecting consumers to one of its advertised listeners,
545
+ # which it will get wrong if it is running in a private container network.
546
+ # To fix this, we need to tell it what to advertise, which means we must
547
+ # know what port will be visible from the host system, and we cannot use an
548
+ # ephemeral port, which would be known to us only after the container is
549
+ # started. Since we have to pick something, pick 9092, which means that
550
+ # these tests cannot run if there is already an instance of Kafka running on
551
+ # the same host.
552
+ ports={"9092/tcp": 9092},
553
+ command=["/root/runServer","--advertisedListener","SASL_SSL://localhost:9092"],
378
554
  )
379
555
 
380
556
  def get_or_create_docker_network(self):
@@ -1,54 +0,0 @@
1
- import logging
2
- from typing import Callable
3
-
4
- import confluent_kafka # type: ignore
5
-
6
- logger = logging.getLogger("adc-streaming")
7
-
8
-
9
- ErrorCallback = Callable[[confluent_kafka.KafkaError], None]
10
-
11
-
12
- def log_client_errors(kafka_error: confluent_kafka.KafkaError):
13
- if kafka_error.code() == confluent_kafka.KafkaError._ALL_BROKERS_DOWN:
14
- # This error occurs very frequently. It's not nearly as fatal as it
15
- # sounds: it really indicates that the client's broker metadata has
16
- # timed out. It appears to get triggered in races during client
17
- # shutdown, too. See https://github.com/edenhill/librdkafka/issues/2543
18
- # for more background.
19
- logger.warn("client is currently disconnected from all brokers")
20
- else:
21
- logger.error(f"internal kafka error: {kafka_error}")
22
-
23
-
24
- DeliveryCallback = Callable[[confluent_kafka.KafkaError, confluent_kafka.Message], None]
25
-
26
-
27
- def log_delivery_errors(
28
- kafka_error: confluent_kafka.KafkaError,
29
- msg: confluent_kafka.Message) -> None:
30
- if kafka_error is not None:
31
- logger.error(f"delivery error: {kafka_error}")
32
-
33
-
34
- def raise_delivery_errors(kafka_error: confluent_kafka.KafkaError,
35
- msg: confluent_kafka.Message) -> None:
36
- if kafka_error is not None:
37
- raise KafkaException.from_kafka_error(kafka_error)
38
- elif msg.error() is not None:
39
- raise KafkaException.from_kafka_error(msg.error())
40
-
41
-
42
- class KafkaException(Exception):
43
- @classmethod
44
- def from_kafka_error(cls, error):
45
- return cls(error)
46
-
47
- def __init__(self, error):
48
- self.error = error
49
- self.name = error.name()
50
- self.reason = error.str()
51
- self.retriable = error.retriable()
52
- self.fatal = error.fatal()
53
- msg = f"Error communicating with Kafka: code={self.name} {self.reason}"
54
- super(KafkaException, self).__init__(msg)
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes