logreducer 3.4.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
logreducer/kafka.py ADDED
@@ -0,0 +1,300 @@
1
+ """Kafka input source and output sink for logreducer (optional ``kafka`` extra).
2
+
3
+ Consume log lines from a Kafka topic to reduce them, and/or produce the reduced
4
+ lines back to a topic. Built on ``confluent-kafka`` (librdkafka).
5
+
6
+ The client defaults (``PRODUCER_DEFAULTS`` / ``CONSUMER_DEFAULTS``, librdkafka
7
+ config names), the ``merge_config`` / credential-masking idioms and the
8
+ ``KafkaConsumerError`` / ``_PARTITION_EOF`` handling follow standard
9
+ production Kafka client practice, so a config authored for another
10
+ librdkafka-based client behaves the same here.
11
+
12
+ Two intentional consumer design choices, because a logreducer ``Source`` must
13
+ be finite and re-iterable (the reducer counts lines by exhausting the source,
14
+ and re-reads it for multi-pass modes):
15
+
16
+ * **Bounded.** The read stops when every assigned partition has reached its end
17
+ (``enable.partition.eof`` -> ``_PARTITION_EOF``), which is the deterministic
18
+ "caught up with the topic" signal. A consecutive-idle-poll count is only a
19
+ backstop for the never-assigned case (e.g. a missing topic).
20
+ * **No commit.** Offsets are never committed, so each pass re-reads from the
21
+ earliest offset - that is what makes the source re-iterable. An application
22
+ that needs offset tracking should consume itself and hand the reducer its own
23
+ iterable of ``str``.
24
+
25
+ Install: ``pip install 'logreducer[kafka]'``.
26
+ """
27
+
28
+ from __future__ import annotations
29
+
30
+ import importlib.util
31
+ import time
32
+ from collections.abc import Iterable, Iterator
33
+ from typing import Any
34
+
35
+ # Production-lean librdkafka defaults: durable produces, bounded timeouts,
36
+ # cheap batching. Any key can be overridden by the user config.
37
+ PRODUCER_DEFAULTS: dict[str, Any] = {
38
+ "acks": "all", # Wait for all replicas (durability)
39
+ "retries": 5, # Retry on transient failures
40
+ "retry.backoff.ms": 100, # Backoff between retries
41
+ "delivery.timeout.ms": 120000, # 2 minutes max delivery time
42
+ "request.timeout.ms": 30000, # 30 seconds per request
43
+ "linger.ms": 5, # Small delay for batching
44
+ "compression.type": "lz4", # Fast compression
45
+ "batch.size": 16384, # 16KB batch size
46
+ }
47
+
48
+ CONSUMER_DEFAULTS: dict[str, Any] = {
49
+ "auto.offset.reset": "earliest", # Start from beginning if no offset
50
+ "enable.auto.commit": False, # Manual control (this source never commits)
51
+ "session.timeout.ms": 45000, # 45 seconds session timeout
52
+ "heartbeat.interval.ms": 3000, # 3 seconds heartbeat
53
+ "max.poll.interval.ms": 300000, # 5 minutes max poll interval
54
+ "fetch.min.bytes": 1, # Return immediately with any data
55
+ "fetch.wait.max.ms": 500, # Max wait for fetch.min.bytes
56
+ # logreducer-specific: needed so a drained partition reports EOF, which is
57
+ # how the bounded read knows it has caught up with the topic.
58
+ "enable.partition.eof": True,
59
+ }
60
+
61
+ _CREDENTIAL_KEYS: frozenset[str] = frozenset(
62
+ {
63
+ "sasl.password",
64
+ "sasl.username",
65
+ "ssl.key.password",
66
+ "ssl.keystore.password",
67
+ "ssl.truststore.password",
68
+ }
69
+ )
70
+
71
+ _INSTALL_HINT = (
72
+ "logreducer's Kafka source/sink needs confluent-kafka. Install the extra:\n pip install 'logreducer[kafka]'"
73
+ )
74
+
75
+
76
+ def merge_config(user_config: dict[str, Any], defaults: dict[str, Any], verify_ssl: bool = True) -> dict[str, Any]:
77
+ """Overlay ``user_config`` on ``defaults``; optionally disable TLS verify.
78
+
79
+ User values win over defaults.
80
+ """
81
+ merged = {**defaults, **user_config}
82
+ if not verify_ssl:
83
+ merged["enable.ssl.certificate.verification"] = "false"
84
+ return merged
85
+
86
+
87
+ def mask_credentials(config: dict[str, Any]) -> dict[str, Any]:
88
+ """Return a copy of ``config`` with credential values replaced by ``***``.
89
+
90
+ Used in ``__repr__`` so a Kafka config never leaks secrets into logs.
91
+ """
92
+ return {k: ("***" if k in _CREDENTIAL_KEYS and v not in (None, "") else v) for k, v in config.items()}
93
+
94
+
95
+ class KafkaConsumerError(Exception):
96
+ """A fatal Kafka consumer error, carrying the broker error code if known."""
97
+
98
+ def __init__(self, message: str, error_code: int | None = None) -> None:
99
+ self.error_code = error_code
100
+ super().__init__(message)
101
+
102
+
103
+ def _normalise_config(config: str | dict[str, Any]) -> dict[str, Any]:
104
+ """A bare string is treated as ``bootstrap.servers``; a dict is copied."""
105
+ if isinstance(config, str):
106
+ return {"bootstrap.servers": config}
107
+ return dict(config)
108
+
109
+
110
+ class KafkaSource:
111
+ """A bounded, re-iterable stream of log lines from a Kafka topic.
112
+
113
+ Each iteration creates a fresh consumer, subscribes, and reads until every
114
+ assigned partition reports end-of-partition (i.e. the topic is drained),
115
+ decoding each message value to a ``str`` line. Offsets are never committed,
116
+ so every pass re-reads from the earliest offset.
117
+ """
118
+
119
+ def __init__(
120
+ self,
121
+ config: str | dict[str, Any],
122
+ group_id: str,
123
+ topics: str | list[str],
124
+ *,
125
+ verify_ssl: bool = True,
126
+ max_messages: int | None = None,
127
+ poll_timeout: float = 1.0,
128
+ idle_polls: int = 3,
129
+ assignment_timeout: float = 30.0,
130
+ encoding: str = "utf-8",
131
+ ) -> None:
132
+ """Build a Kafka source.
133
+
134
+ Args:
135
+ config: ``bootstrap.servers`` string, or a full librdkafka config.
136
+ group_id: Consumer group id.
137
+ topics: A topic name or list of names to subscribe to.
138
+ verify_ssl: If False, disable TLS certificate verification.
139
+ max_messages: Optional hard cap on lines read per pass.
140
+ poll_timeout: Seconds to block per ``poll`` call.
141
+ idle_polls: Backstop for a never-EOF broker - once partitions are
142
+ assigned, stop after this many consecutive empty polls.
143
+ assignment_timeout: Backstop for the never-assigned case (unreachable
144
+ broker, missing topic, auth failure). If no partition is assigned
145
+ within this many seconds, iteration raises instead of hanging.
146
+ encoding: Text encoding for decoding message values.
147
+ """
148
+ if importlib.util.find_spec("confluent_kafka") is None: # pragma: no cover - only without the extra
149
+ raise ImportError(_INSTALL_HINT)
150
+
151
+ config = _normalise_config(config)
152
+ config["group.id"] = group_id
153
+ self._config = merge_config(config, CONSUMER_DEFAULTS, verify_ssl=verify_ssl)
154
+ self.topics = [topics] if isinstance(topics, str) else list(topics)
155
+ self.max_messages = max_messages
156
+ self.poll_timeout = poll_timeout
157
+ self.idle_polls = idle_polls
158
+ self.assignment_timeout = assignment_timeout
159
+ self.encoding = encoding
160
+
161
+ def __iter__(self) -> Iterator[str]:
162
+ from confluent_kafka import Consumer, KafkaError
163
+
164
+ consumer = Consumer(self._config)
165
+ consumer.subscribe(self.topics)
166
+ eof_partitions: set[tuple[str, int]] = set()
167
+ empty_polls = 0
168
+ emitted = 0
169
+ # Never-assigned backstop: an unreachable broker, a missing topic, or an
170
+ # auth failure leaves assignment() empty forever, so empty_polls (gated
171
+ # on assignment) never fires and the loop would spin indefinitely. Bail
172
+ # if no partition is ever assigned within the timeout - a stuck
173
+ # dependency surfaced as a real error, not a silent hang.
174
+ ever_assigned = False
175
+ assignment_deadline = time.monotonic() + self.assignment_timeout
176
+ try:
177
+ while True:
178
+ if self.max_messages is not None and emitted >= self.max_messages:
179
+ break
180
+
181
+ msg = consumer.poll(self.poll_timeout)
182
+
183
+ if consumer.assignment():
184
+ ever_assigned = True
185
+ elif not ever_assigned and time.monotonic() >= assignment_deadline:
186
+ raise KafkaConsumerError(
187
+ f"no partition assignment within {self.assignment_timeout:.0f}s for "
188
+ f"topics {self.topics!r} (broker unreachable, missing topic, or auth failure?)"
189
+ )
190
+
191
+ if msg is None:
192
+ # Before partitions are assigned an empty poll just means the
193
+ # group is still joining - don't count it. Once assigned,
194
+ # repeated empties are the backstop for a never-EOF broker.
195
+ if consumer.assignment():
196
+ empty_polls += 1
197
+ if empty_polls >= self.idle_polls:
198
+ break
199
+ continue
200
+ empty_polls = 0
201
+
202
+ error = msg.error()
203
+ if error:
204
+ if error.code() == KafkaError._PARTITION_EOF:
205
+ topic, partition = msg.topic(), msg.partition()
206
+ if topic is not None and partition is not None:
207
+ eof_partitions.add((topic, partition))
208
+ assigned = {(tp.topic, tp.partition) for tp in consumer.assignment()}
209
+ # Drained: every partition we hold has reported its end.
210
+ if assigned and eof_partitions >= assigned:
211
+ break
212
+ continue
213
+ raise KafkaConsumerError(f"Kafka error: {error.str()}", error_code=error.code())
214
+
215
+ value = msg.value()
216
+ if value is None:
217
+ continue
218
+ line = value.decode(self.encoding, errors="replace").strip()
219
+ if line:
220
+ emitted += 1
221
+ yield line
222
+ finally:
223
+ consumer.close()
224
+
225
+ def __repr__(self) -> str:
226
+ return f"KafkaSource(topics={self.topics!r}, config={mask_credentials(self._config)!r})"
227
+
228
+
229
+ class KafkaSink:
230
+ """A Sink that produces reduced lines to a Kafka topic.
231
+
232
+ ``write`` produces each line as a UTF-8 message value and flushes at the
233
+ end, returning the number of lines produced. Uses the standard
234
+ produce/poll/flush shape over ``PRODUCER_DEFAULTS``.
235
+ """
236
+
237
+ def __init__(
238
+ self,
239
+ config: str | dict[str, Any],
240
+ topic: str,
241
+ *,
242
+ key: str | None = None,
243
+ verify_ssl: bool = True,
244
+ flush_timeout: float | None = None,
245
+ encoding: str = "utf-8",
246
+ ) -> None:
247
+ """Build a Kafka sink.
248
+
249
+ Args:
250
+ config: ``bootstrap.servers`` string, or a full librdkafka config.
251
+ topic: Target topic for produced lines.
252
+ key: Optional fixed message key applied to every line.
253
+ verify_ssl: If False, disable TLS certificate verification.
254
+ flush_timeout: Seconds to wait on the final flush (None = infinite).
255
+ encoding: Text encoding for the produced message values.
256
+ """
257
+ try:
258
+ from confluent_kafka import Producer
259
+ except ImportError as exc: # pragma: no cover - exercised only without the extra
260
+ raise ImportError(_INSTALL_HINT) from exc
261
+
262
+ self._config = merge_config(_normalise_config(config), PRODUCER_DEFAULTS, verify_ssl=verify_ssl)
263
+ self.topic = topic
264
+ self.key = key
265
+ self.flush_timeout = flush_timeout
266
+ self.encoding = encoding
267
+ self._producer = Producer(self._config)
268
+
269
+ def write(self, lines: Iterable[str]) -> int:
270
+ """Produce each line to the topic and flush; return the number produced.
271
+
272
+ With the default ``flush_timeout=None`` the final flush blocks until the
273
+ queue drains, so the count is effectively a delivery count. With a finite
274
+ ``flush_timeout`` some messages may still be in flight when flush returns.
275
+ """
276
+ key_bytes = self.key.encode(self.encoding) if self.key is not None else None
277
+ count = 0
278
+ for line in lines:
279
+ self._producer.produce(self.topic, value=line.encode(self.encoding), key=key_bytes)
280
+ self._producer.poll(0) # Trigger delivery callbacks (non-blocking)
281
+ count += 1
282
+ self.flush()
283
+ return count
284
+
285
+ def flush(self) -> int:
286
+ """Wait for outstanding messages; return the count still in queue."""
287
+ if self.flush_timeout is not None:
288
+ result: int = self._producer.flush(self.flush_timeout)
289
+ else:
290
+ result = self._producer.flush()
291
+ return result
292
+
293
+ def __enter__(self) -> KafkaSink:
294
+ return self
295
+
296
+ def __exit__(self, *exc: object) -> None:
297
+ self.flush()
298
+
299
+ def __repr__(self) -> str:
300
+ return f"KafkaSink(topic={self.topic!r}, config={mask_credentials(self._config)!r})"
@@ -0,0 +1,193 @@
1
+ """Logging configuration for LogReducer.
2
+
3
+ Output follows one strict standard so logreducer's logs sit uniformly beside a
4
+ host application's logs in a shared deployment: RFC 3339 timestamps, a
5
+ ``time | LEVEL | name:function:line - message`` layout, ASCII-only in files,
6
+ solarized colours in an interactive terminal, a plain prefix in CI, and one
7
+ JSON object per line when ``LOG_FORMAT=json``.
8
+
9
+ logreducer is a library, not a service, so it only ever manages its OWN loguru
10
+ handlers (filtered to ``logreducer`` records) and never removes a host
11
+ application's handlers - embedding logreducer inside another application must
12
+ not disturb that application's logging.
13
+
14
+ ENV overrides:
15
+ LOG_LEVEL=DEBUG
16
+ LOG_FORMAT=json|text
17
+ LOG_OUTPUT=stdout|stderr
18
+ """
19
+
20
+ import contextlib
21
+ import os
22
+ import sys
23
+ from pathlib import Path
24
+ from typing import TYPE_CHECKING
25
+
26
+ from loguru import logger
27
+
28
+ if TYPE_CHECKING:
29
+ from loguru import Logger
30
+
31
+ # Solarized palette (https://ethanschoonover.com/solarized/) for the console
32
+ # sink - readable on both light and dark terminal backgrounds.
33
+ SOLARIZED = {
34
+ "base01": "#586e75",
35
+ "green": "#859900",
36
+ "cyan": "#2aa198",
37
+ "blue": "#268bd2",
38
+ "yellow": "#b58900",
39
+ "orange": "#cb4b16",
40
+ "red": "#dc322f",
41
+ }
42
+
43
+ # Handler ids this module has registered, so re-configuring only tears down our
44
+ # own sinks - never the host application's.
45
+ _HANDLER_IDS: list[int] = []
46
+
47
+
48
+ def _is_ci() -> bool:
49
+ """True in a CI environment (colours off, plain prefix format)."""
50
+ return (
51
+ os.getenv("CI") == "true"
52
+ or os.getenv("GITHUB_ACTIONS") == "true"
53
+ or os.getenv("GITLAB_CI") == "true"
54
+ or os.getenv("JENKINS_URL") is not None
55
+ )
56
+
57
+
58
+ def _is_interactive() -> bool:
59
+ """True only for an interactive UTF-8 terminal (never a pipe/container)."""
60
+ if _is_ci() or not sys.stderr.isatty():
61
+ return False
62
+ term = os.getenv("TERM", "")
63
+ if not term or term == "dumb":
64
+ return False
65
+ locale = (os.getenv("LC_ALL") or os.getenv("LANG") or "").upper()
66
+ return "UTF-8" in locale or "UTF8" in locale
67
+
68
+
69
+ def _console_format(colorize: bool, ci: bool) -> str:
70
+ """Console format string - solarized colours, or a plain CI prefix."""
71
+ if ci or not colorize:
72
+ return "[{level: <8}] {name}:{function}:{line} - {message}"
73
+ return (
74
+ f"<fg {SOLARIZED['green']}>{{time:YYYY-MM-DDTHH:mm:ss.SSSZZ}}</fg {SOLARIZED['green']}> | "
75
+ f"<level>{{level: <8}}</level> | "
76
+ f"<fg {SOLARIZED['cyan']}>{{name}}</fg {SOLARIZED['cyan']}>:"
77
+ f"<fg {SOLARIZED['cyan']}>{{function}}</fg {SOLARIZED['cyan']}>:"
78
+ f"<fg {SOLARIZED['cyan']}>{{line}}</fg {SOLARIZED['cyan']}> - "
79
+ f"<level>{{message}}</level>"
80
+ )
81
+
82
+
83
+ def _file_format() -> str:
84
+ """File format string - ASCII only, RFC 3339 timestamp with bracketed level."""
85
+ return "{time:YYYY-MM-DDTHH:mm:ss.SSSZZ} [{level: <8}] {name}:{function}:{line} - {message}"
86
+
87
+
88
+ def _apply_solarized_levels() -> None:
89
+ """Colour the loguru levels with the solarized scheme."""
90
+ logger.level("TRACE", color=f"<fg {SOLARIZED['base01']}>")
91
+ logger.level("DEBUG", color=f"<fg {SOLARIZED['base01']}>")
92
+ logger.level("INFO", color=f"<fg {SOLARIZED['blue']}>")
93
+ logger.level("SUCCESS", color=f"<fg {SOLARIZED['green']}>")
94
+ logger.level("WARNING", color=f"<fg {SOLARIZED['yellow']}>")
95
+ logger.level("ERROR", color=f"<fg {SOLARIZED['orange']}>")
96
+ logger.level("CRITICAL", color=f"<fg {SOLARIZED['red']}>")
97
+
98
+
99
+ def setup_logging(
100
+ enable: bool = False,
101
+ log_file: str | None = None,
102
+ log_level: str = "INFO",
103
+ log_format: str = "text",
104
+ console: bool = False,
105
+ own_sinks: bool = True,
106
+ ) -> None:
107
+ """Configure logreducer's own logging sinks.
108
+
109
+ Args:
110
+ enable: Whether logreducer emits logs at all (default False).
111
+ log_file: Optional path for an ASCII, rotating file sink.
112
+ log_level: Threshold level (overridden by ``LOG_LEVEL``).
113
+ log_format: ``text`` (human) or ``json`` (overridden by ``LOG_FORMAT``).
114
+ console: Also log to the console (stderr, or ``LOG_OUTPUT=stdout``).
115
+ own_sinks: When False, register NO handlers - just enable the package
116
+ and let logreducer's records flow through the HOST application's
117
+ loguru handlers. This is the embedding seam: a host app that owns
118
+ logging calls ``setup_logging(enable=True, own_sinks=False)`` and
119
+ logreducer's output lands in the host's sinks, formatted by the
120
+ host's standard.
121
+ """
122
+ # Env beats caller, so a deployment can retune logging without code changes.
123
+ log_level = os.environ.get("LOG_LEVEL", log_level).upper()
124
+ fmt_selector = (os.environ.get("LOG_FORMAT") or log_format or "text").strip().lower()
125
+ serialize = fmt_selector == "json"
126
+
127
+ # Tear down ONLY our previously registered handlers.
128
+ global _HANDLER_IDS
129
+ for handler_id in _HANDLER_IDS:
130
+ try:
131
+ logger.remove(handler_id)
132
+ except ValueError:
133
+ pass
134
+ _HANDLER_IDS = []
135
+
136
+ if not enable:
137
+ logger.disable("logreducer")
138
+ return
139
+
140
+ if not own_sinks:
141
+ # Host-owned logging: enable the package, register nothing.
142
+ logger.enable("logreducer")
143
+ return
144
+
145
+ ci = _is_ci()
146
+ stream = sys.stdout if os.environ.get("LOG_OUTPUT") == "stdout" else sys.stderr
147
+
148
+ if console:
149
+ # CLI/app mode owns the console: drop loguru's default stderr handler
150
+ # (id 0) so our formatted line is not printed twice. Only the pristine
151
+ # default is removed - a host app's own handlers (ids > 0) are untouched
152
+ # (embedding logreducer only ever goes through the console=False path).
153
+ with contextlib.suppress(ValueError):
154
+ logger.remove(0)
155
+ colorize = not serialize and not ci and _is_interactive()
156
+ if colorize:
157
+ _apply_solarized_levels()
158
+ _HANDLER_IDS.append(
159
+ logger.add(
160
+ stream,
161
+ level=log_level,
162
+ format=_console_format(colorize, ci),
163
+ colorize=colorize,
164
+ serialize=serialize,
165
+ filter="logreducer",
166
+ )
167
+ )
168
+
169
+ if log_file:
170
+ try:
171
+ Path(log_file).parent.mkdir(parents=True, exist_ok=True)
172
+ _HANDLER_IDS.append(
173
+ logger.add(
174
+ log_file,
175
+ level=log_level,
176
+ format=_file_format(),
177
+ rotation="10 MB",
178
+ retention="7 days",
179
+ encoding="utf-8",
180
+ serialize=serialize,
181
+ filter="logreducer",
182
+ )
183
+ )
184
+ except OSError as exc:
185
+ # A bad log path should never take the process down.
186
+ logger.warning(f"Could not create log file {log_file}: {exc}")
187
+
188
+ logger.enable("logreducer")
189
+
190
+
191
+ def get_logger(name: str = "logreducer") -> "Logger":
192
+ """Return a logger bound for the given logreducer submodule name."""
193
+ return logger.bind(name=name)