logreducer 3.4.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
logreducer/memory.py ADDED
@@ -0,0 +1,247 @@
1
+ """
2
+ Memory management and monitoring utilities
3
+ """
4
+
5
+ import gc
6
+ import os
7
+ import random
8
+ from collections import deque
9
+ from collections.abc import Iterable, Iterator
10
+ from typing import Any
11
+
12
+ import psutil
13
+
14
+
15
+ class MemoryMonitor:
16
+ """Monitor and control memory usage during processing"""
17
+
18
+ def __init__(self, max_memory_gb: float = 1.0):
19
+ self.max_memory_gb = max_memory_gb
20
+ self.max_memory_bytes = max_memory_gb * 1024 * 1024 * 1024
21
+ self.effective_limit = self.max_memory_bytes * 0.8 # 80% safety margin
22
+ self.process = psutil.Process()
23
+ self.peak_usage_bytes = 0
24
+
25
+ # Initialize peak usage
26
+ current_usage = self.process.memory_info().rss
27
+ self.peak_usage_bytes = current_usage
28
+
29
+ # Calculate safe parameters
30
+ avg_line_size = 200 # bytes
31
+ python_overhead = 3
32
+ self.max_lines_in_memory = int(self.effective_limit / (avg_line_size * python_overhead))
33
+ self.safe_chunk_size = max(1000, self.max_lines_in_memory // 10)
34
+
35
+ def check_memory(self) -> tuple[float, bool]:
36
+ """Return (current_gb, is_safe) against the SOFT 80% threshold.
37
+
38
+ Not a pure getter: when the soft threshold is crossed this runs a
39
+ gc.collect() and re-tests, so transient garbage does not trigger
40
+ degradation. Callers use this as the soft tier and
41
+ ``is_limit_exceeded`` (the full limit) as the hard tier.
42
+ """
43
+ current = self.process.memory_info().rss
44
+ current_gb = current / (1024**3)
45
+ is_safe = current < self.effective_limit
46
+
47
+ # Update peak usage
48
+ if current > self.peak_usage_bytes:
49
+ self.peak_usage_bytes = current
50
+
51
+ if not is_safe:
52
+ gc.collect()
53
+ current = self.process.memory_info().rss
54
+ current_gb = current / (1024**3)
55
+ is_safe = current < self.effective_limit
56
+
57
+ return current_gb, is_safe
58
+
59
+ def get_current_usage_gb(self) -> float:
60
+ """Get current memory usage in GB"""
61
+ current = self.process.memory_info().rss
62
+ # Update peak usage
63
+ if current > self.peak_usage_bytes:
64
+ self.peak_usage_bytes = current
65
+ return float(current / (1024**3))
66
+
67
+ def get_peak_usage_gb(self) -> float:
68
+ """Get peak memory usage in GB"""
69
+ # Update current peak if needed
70
+ current = self.process.memory_info().rss
71
+ if current > self.peak_usage_bytes:
72
+ self.peak_usage_bytes = current
73
+ return self.peak_usage_bytes / (1024**3)
74
+
75
+ def is_limit_exceeded(self) -> bool:
76
+ """True when RSS exceeds the FULL configured limit (the hard tier).
77
+
78
+ Deliberately compares against ``max_memory_bytes``, not the 80%
79
+ ``effective_limit`` used by ``check_memory`` - the gap between the two
80
+ is what gives callers room to soft-degrade (shrink batches, sample)
81
+ before the hard stop fires.
82
+ """
83
+ current = self.process.memory_info().rss
84
+ return bool(current > self.max_memory_bytes)
85
+
86
+ def reset(self) -> None:
87
+ """Reset peak memory tracking"""
88
+ current_usage = self.process.memory_info().rss
89
+ self.peak_usage_bytes = current_usage
90
+
91
+ def estimate_file_strategy(self, file_size_bytes: int) -> str:
92
+ """Determine processing strategy"""
93
+ file_size_gb = file_size_bytes / (1024**3)
94
+
95
+ if file_size_gb < self.max_memory_gb * 0.2:
96
+ return "full"
97
+ elif file_size_gb < self.max_memory_gb * 5:
98
+ return "chunked"
99
+ else:
100
+ return "sampled"
101
+
102
+
103
+ class StreamingProcessor:
104
+ """Process large files with constant memory usage"""
105
+
106
+ def __init__(self, memory_monitor: MemoryMonitor):
107
+ self.memory_monitor = memory_monitor
108
+ self.chunk_size = memory_monitor.safe_chunk_size
109
+
110
+ def read_file_streaming(self, file_path: str) -> Iterator[str]:
111
+ """Read file with minimal memory usage"""
112
+ file_size = os.path.getsize(file_path)
113
+ strategy = self.memory_monitor.estimate_file_strategy(file_size)
114
+
115
+ if strategy == "full":
116
+ with open(file_path, encoding="utf-8", errors="ignore") as f:
117
+ for line in f:
118
+ line = line.strip()
119
+ if line:
120
+ yield line
121
+
122
+ elif strategy == "chunked":
123
+ yield from self._read_chunked(file_path)
124
+
125
+ else: # sampled
126
+ yield from self._read_sampled(file_path)
127
+
128
+ def _read_chunked(self, file_path: str) -> Iterator[str]:
129
+ """Read file in chunks, degrading to sampling under memory pressure.
130
+
131
+ The memory check runs on its own line counter, NOT on chunk-flush
132
+ boundaries - a flush count and a line count almost never coincide, so
133
+ gating the check on both used to make this fallback effectively dead.
134
+ """
135
+ chunk: list[str] = []
136
+ check_every = 100_000
137
+ next_check = check_every
138
+
139
+ with open(file_path, encoding="utf-8", errors="ignore") as f:
140
+ for line_num, raw in enumerate(line.strip() for line in f):
141
+ if not raw:
142
+ continue
143
+
144
+ chunk.append(raw)
145
+
146
+ if len(chunk) >= self.chunk_size:
147
+ yield from chunk
148
+ chunk = []
149
+
150
+ if line_num >= next_check:
151
+ next_check += check_every
152
+ _, is_safe = self.memory_monitor.check_memory()
153
+ if not is_safe:
154
+ yield from chunk
155
+ yield from self._read_sampled_remainder(f)
156
+ return
157
+
158
+ if chunk:
159
+ yield from chunk
160
+
161
+ def _read_sampled(self, file_path: str) -> Iterator[str]:
162
+ """Reservoir sampling for huge files (uniform, reproducible).
163
+
164
+ Uses Algorithm L with a fixed-seed RNG so every pass over this
165
+ re-iterable FileSource yields the SAME sample - the reducer's multi-pass
166
+ modes need a stable input. (The old ``hash(line) % (i+1)`` reservoir was
167
+ neither uniform nor stable across processes.)
168
+ """
169
+ from .sampling import reservoir_sample
170
+
171
+ reservoir_size = min(self.chunk_size, 100000)
172
+
173
+ def _lines() -> Iterator[str]:
174
+ with open(file_path, encoding="utf-8", errors="ignore") as f:
175
+ for line in f:
176
+ stripped = line.strip()
177
+ if stripped:
178
+ yield stripped
179
+
180
+ yield from reservoir_sample(_lines(), reservoir_size, random.Random(0))
181
+
182
+ def _read_sampled_remainder(self, file_handle: Any) -> Iterator[str]:
183
+ """Reservoir-sample the rest of an already-open file (mid-chunk fallback)."""
184
+ from .sampling import reservoir_sample
185
+
186
+ def _lines() -> Iterator[str]:
187
+ for line in file_handle:
188
+ stripped = line.strip()
189
+ if stripped:
190
+ yield stripped
191
+
192
+ yield from reservoir_sample(_lines(), 10000, random.Random(0))
193
+
194
+
195
+ class BoundedDeduplicator:
196
+ """Memory-bounded exact deduplication (sliding-window semantics).
197
+
198
+ The seen-set is capped at ``max_cache_size`` hashes with FIFO eviction, so
199
+ on very high-cardinality input a line can re-emit after its hash has been
200
+ evicted - bounded memory is traded for global dedup guarantees.
201
+ """
202
+
203
+ def __init__(self, max_cache_size: int = 100000, hash_algorithm: str = "xxhash"):
204
+ self.max_cache_size = max_cache_size
205
+
206
+ # Choose hash function
207
+ if hash_algorithm == "xxhash":
208
+ try:
209
+ import xxhash
210
+
211
+ self.hash_func = lambda x: xxhash.xxh64(x.encode()).hexdigest()
212
+ except ImportError:
213
+ import hashlib
214
+
215
+ self.hash_func = lambda x: hashlib.blake2b(x.encode(), digest_size=16).hexdigest()
216
+ else:
217
+ import hashlib
218
+
219
+ self.hash_func = lambda x: hashlib.md5(x.encode(), usedforsecurity=False).hexdigest()
220
+
221
+ self.seen_hashes: deque[str] = deque(maxlen=max_cache_size)
222
+ self.seen_set: set[str] = set()
223
+ self.stats = {"total": 0, "unique": 0, "duplicates": 0}
224
+
225
+ def is_duplicate(self, line: str) -> bool:
226
+ """Check if line is duplicate"""
227
+ self.stats["total"] += 1
228
+ line_hash = self.hash_func(line)
229
+
230
+ if line_hash in self.seen_set:
231
+ self.stats["duplicates"] += 1
232
+ return True
233
+
234
+ if len(self.seen_hashes) >= self.max_cache_size:
235
+ old_hash = self.seen_hashes[0]
236
+ self.seen_set.discard(old_hash)
237
+
238
+ self.seen_hashes.append(line_hash)
239
+ self.seen_set.add(line_hash)
240
+ self.stats["unique"] += 1
241
+ return False
242
+
243
+ def deduplicate_lines(self, lines: Iterable[str]) -> Iterator[str]:
244
+ """Deduplicate a stream of lines"""
245
+ for line in lines:
246
+ if not self.is_duplicate(line):
247
+ yield line
logreducer/patterns.py ADDED
@@ -0,0 +1,154 @@
1
+ """Pattern extraction (Drain3 template mining) and fuzzy deduplication."""
2
+
3
+ from collections.abc import Iterable, Iterator
4
+ from dataclasses import dataclass, field
5
+ from typing import TYPE_CHECKING
6
+
7
+ from drain3 import TemplateMiner
8
+ from drain3.template_miner_config import TemplateMinerConfig
9
+ from loguru import logger
10
+
11
+ if TYPE_CHECKING:
12
+ from .config import BigDialConfig
13
+
14
+ # datasketch powers fuzzy dedup and ships in the optional `enhanced` extra;
15
+ # without it FuzzyDeduplicator degrades to a pass-through (with a warning).
16
+ try:
17
+ from datasketch import MinHash, MinHashLSH
18
+
19
+ MINHASH_AVAILABLE = True
20
+ except ImportError:
21
+ MINHASH_AVAILABLE = False
22
+
23
+
24
+ @dataclass
25
+ class LogPattern:
26
+ """Represents a unique log pattern."""
27
+
28
+ template: str
29
+ examples: list[str] = field(default_factory=list)
30
+ count: int = 0
31
+ priority: float = 0.0
32
+ anomaly_score: float = 0.0
33
+ metadata: dict = field(default_factory=dict)
34
+
35
+
36
+ class PatternExtractor:
37
+ """Extract patterns using Drain3's online template miner."""
38
+
39
+ def __init__(self, config: "BigDialConfig") -> None:
40
+ self.config = config
41
+ self.setup_drain3()
42
+
43
+ def setup_drain3(self) -> None:
44
+ """Configure Drain3."""
45
+ drain_config = TemplateMinerConfig()
46
+ drain_config.profiling_enabled = False
47
+ drain_config.drain_sim_th = self.config.drain_similarity
48
+ drain_config.drain_depth = 4
49
+ drain_config.snapshot_interval_minutes = 0 # Disable snapshots
50
+ drain_config.snapshot_compress_state = False
51
+ # Bound the template store when configured: Drain3 LRU-evicts beyond
52
+ # drain_max_clusters, keeping memory flat on high-cardinality logs.
53
+ drain_config.drain_max_clusters = self.config.max_clusters
54
+ self.miner = TemplateMiner(config=drain_config)
55
+
56
+ def extract_patterns(self, lines: Iterable[str]) -> list[LogPattern]:
57
+ """Extract patterns from lines (accepts a stream; only clusters are held).
58
+
59
+ Feeds each line into Drain3's online miner, so the caller can pass a
60
+ generator - no need to materialise the whole line set. Memory is bounded
61
+ by the number of clusters (see ``max_clusters``), not the line count.
62
+ """
63
+ pattern_map = {}
64
+
65
+ for line in lines:
66
+ result = self.miner.add_log_message(line)
67
+ cluster_id = result["cluster_id"]
68
+
69
+ if cluster_id not in pattern_map:
70
+ cluster = self.miner.drain.id_to_cluster[cluster_id]
71
+ pattern = LogPattern(template=cluster.get_template(), count=0)
72
+ pattern_map[cluster_id] = pattern
73
+
74
+ pattern = pattern_map[cluster_id]
75
+ pattern.count += 1
76
+
77
+ if len(pattern.examples) < self.config.examples_per_pattern:
78
+ pattern.examples.append(line)
79
+
80
+ patterns = list(pattern_map.values())
81
+
82
+ # Apply filters
83
+ patterns = self._filter_by_occurrence(patterns)
84
+ patterns = self._calculate_priority(patterns)
85
+
86
+ # Sort by priority and limit
87
+ patterns.sort(key=lambda p: p.priority, reverse=True)
88
+ return patterns[: self.config.max_patterns]
89
+
90
+ def _filter_by_occurrence(self, patterns: list[LogPattern]) -> list[LogPattern]:
91
+ """Filter patterns by minimum occurrence."""
92
+ return [p for p in patterns if p.count >= self.config.min_pattern_occurrences]
93
+
94
+ def _calculate_priority(self, patterns: list[LogPattern]) -> list[LogPattern]:
95
+ """Calculate priority scores.
96
+
97
+ Severity boosts are checked highest-first so a template containing both
98
+ (e.g. "CRITICAL: write FAILED") scores as critical, not merely error.
99
+ """
100
+ for pattern in patterns:
101
+ priority: float = pattern.count
102
+
103
+ template_upper = pattern.template.upper()
104
+ if "CRITICAL" in template_upper or "FATAL" in template_upper:
105
+ priority *= 200
106
+ elif "ERROR" in template_upper or "FAIL" in template_upper:
107
+ priority *= 100
108
+ elif "WARN" in template_upper:
109
+ priority *= 50
110
+
111
+ # Boost complex patterns
112
+ priority *= 1 + pattern.template.count("<*>") * 0.5
113
+ priority *= 1 + len(pattern.template.split()) * 0.1
114
+
115
+ pattern.priority = priority
116
+
117
+ return patterns
118
+
119
+
120
+ class FuzzyDeduplicator:
121
+ """Fuzzy (near-duplicate) deduplication using MinHash LSH."""
122
+
123
+ def __init__(self, threshold: float = 0.8):
124
+ self.threshold = threshold
125
+ self.enabled = MINHASH_AVAILABLE
126
+
127
+ if self.enabled:
128
+ self.lsh = MinHashLSH(threshold=threshold, num_perm=64)
129
+ else:
130
+ logger.warning("datasketch not installed - fuzzy dedup disabled (pip install 'logreducer[enhanced]')")
131
+
132
+ def deduplicate_stream(self, lines: Iterable[str]) -> Iterator[str]:
133
+ """Yield near-unique lines as they arrive (streaming near-dup filter).
134
+
135
+ Queries the LSH per line and inserts only new ones, so the reducer never
136
+ materialises the full unique-line list. Note the LSH itself grows with
137
+ the number of near-unique lines - that is inherent to fuzzy dedup.
138
+ """
139
+ if not self.enabled:
140
+ yield from lines
141
+ return
142
+
143
+ for i, line in enumerate(lines):
144
+ m = MinHash(num_perm=64)
145
+ for word in line.split():
146
+ m.update(word.encode("utf8"))
147
+
148
+ if not self.lsh.query(m):
149
+ self.lsh.insert(f"line_{i}", m)
150
+ yield line
151
+
152
+ def deduplicate(self, lines: Iterable[str]) -> list[str]:
153
+ """Remove near-duplicates, returning a list (eager wrapper of the stream)."""
154
+ return list(self.deduplicate_stream(lines))
logreducer/py.typed ADDED
@@ -0,0 +1 @@
1
+ # Marker file for PEP 561 - this package supports type checking
logreducer/sampling.py ADDED
@@ -0,0 +1,211 @@
1
+ """Sampling helpers: dialect-aware SQL sampling, reservoir sampling, batch sizing.
2
+
3
+ Three concerns, all pure and server-free so they unit-test without a database:
4
+
5
+ * ``build_sample_sql`` - wrap an arbitrary user query in a deterministic,
6
+ per-dialect random predicate so a fraction of rows is returned reproducibly
7
+ (needed because ``TABLESAMPLE`` only applies to a base table, not the
8
+ arbitrary ``SELECT`` a Source is given). Mirrors ibis: raise when a seed
9
+ cannot be honoured deterministically rather than fake reproducibility.
10
+ * ``build_sample_batch_sql`` - one fresh random batch of ``n`` rows
11
+ (``ORDER BY <rand> LIMIT n``) for the reduce-to-target loop (with-replacement).
12
+ * ``reservoir_sample`` (Algorithm L) and ``estimate_batch_rows`` - the
13
+ client-side primitives the target orchestrator and the memory watchdog use.
14
+ """
15
+
16
+ from __future__ import annotations
17
+
18
+ import math
19
+ import random
20
+ from collections.abc import Iterable
21
+ from itertools import islice
22
+
23
+
24
+ class SamplingNotSupported(Exception):
25
+ """A sampling request an engine cannot honour (e.g. a seed on SQLite)."""
26
+
27
+
28
+ # Per-dialect random function, used by ORDER BY <rand> for fresh batches.
29
+ RANDOM_FN: dict[str, str] = {
30
+ "postgresql": "random()",
31
+ "mysql": "rand()",
32
+ "mariadb": "rand()",
33
+ "sqlite": "random()",
34
+ "clickhouse": "rand()",
35
+ }
36
+
37
+
38
+ def _pg_sample(query: str, fraction: float, seed: int | None) -> tuple[str | None, str]:
39
+ # setseed makes random() deterministic for the rest of the connection; it
40
+ # takes a value in [-1, 1]. Each pass opens a fresh connection and re-seeds,
41
+ # so the sample is identical across passes (re-iterable). Caveat: strict
42
+ # identity also depends on PostgreSQL visiting rows in the same order -
43
+ # synchronized_seqscans on a busy table can rotate the scan start; ORDER BY
44
+ # in the query if bit-exact re-reads matter.
45
+ setup = None
46
+ if seed is not None:
47
+ s = ((seed % 2_000_000) / 1_000_000.0) - 1.0
48
+ setup = f"SELECT setseed({s!r})"
49
+ return setup, f"SELECT * FROM ({query}) AS _lr_sample WHERE random() < {float(fraction)!r}"
50
+
51
+
52
+ def _mysql_sample(query: str, fraction: float, seed: int | None) -> tuple[str | None, str]:
53
+ # RAND(seed) with a constant seed is a repeatable per-row sequence.
54
+ rand = f"rand({int(seed)})" if seed is not None else "rand()"
55
+ return None, f"SELECT * FROM ({query}) AS _lr_sample WHERE {rand} < {float(fraction)!r}"
56
+
57
+
58
+ def _sqlite_sample(query: str, fraction: float, seed: int | None) -> tuple[str | None, str]:
59
+ if seed is not None:
60
+ raise SamplingNotSupported(
61
+ "sqlite has no seedable RNG, so seeded (reproducible) sampling is unsupported; "
62
+ "omit sample_seed for best-effort sampling, or use a seedable engine (postgresql, mysql)"
63
+ )
64
+ threshold = int(float(fraction) * 1_000_000)
65
+ return None, f"SELECT * FROM ({query}) AS _lr_sample WHERE (abs(random()) % 1000000) < {threshold}"
66
+
67
+
68
+ # dialect.name -> builder. ClickHouse is handled by its own source (native
69
+ # SAMPLE clause), so it is intentionally not here.
70
+ DIALECT_SAMPLERS = {
71
+ "postgresql": _pg_sample,
72
+ "mysql": _mysql_sample,
73
+ "mariadb": _mysql_sample,
74
+ "sqlite": _sqlite_sample,
75
+ }
76
+
77
+
78
+ def _check_fraction(fraction: float) -> None:
79
+ if not (0.0 < fraction <= 1.0):
80
+ raise ValueError(f"sample fraction must be in (0, 1], got {fraction!r}")
81
+
82
+
83
+ def build_sample_sql(dialect: str, query: str, fraction: float, seed: int | None = None) -> tuple[str | None, str]:
84
+ """Wrap ``query`` so it returns ~``fraction`` of its rows, deterministically.
85
+
86
+ Returns ``(setup_sql, sampled_sql)`` where ``setup_sql`` is an optional
87
+ per-connection statement to run first (PostgreSQL ``setseed``), or None.
88
+
89
+ Raises ``ValueError`` for a bad fraction and ``SamplingNotSupported`` for a
90
+ dialect with no sampler or a seed the dialect cannot honour.
91
+ """
92
+ _check_fraction(fraction)
93
+ builder = DIALECT_SAMPLERS.get(dialect)
94
+ if builder is None:
95
+ raise SamplingNotSupported(f"no SQL sampler for dialect {dialect!r} (supported: {sorted(DIALECT_SAMPLERS)})")
96
+ return builder(query, fraction, seed)
97
+
98
+
99
+ def build_sample_batch_sql(dialect: str, query: str, n: int) -> str:
100
+ """A fresh random batch of up to ``n`` rows (``ORDER BY <rand> LIMIT n``).
101
+
102
+ Deliberately unseeded: each call returns a different sample (with-replacement),
103
+ which is what the reduce-to-target loop wants.
104
+ """
105
+ if n <= 0:
106
+ raise ValueError(f"batch size must be positive, got {n!r}")
107
+ rand = RANDOM_FN.get(dialect)
108
+ if rand is None:
109
+ raise SamplingNotSupported(f"no random function for dialect {dialect!r} (supported: {sorted(RANDOM_FN)})")
110
+ return f"SELECT * FROM ({query}) AS _lr_batch ORDER BY {rand} LIMIT {int(n)}"
111
+
112
+
113
+ def build_table_sample_query(
114
+ dialect: str,
115
+ table: str,
116
+ column: str,
117
+ *,
118
+ fraction: float,
119
+ seed: int | None = None,
120
+ method: str = "system",
121
+ where: str | None = None,
122
+ ) -> str:
123
+ """Build a native table-sampling query (``table``/``column`` already quoted).
124
+
125
+ PostgreSQL uses page-level ``TABLESAMPLE SYSTEM`` (or row-level ``BERNOULLI``)
126
+ with ``REPEATABLE(seed)`` - genuinely sub-linear, unlike the full-scan random
127
+ predicate used for arbitrary queries. Engines without TABLESAMPLE (MySQL,
128
+ SQLite) fall back to a predicate on the table (still a full scan).
129
+
130
+ Raises ``SamplingNotSupported`` for a dialect with no table sampler or a seed
131
+ it cannot honour, and ``ValueError`` for a bad fraction or method.
132
+ """
133
+ _check_fraction(fraction)
134
+ conds = [where] if where else []
135
+
136
+ if dialect == "postgresql":
137
+ m = method.lower()
138
+ if m not in ("system", "bernoulli"):
139
+ raise ValueError(f"method must be 'system' or 'bernoulli', got {method!r}")
140
+ pct = float(fraction) * 100.0
141
+ repeatable = f" REPEATABLE ({int(seed)})" if seed is not None else ""
142
+ where_sql = f" WHERE {' AND '.join(conds)}" if conds else ""
143
+ return f"SELECT {column} FROM {table} TABLESAMPLE {m.upper()} ({pct!r}){repeatable}{where_sql}"
144
+
145
+ # No TABLESAMPLE on these - a predicate on the table (full scan, but native).
146
+ if dialect in ("mysql", "mariadb"):
147
+ pred = f"rand({int(seed)}) < {float(fraction)!r}" if seed is not None else f"rand() < {float(fraction)!r}"
148
+ elif dialect == "sqlite":
149
+ if seed is not None:
150
+ raise SamplingNotSupported("sqlite has no seedable RNG; seeded table sampling is unsupported")
151
+ pred = f"(abs(random()) % 1000000) < {int(float(fraction) * 1_000_000)}"
152
+ else:
153
+ raise SamplingNotSupported(f"no table sampler for dialect {dialect!r}")
154
+
155
+ conds.append(pred)
156
+ return f"SELECT {column} FROM {table} WHERE {' AND '.join(conds)}"
157
+
158
+
159
+ def reservoir_sample(items: Iterable[str], k: int, rng: random.Random) -> list[str]:
160
+ """Uniform sample of ``k`` items from a stream of unknown length (Algorithm L).
161
+
162
+ Without replacement, O(k) memory, O(k(1 + log(n/k))) expected time. Replaces
163
+ the old ``hash(line) % (i+1)`` reservoir, which was neither uniform nor
164
+ reproducible (it depended on PYTHONHASHSEED). Seed ``rng`` for reproducibility.
165
+ """
166
+ if k <= 0:
167
+ return []
168
+ it = iter(items)
169
+ reservoir = list(islice(it, k))
170
+ if len(reservoir) < k:
171
+ return reservoir # fewer than k items - keep them all
172
+
173
+ def _u() -> float:
174
+ # Clamp away from 0.0 and 1.0 so log() and log1p(-w) never blow up.
175
+ return min(max(rng.random(), 1e-12), 1.0 - 1e-12)
176
+
177
+ w = math.exp(math.log(_u()) / k)
178
+ while True:
179
+ skip = math.floor(math.log(_u()) / math.log1p(-w))
180
+ # islice(it, skip, skip + 1) discards `skip` items and yields the next.
181
+ nxt = next(islice(it, skip, skip + 1), None)
182
+ if nxt is None:
183
+ return reservoir
184
+ reservoir[rng.randrange(k)] = nxt
185
+ w *= math.exp(math.log(_u()) / k)
186
+
187
+
188
+ def estimate_batch_rows(
189
+ sample_lines: list[str],
190
+ budget_bytes: int,
191
+ *,
192
+ floor: int = 1000,
193
+ ceil: int = 1_000_000,
194
+ align: int = 1000,
195
+ overhead: int = 3,
196
+ ) -> int:
197
+ """Rows that fit ``budget_bytes``, from the average byte size of a sample.
198
+
199
+ ``overhead`` approximates Python's per-object memory multiplier (matching
200
+ MemoryMonitor). Result is clamped to ``[floor, ceil]`` and rounded down to a
201
+ multiple of ``align`` so it lines up with the cursor fetch batch.
202
+ """
203
+ if not sample_lines or budget_bytes <= 0:
204
+ return floor
205
+ avg_bytes = sum(len(s.encode("utf-8")) for s in sample_lines) / len(sample_lines)
206
+ per_row = max(1.0, avg_bytes * overhead)
207
+ rows = int(budget_bytes / per_row)
208
+ if align > 1:
209
+ rows = (rows // align) * align
210
+ # Clamp LAST so alignment can never push the result outside [floor, ceil].
211
+ return max(floor, min(ceil, rows))
logreducer/sinks.py ADDED
@@ -0,0 +1,81 @@
1
+ """Output sinks for logreducer.
2
+
3
+ The reduction engine returns its result in memory (a ``list[str]``) - that is
4
+ the primary, dependency-free output. A *sink* is the abstraction for sending
5
+ those reduced lines somewhere else: a file, a Kafka topic, a database. Like
6
+ ``Source`` on the input side, a sink is a thin seam - the package does not own
7
+ the connection or the delivery guarantees; an application can pass its own.
8
+
9
+ ``Sink`` is a structural protocol: anything with a ``write(lines) -> int`` is a
10
+ sink. ``FileSink`` is built in; the Kafka producer sink lives behind the
11
+ ``kafka`` optional extra (see the kafka submodule).
12
+ """
13
+
14
+ from __future__ import annotations
15
+
16
+ import json
17
+ import os
18
+ from collections.abc import Iterable
19
+ from datetime import datetime
20
+ from pathlib import Path
21
+ from typing import Protocol, runtime_checkable
22
+
23
+
24
+ @runtime_checkable
25
+ class Sink(Protocol):
26
+ """A destination for reduced log lines.
27
+
28
+ ``write`` consumes an iterable of ``str`` lines and returns the number of
29
+ lines written. It may be called with a generator, so implementations should
30
+ iterate lazily rather than materialising the whole batch.
31
+ """
32
+
33
+ def write(self, lines: Iterable[str]) -> int: ...
34
+
35
+
36
+ class FileSink:
37
+ """Write reduced lines to a text file in one of three formats.
38
+
39
+ A standalone, format-aware writer (``line`` / ``json`` / ``jsonl``) that an
40
+ application can use without touching the reducer. It streams the ``line``
41
+ and ``jsonl`` formats a row at a time (constant memory); ``json`` buffers
42
+ the list because the format is a single document.
43
+
44
+ This is deliberately simpler than the reducer's own ``output_file`` path,
45
+ which additionally writes a ``.meta.json`` sidecar of run stats - that needs
46
+ the reducer's context, whereas a sink only sees the lines.
47
+ """
48
+
49
+ def __init__(self, path: str | os.PathLike[str], *, output_format: str = "line") -> None:
50
+ self.path = os.fspath(path)
51
+ fmt = output_format.lower()
52
+ if fmt not in ("line", "json", "jsonl"):
53
+ raise ValueError(f"Unknown output_format {output_format!r} (use line, json or jsonl)")
54
+ self.output_format = fmt
55
+
56
+ def write(self, lines: Iterable[str]) -> int:
57
+ output_path = Path(self.path)
58
+ output_path.parent.mkdir(parents=True, exist_ok=True)
59
+
60
+ count = 0
61
+ if self.output_format == "json":
62
+ # A JSON document is a single value - the list has to be built.
63
+ materialised = list(lines)
64
+ count = len(materialised)
65
+ with open(output_path, "w", encoding="utf-8") as f:
66
+ json.dump({"lines": materialised, "timestamp": datetime.now().isoformat()}, f, indent=2)
67
+ elif self.output_format == "jsonl":
68
+ with open(output_path, "w", encoding="utf-8") as f:
69
+ for line in lines:
70
+ json.dump({"line": line}, f)
71
+ f.write("\n")
72
+ count += 1
73
+ else: # line
74
+ with open(output_path, "w", encoding="utf-8") as f:
75
+ for line in lines:
76
+ f.write(line + "\n")
77
+ count += 1
78
+ return count
79
+
80
+ def __repr__(self) -> str:
81
+ return f"FileSink({self.path!r}, output_format={self.output_format!r})"