logreducer 3.4.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- logreducer/__init__.py +40 -0
- logreducer/anomaly.py +75 -0
- logreducer/cli.py +359 -0
- logreducer/clickhouse.py +171 -0
- logreducer/config.py +166 -0
- logreducer/core.py +493 -0
- logreducer/kafka.py +300 -0
- logreducer/logging_config.py +193 -0
- logreducer/memory.py +247 -0
- logreducer/patterns.py +154 -0
- logreducer/py.typed +1 -0
- logreducer/sampling.py +211 -0
- logreducer/sinks.py +81 -0
- logreducer/sources.py +78 -0
- logreducer/sql.py +209 -0
- logreducer/target.py +164 -0
- logreducer/temporal.py +163 -0
- logreducer-3.4.0.dist-info/METADATA +387 -0
- logreducer-3.4.0.dist-info/RECORD +23 -0
- logreducer-3.4.0.dist-info/WHEEL +4 -0
- logreducer-3.4.0.dist-info/entry_points.txt +3 -0
- logreducer-3.4.0.dist-info/licenses/LICENSE +201 -0
- logreducer-3.4.0.dist-info/licenses/NOTICE +53 -0
logreducer/memory.py
ADDED
|
@@ -0,0 +1,247 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Memory management and monitoring utilities
|
|
3
|
+
"""
|
|
4
|
+
|
|
5
|
+
import gc
|
|
6
|
+
import os
|
|
7
|
+
import random
|
|
8
|
+
from collections import deque
|
|
9
|
+
from collections.abc import Iterable, Iterator
|
|
10
|
+
from typing import Any
|
|
11
|
+
|
|
12
|
+
import psutil
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
class MemoryMonitor:
|
|
16
|
+
"""Monitor and control memory usage during processing"""
|
|
17
|
+
|
|
18
|
+
def __init__(self, max_memory_gb: float = 1.0):
|
|
19
|
+
self.max_memory_gb = max_memory_gb
|
|
20
|
+
self.max_memory_bytes = max_memory_gb * 1024 * 1024 * 1024
|
|
21
|
+
self.effective_limit = self.max_memory_bytes * 0.8 # 80% safety margin
|
|
22
|
+
self.process = psutil.Process()
|
|
23
|
+
self.peak_usage_bytes = 0
|
|
24
|
+
|
|
25
|
+
# Initialize peak usage
|
|
26
|
+
current_usage = self.process.memory_info().rss
|
|
27
|
+
self.peak_usage_bytes = current_usage
|
|
28
|
+
|
|
29
|
+
# Calculate safe parameters
|
|
30
|
+
avg_line_size = 200 # bytes
|
|
31
|
+
python_overhead = 3
|
|
32
|
+
self.max_lines_in_memory = int(self.effective_limit / (avg_line_size * python_overhead))
|
|
33
|
+
self.safe_chunk_size = max(1000, self.max_lines_in_memory // 10)
|
|
34
|
+
|
|
35
|
+
def check_memory(self) -> tuple[float, bool]:
|
|
36
|
+
"""Return (current_gb, is_safe) against the SOFT 80% threshold.
|
|
37
|
+
|
|
38
|
+
Not a pure getter: when the soft threshold is crossed this runs a
|
|
39
|
+
gc.collect() and re-tests, so transient garbage does not trigger
|
|
40
|
+
degradation. Callers use this as the soft tier and
|
|
41
|
+
``is_limit_exceeded`` (the full limit) as the hard tier.
|
|
42
|
+
"""
|
|
43
|
+
current = self.process.memory_info().rss
|
|
44
|
+
current_gb = current / (1024**3)
|
|
45
|
+
is_safe = current < self.effective_limit
|
|
46
|
+
|
|
47
|
+
# Update peak usage
|
|
48
|
+
if current > self.peak_usage_bytes:
|
|
49
|
+
self.peak_usage_bytes = current
|
|
50
|
+
|
|
51
|
+
if not is_safe:
|
|
52
|
+
gc.collect()
|
|
53
|
+
current = self.process.memory_info().rss
|
|
54
|
+
current_gb = current / (1024**3)
|
|
55
|
+
is_safe = current < self.effective_limit
|
|
56
|
+
|
|
57
|
+
return current_gb, is_safe
|
|
58
|
+
|
|
59
|
+
def get_current_usage_gb(self) -> float:
|
|
60
|
+
"""Get current memory usage in GB"""
|
|
61
|
+
current = self.process.memory_info().rss
|
|
62
|
+
# Update peak usage
|
|
63
|
+
if current > self.peak_usage_bytes:
|
|
64
|
+
self.peak_usage_bytes = current
|
|
65
|
+
return float(current / (1024**3))
|
|
66
|
+
|
|
67
|
+
def get_peak_usage_gb(self) -> float:
|
|
68
|
+
"""Get peak memory usage in GB"""
|
|
69
|
+
# Update current peak if needed
|
|
70
|
+
current = self.process.memory_info().rss
|
|
71
|
+
if current > self.peak_usage_bytes:
|
|
72
|
+
self.peak_usage_bytes = current
|
|
73
|
+
return self.peak_usage_bytes / (1024**3)
|
|
74
|
+
|
|
75
|
+
def is_limit_exceeded(self) -> bool:
|
|
76
|
+
"""True when RSS exceeds the FULL configured limit (the hard tier).
|
|
77
|
+
|
|
78
|
+
Deliberately compares against ``max_memory_bytes``, not the 80%
|
|
79
|
+
``effective_limit`` used by ``check_memory`` - the gap between the two
|
|
80
|
+
is what gives callers room to soft-degrade (shrink batches, sample)
|
|
81
|
+
before the hard stop fires.
|
|
82
|
+
"""
|
|
83
|
+
current = self.process.memory_info().rss
|
|
84
|
+
return bool(current > self.max_memory_bytes)
|
|
85
|
+
|
|
86
|
+
def reset(self) -> None:
|
|
87
|
+
"""Reset peak memory tracking"""
|
|
88
|
+
current_usage = self.process.memory_info().rss
|
|
89
|
+
self.peak_usage_bytes = current_usage
|
|
90
|
+
|
|
91
|
+
def estimate_file_strategy(self, file_size_bytes: int) -> str:
|
|
92
|
+
"""Determine processing strategy"""
|
|
93
|
+
file_size_gb = file_size_bytes / (1024**3)
|
|
94
|
+
|
|
95
|
+
if file_size_gb < self.max_memory_gb * 0.2:
|
|
96
|
+
return "full"
|
|
97
|
+
elif file_size_gb < self.max_memory_gb * 5:
|
|
98
|
+
return "chunked"
|
|
99
|
+
else:
|
|
100
|
+
return "sampled"
|
|
101
|
+
|
|
102
|
+
|
|
103
|
+
class StreamingProcessor:
|
|
104
|
+
"""Process large files with constant memory usage"""
|
|
105
|
+
|
|
106
|
+
def __init__(self, memory_monitor: MemoryMonitor):
|
|
107
|
+
self.memory_monitor = memory_monitor
|
|
108
|
+
self.chunk_size = memory_monitor.safe_chunk_size
|
|
109
|
+
|
|
110
|
+
def read_file_streaming(self, file_path: str) -> Iterator[str]:
|
|
111
|
+
"""Read file with minimal memory usage"""
|
|
112
|
+
file_size = os.path.getsize(file_path)
|
|
113
|
+
strategy = self.memory_monitor.estimate_file_strategy(file_size)
|
|
114
|
+
|
|
115
|
+
if strategy == "full":
|
|
116
|
+
with open(file_path, encoding="utf-8", errors="ignore") as f:
|
|
117
|
+
for line in f:
|
|
118
|
+
line = line.strip()
|
|
119
|
+
if line:
|
|
120
|
+
yield line
|
|
121
|
+
|
|
122
|
+
elif strategy == "chunked":
|
|
123
|
+
yield from self._read_chunked(file_path)
|
|
124
|
+
|
|
125
|
+
else: # sampled
|
|
126
|
+
yield from self._read_sampled(file_path)
|
|
127
|
+
|
|
128
|
+
def _read_chunked(self, file_path: str) -> Iterator[str]:
|
|
129
|
+
"""Read file in chunks, degrading to sampling under memory pressure.
|
|
130
|
+
|
|
131
|
+
The memory check runs on its own line counter, NOT on chunk-flush
|
|
132
|
+
boundaries - a flush count and a line count almost never coincide, so
|
|
133
|
+
gating the check on both used to make this fallback effectively dead.
|
|
134
|
+
"""
|
|
135
|
+
chunk: list[str] = []
|
|
136
|
+
check_every = 100_000
|
|
137
|
+
next_check = check_every
|
|
138
|
+
|
|
139
|
+
with open(file_path, encoding="utf-8", errors="ignore") as f:
|
|
140
|
+
for line_num, raw in enumerate(line.strip() for line in f):
|
|
141
|
+
if not raw:
|
|
142
|
+
continue
|
|
143
|
+
|
|
144
|
+
chunk.append(raw)
|
|
145
|
+
|
|
146
|
+
if len(chunk) >= self.chunk_size:
|
|
147
|
+
yield from chunk
|
|
148
|
+
chunk = []
|
|
149
|
+
|
|
150
|
+
if line_num >= next_check:
|
|
151
|
+
next_check += check_every
|
|
152
|
+
_, is_safe = self.memory_monitor.check_memory()
|
|
153
|
+
if not is_safe:
|
|
154
|
+
yield from chunk
|
|
155
|
+
yield from self._read_sampled_remainder(f)
|
|
156
|
+
return
|
|
157
|
+
|
|
158
|
+
if chunk:
|
|
159
|
+
yield from chunk
|
|
160
|
+
|
|
161
|
+
def _read_sampled(self, file_path: str) -> Iterator[str]:
|
|
162
|
+
"""Reservoir sampling for huge files (uniform, reproducible).
|
|
163
|
+
|
|
164
|
+
Uses Algorithm L with a fixed-seed RNG so every pass over this
|
|
165
|
+
re-iterable FileSource yields the SAME sample - the reducer's multi-pass
|
|
166
|
+
modes need a stable input. (The old ``hash(line) % (i+1)`` reservoir was
|
|
167
|
+
neither uniform nor stable across processes.)
|
|
168
|
+
"""
|
|
169
|
+
from .sampling import reservoir_sample
|
|
170
|
+
|
|
171
|
+
reservoir_size = min(self.chunk_size, 100000)
|
|
172
|
+
|
|
173
|
+
def _lines() -> Iterator[str]:
|
|
174
|
+
with open(file_path, encoding="utf-8", errors="ignore") as f:
|
|
175
|
+
for line in f:
|
|
176
|
+
stripped = line.strip()
|
|
177
|
+
if stripped:
|
|
178
|
+
yield stripped
|
|
179
|
+
|
|
180
|
+
yield from reservoir_sample(_lines(), reservoir_size, random.Random(0))
|
|
181
|
+
|
|
182
|
+
def _read_sampled_remainder(self, file_handle: Any) -> Iterator[str]:
|
|
183
|
+
"""Reservoir-sample the rest of an already-open file (mid-chunk fallback)."""
|
|
184
|
+
from .sampling import reservoir_sample
|
|
185
|
+
|
|
186
|
+
def _lines() -> Iterator[str]:
|
|
187
|
+
for line in file_handle:
|
|
188
|
+
stripped = line.strip()
|
|
189
|
+
if stripped:
|
|
190
|
+
yield stripped
|
|
191
|
+
|
|
192
|
+
yield from reservoir_sample(_lines(), 10000, random.Random(0))
|
|
193
|
+
|
|
194
|
+
|
|
195
|
+
class BoundedDeduplicator:
|
|
196
|
+
"""Memory-bounded exact deduplication (sliding-window semantics).
|
|
197
|
+
|
|
198
|
+
The seen-set is capped at ``max_cache_size`` hashes with FIFO eviction, so
|
|
199
|
+
on very high-cardinality input a line can re-emit after its hash has been
|
|
200
|
+
evicted - bounded memory is traded for global dedup guarantees.
|
|
201
|
+
"""
|
|
202
|
+
|
|
203
|
+
def __init__(self, max_cache_size: int = 100000, hash_algorithm: str = "xxhash"):
|
|
204
|
+
self.max_cache_size = max_cache_size
|
|
205
|
+
|
|
206
|
+
# Choose hash function
|
|
207
|
+
if hash_algorithm == "xxhash":
|
|
208
|
+
try:
|
|
209
|
+
import xxhash
|
|
210
|
+
|
|
211
|
+
self.hash_func = lambda x: xxhash.xxh64(x.encode()).hexdigest()
|
|
212
|
+
except ImportError:
|
|
213
|
+
import hashlib
|
|
214
|
+
|
|
215
|
+
self.hash_func = lambda x: hashlib.blake2b(x.encode(), digest_size=16).hexdigest()
|
|
216
|
+
else:
|
|
217
|
+
import hashlib
|
|
218
|
+
|
|
219
|
+
self.hash_func = lambda x: hashlib.md5(x.encode(), usedforsecurity=False).hexdigest()
|
|
220
|
+
|
|
221
|
+
self.seen_hashes: deque[str] = deque(maxlen=max_cache_size)
|
|
222
|
+
self.seen_set: set[str] = set()
|
|
223
|
+
self.stats = {"total": 0, "unique": 0, "duplicates": 0}
|
|
224
|
+
|
|
225
|
+
def is_duplicate(self, line: str) -> bool:
|
|
226
|
+
"""Check if line is duplicate"""
|
|
227
|
+
self.stats["total"] += 1
|
|
228
|
+
line_hash = self.hash_func(line)
|
|
229
|
+
|
|
230
|
+
if line_hash in self.seen_set:
|
|
231
|
+
self.stats["duplicates"] += 1
|
|
232
|
+
return True
|
|
233
|
+
|
|
234
|
+
if len(self.seen_hashes) >= self.max_cache_size:
|
|
235
|
+
old_hash = self.seen_hashes[0]
|
|
236
|
+
self.seen_set.discard(old_hash)
|
|
237
|
+
|
|
238
|
+
self.seen_hashes.append(line_hash)
|
|
239
|
+
self.seen_set.add(line_hash)
|
|
240
|
+
self.stats["unique"] += 1
|
|
241
|
+
return False
|
|
242
|
+
|
|
243
|
+
def deduplicate_lines(self, lines: Iterable[str]) -> Iterator[str]:
|
|
244
|
+
"""Deduplicate a stream of lines"""
|
|
245
|
+
for line in lines:
|
|
246
|
+
if not self.is_duplicate(line):
|
|
247
|
+
yield line
|
logreducer/patterns.py
ADDED
|
@@ -0,0 +1,154 @@
|
|
|
1
|
+
"""Pattern extraction (Drain3 template mining) and fuzzy deduplication."""
|
|
2
|
+
|
|
3
|
+
from collections.abc import Iterable, Iterator
|
|
4
|
+
from dataclasses import dataclass, field
|
|
5
|
+
from typing import TYPE_CHECKING
|
|
6
|
+
|
|
7
|
+
from drain3 import TemplateMiner
|
|
8
|
+
from drain3.template_miner_config import TemplateMinerConfig
|
|
9
|
+
from loguru import logger
|
|
10
|
+
|
|
11
|
+
if TYPE_CHECKING:
|
|
12
|
+
from .config import BigDialConfig
|
|
13
|
+
|
|
14
|
+
# datasketch powers fuzzy dedup and ships in the optional `enhanced` extra;
|
|
15
|
+
# without it FuzzyDeduplicator degrades to a pass-through (with a warning).
|
|
16
|
+
try:
|
|
17
|
+
from datasketch import MinHash, MinHashLSH
|
|
18
|
+
|
|
19
|
+
MINHASH_AVAILABLE = True
|
|
20
|
+
except ImportError:
|
|
21
|
+
MINHASH_AVAILABLE = False
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
@dataclass
|
|
25
|
+
class LogPattern:
|
|
26
|
+
"""Represents a unique log pattern."""
|
|
27
|
+
|
|
28
|
+
template: str
|
|
29
|
+
examples: list[str] = field(default_factory=list)
|
|
30
|
+
count: int = 0
|
|
31
|
+
priority: float = 0.0
|
|
32
|
+
anomaly_score: float = 0.0
|
|
33
|
+
metadata: dict = field(default_factory=dict)
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
class PatternExtractor:
|
|
37
|
+
"""Extract patterns using Drain3's online template miner."""
|
|
38
|
+
|
|
39
|
+
def __init__(self, config: "BigDialConfig") -> None:
|
|
40
|
+
self.config = config
|
|
41
|
+
self.setup_drain3()
|
|
42
|
+
|
|
43
|
+
def setup_drain3(self) -> None:
|
|
44
|
+
"""Configure Drain3."""
|
|
45
|
+
drain_config = TemplateMinerConfig()
|
|
46
|
+
drain_config.profiling_enabled = False
|
|
47
|
+
drain_config.drain_sim_th = self.config.drain_similarity
|
|
48
|
+
drain_config.drain_depth = 4
|
|
49
|
+
drain_config.snapshot_interval_minutes = 0 # Disable snapshots
|
|
50
|
+
drain_config.snapshot_compress_state = False
|
|
51
|
+
# Bound the template store when configured: Drain3 LRU-evicts beyond
|
|
52
|
+
# drain_max_clusters, keeping memory flat on high-cardinality logs.
|
|
53
|
+
drain_config.drain_max_clusters = self.config.max_clusters
|
|
54
|
+
self.miner = TemplateMiner(config=drain_config)
|
|
55
|
+
|
|
56
|
+
def extract_patterns(self, lines: Iterable[str]) -> list[LogPattern]:
|
|
57
|
+
"""Extract patterns from lines (accepts a stream; only clusters are held).
|
|
58
|
+
|
|
59
|
+
Feeds each line into Drain3's online miner, so the caller can pass a
|
|
60
|
+
generator - no need to materialise the whole line set. Memory is bounded
|
|
61
|
+
by the number of clusters (see ``max_clusters``), not the line count.
|
|
62
|
+
"""
|
|
63
|
+
pattern_map = {}
|
|
64
|
+
|
|
65
|
+
for line in lines:
|
|
66
|
+
result = self.miner.add_log_message(line)
|
|
67
|
+
cluster_id = result["cluster_id"]
|
|
68
|
+
|
|
69
|
+
if cluster_id not in pattern_map:
|
|
70
|
+
cluster = self.miner.drain.id_to_cluster[cluster_id]
|
|
71
|
+
pattern = LogPattern(template=cluster.get_template(), count=0)
|
|
72
|
+
pattern_map[cluster_id] = pattern
|
|
73
|
+
|
|
74
|
+
pattern = pattern_map[cluster_id]
|
|
75
|
+
pattern.count += 1
|
|
76
|
+
|
|
77
|
+
if len(pattern.examples) < self.config.examples_per_pattern:
|
|
78
|
+
pattern.examples.append(line)
|
|
79
|
+
|
|
80
|
+
patterns = list(pattern_map.values())
|
|
81
|
+
|
|
82
|
+
# Apply filters
|
|
83
|
+
patterns = self._filter_by_occurrence(patterns)
|
|
84
|
+
patterns = self._calculate_priority(patterns)
|
|
85
|
+
|
|
86
|
+
# Sort by priority and limit
|
|
87
|
+
patterns.sort(key=lambda p: p.priority, reverse=True)
|
|
88
|
+
return patterns[: self.config.max_patterns]
|
|
89
|
+
|
|
90
|
+
def _filter_by_occurrence(self, patterns: list[LogPattern]) -> list[LogPattern]:
|
|
91
|
+
"""Filter patterns by minimum occurrence."""
|
|
92
|
+
return [p for p in patterns if p.count >= self.config.min_pattern_occurrences]
|
|
93
|
+
|
|
94
|
+
def _calculate_priority(self, patterns: list[LogPattern]) -> list[LogPattern]:
|
|
95
|
+
"""Calculate priority scores.
|
|
96
|
+
|
|
97
|
+
Severity boosts are checked highest-first so a template containing both
|
|
98
|
+
(e.g. "CRITICAL: write FAILED") scores as critical, not merely error.
|
|
99
|
+
"""
|
|
100
|
+
for pattern in patterns:
|
|
101
|
+
priority: float = pattern.count
|
|
102
|
+
|
|
103
|
+
template_upper = pattern.template.upper()
|
|
104
|
+
if "CRITICAL" in template_upper or "FATAL" in template_upper:
|
|
105
|
+
priority *= 200
|
|
106
|
+
elif "ERROR" in template_upper or "FAIL" in template_upper:
|
|
107
|
+
priority *= 100
|
|
108
|
+
elif "WARN" in template_upper:
|
|
109
|
+
priority *= 50
|
|
110
|
+
|
|
111
|
+
# Boost complex patterns
|
|
112
|
+
priority *= 1 + pattern.template.count("<*>") * 0.5
|
|
113
|
+
priority *= 1 + len(pattern.template.split()) * 0.1
|
|
114
|
+
|
|
115
|
+
pattern.priority = priority
|
|
116
|
+
|
|
117
|
+
return patterns
|
|
118
|
+
|
|
119
|
+
|
|
120
|
+
class FuzzyDeduplicator:
|
|
121
|
+
"""Fuzzy (near-duplicate) deduplication using MinHash LSH."""
|
|
122
|
+
|
|
123
|
+
def __init__(self, threshold: float = 0.8):
|
|
124
|
+
self.threshold = threshold
|
|
125
|
+
self.enabled = MINHASH_AVAILABLE
|
|
126
|
+
|
|
127
|
+
if self.enabled:
|
|
128
|
+
self.lsh = MinHashLSH(threshold=threshold, num_perm=64)
|
|
129
|
+
else:
|
|
130
|
+
logger.warning("datasketch not installed - fuzzy dedup disabled (pip install 'logreducer[enhanced]')")
|
|
131
|
+
|
|
132
|
+
def deduplicate_stream(self, lines: Iterable[str]) -> Iterator[str]:
|
|
133
|
+
"""Yield near-unique lines as they arrive (streaming near-dup filter).
|
|
134
|
+
|
|
135
|
+
Queries the LSH per line and inserts only new ones, so the reducer never
|
|
136
|
+
materialises the full unique-line list. Note the LSH itself grows with
|
|
137
|
+
the number of near-unique lines - that is inherent to fuzzy dedup.
|
|
138
|
+
"""
|
|
139
|
+
if not self.enabled:
|
|
140
|
+
yield from lines
|
|
141
|
+
return
|
|
142
|
+
|
|
143
|
+
for i, line in enumerate(lines):
|
|
144
|
+
m = MinHash(num_perm=64)
|
|
145
|
+
for word in line.split():
|
|
146
|
+
m.update(word.encode("utf8"))
|
|
147
|
+
|
|
148
|
+
if not self.lsh.query(m):
|
|
149
|
+
self.lsh.insert(f"line_{i}", m)
|
|
150
|
+
yield line
|
|
151
|
+
|
|
152
|
+
def deduplicate(self, lines: Iterable[str]) -> list[str]:
|
|
153
|
+
"""Remove near-duplicates, returning a list (eager wrapper of the stream)."""
|
|
154
|
+
return list(self.deduplicate_stream(lines))
|
logreducer/py.typed
ADDED
|
@@ -0,0 +1 @@
|
|
|
1
|
+
# Marker file for PEP 561 - this package supports type checking
|
logreducer/sampling.py
ADDED
|
@@ -0,0 +1,211 @@
|
|
|
1
|
+
"""Sampling helpers: dialect-aware SQL sampling, reservoir sampling, batch sizing.
|
|
2
|
+
|
|
3
|
+
Three concerns, all pure and server-free so they unit-test without a database:
|
|
4
|
+
|
|
5
|
+
* ``build_sample_sql`` - wrap an arbitrary user query in a deterministic,
|
|
6
|
+
per-dialect random predicate so a fraction of rows is returned reproducibly
|
|
7
|
+
(needed because ``TABLESAMPLE`` only applies to a base table, not the
|
|
8
|
+
arbitrary ``SELECT`` a Source is given). Mirrors ibis: raise when a seed
|
|
9
|
+
cannot be honoured deterministically rather than fake reproducibility.
|
|
10
|
+
* ``build_sample_batch_sql`` - one fresh random batch of ``n`` rows
|
|
11
|
+
(``ORDER BY <rand> LIMIT n``) for the reduce-to-target loop (with-replacement).
|
|
12
|
+
* ``reservoir_sample`` (Algorithm L) and ``estimate_batch_rows`` - the
|
|
13
|
+
client-side primitives the target orchestrator and the memory watchdog use.
|
|
14
|
+
"""
|
|
15
|
+
|
|
16
|
+
from __future__ import annotations
|
|
17
|
+
|
|
18
|
+
import math
|
|
19
|
+
import random
|
|
20
|
+
from collections.abc import Iterable
|
|
21
|
+
from itertools import islice
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
class SamplingNotSupported(Exception):
|
|
25
|
+
"""A sampling request an engine cannot honour (e.g. a seed on SQLite)."""
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
# Per-dialect random function, used by ORDER BY <rand> for fresh batches.
|
|
29
|
+
RANDOM_FN: dict[str, str] = {
|
|
30
|
+
"postgresql": "random()",
|
|
31
|
+
"mysql": "rand()",
|
|
32
|
+
"mariadb": "rand()",
|
|
33
|
+
"sqlite": "random()",
|
|
34
|
+
"clickhouse": "rand()",
|
|
35
|
+
}
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def _pg_sample(query: str, fraction: float, seed: int | None) -> tuple[str | None, str]:
|
|
39
|
+
# setseed makes random() deterministic for the rest of the connection; it
|
|
40
|
+
# takes a value in [-1, 1]. Each pass opens a fresh connection and re-seeds,
|
|
41
|
+
# so the sample is identical across passes (re-iterable). Caveat: strict
|
|
42
|
+
# identity also depends on PostgreSQL visiting rows in the same order -
|
|
43
|
+
# synchronized_seqscans on a busy table can rotate the scan start; ORDER BY
|
|
44
|
+
# in the query if bit-exact re-reads matter.
|
|
45
|
+
setup = None
|
|
46
|
+
if seed is not None:
|
|
47
|
+
s = ((seed % 2_000_000) / 1_000_000.0) - 1.0
|
|
48
|
+
setup = f"SELECT setseed({s!r})"
|
|
49
|
+
return setup, f"SELECT * FROM ({query}) AS _lr_sample WHERE random() < {float(fraction)!r}"
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def _mysql_sample(query: str, fraction: float, seed: int | None) -> tuple[str | None, str]:
|
|
53
|
+
# RAND(seed) with a constant seed is a repeatable per-row sequence.
|
|
54
|
+
rand = f"rand({int(seed)})" if seed is not None else "rand()"
|
|
55
|
+
return None, f"SELECT * FROM ({query}) AS _lr_sample WHERE {rand} < {float(fraction)!r}"
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def _sqlite_sample(query: str, fraction: float, seed: int | None) -> tuple[str | None, str]:
|
|
59
|
+
if seed is not None:
|
|
60
|
+
raise SamplingNotSupported(
|
|
61
|
+
"sqlite has no seedable RNG, so seeded (reproducible) sampling is unsupported; "
|
|
62
|
+
"omit sample_seed for best-effort sampling, or use a seedable engine (postgresql, mysql)"
|
|
63
|
+
)
|
|
64
|
+
threshold = int(float(fraction) * 1_000_000)
|
|
65
|
+
return None, f"SELECT * FROM ({query}) AS _lr_sample WHERE (abs(random()) % 1000000) < {threshold}"
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
# dialect.name -> builder. ClickHouse is handled by its own source (native
|
|
69
|
+
# SAMPLE clause), so it is intentionally not here.
|
|
70
|
+
DIALECT_SAMPLERS = {
|
|
71
|
+
"postgresql": _pg_sample,
|
|
72
|
+
"mysql": _mysql_sample,
|
|
73
|
+
"mariadb": _mysql_sample,
|
|
74
|
+
"sqlite": _sqlite_sample,
|
|
75
|
+
}
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def _check_fraction(fraction: float) -> None:
|
|
79
|
+
if not (0.0 < fraction <= 1.0):
|
|
80
|
+
raise ValueError(f"sample fraction must be in (0, 1], got {fraction!r}")
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
def build_sample_sql(dialect: str, query: str, fraction: float, seed: int | None = None) -> tuple[str | None, str]:
|
|
84
|
+
"""Wrap ``query`` so it returns ~``fraction`` of its rows, deterministically.
|
|
85
|
+
|
|
86
|
+
Returns ``(setup_sql, sampled_sql)`` where ``setup_sql`` is an optional
|
|
87
|
+
per-connection statement to run first (PostgreSQL ``setseed``), or None.
|
|
88
|
+
|
|
89
|
+
Raises ``ValueError`` for a bad fraction and ``SamplingNotSupported`` for a
|
|
90
|
+
dialect with no sampler or a seed the dialect cannot honour.
|
|
91
|
+
"""
|
|
92
|
+
_check_fraction(fraction)
|
|
93
|
+
builder = DIALECT_SAMPLERS.get(dialect)
|
|
94
|
+
if builder is None:
|
|
95
|
+
raise SamplingNotSupported(f"no SQL sampler for dialect {dialect!r} (supported: {sorted(DIALECT_SAMPLERS)})")
|
|
96
|
+
return builder(query, fraction, seed)
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
def build_sample_batch_sql(dialect: str, query: str, n: int) -> str:
|
|
100
|
+
"""A fresh random batch of up to ``n`` rows (``ORDER BY <rand> LIMIT n``).
|
|
101
|
+
|
|
102
|
+
Deliberately unseeded: each call returns a different sample (with-replacement),
|
|
103
|
+
which is what the reduce-to-target loop wants.
|
|
104
|
+
"""
|
|
105
|
+
if n <= 0:
|
|
106
|
+
raise ValueError(f"batch size must be positive, got {n!r}")
|
|
107
|
+
rand = RANDOM_FN.get(dialect)
|
|
108
|
+
if rand is None:
|
|
109
|
+
raise SamplingNotSupported(f"no random function for dialect {dialect!r} (supported: {sorted(RANDOM_FN)})")
|
|
110
|
+
return f"SELECT * FROM ({query}) AS _lr_batch ORDER BY {rand} LIMIT {int(n)}"
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
def build_table_sample_query(
|
|
114
|
+
dialect: str,
|
|
115
|
+
table: str,
|
|
116
|
+
column: str,
|
|
117
|
+
*,
|
|
118
|
+
fraction: float,
|
|
119
|
+
seed: int | None = None,
|
|
120
|
+
method: str = "system",
|
|
121
|
+
where: str | None = None,
|
|
122
|
+
) -> str:
|
|
123
|
+
"""Build a native table-sampling query (``table``/``column`` already quoted).
|
|
124
|
+
|
|
125
|
+
PostgreSQL uses page-level ``TABLESAMPLE SYSTEM`` (or row-level ``BERNOULLI``)
|
|
126
|
+
with ``REPEATABLE(seed)`` - genuinely sub-linear, unlike the full-scan random
|
|
127
|
+
predicate used for arbitrary queries. Engines without TABLESAMPLE (MySQL,
|
|
128
|
+
SQLite) fall back to a predicate on the table (still a full scan).
|
|
129
|
+
|
|
130
|
+
Raises ``SamplingNotSupported`` for a dialect with no table sampler or a seed
|
|
131
|
+
it cannot honour, and ``ValueError`` for a bad fraction or method.
|
|
132
|
+
"""
|
|
133
|
+
_check_fraction(fraction)
|
|
134
|
+
conds = [where] if where else []
|
|
135
|
+
|
|
136
|
+
if dialect == "postgresql":
|
|
137
|
+
m = method.lower()
|
|
138
|
+
if m not in ("system", "bernoulli"):
|
|
139
|
+
raise ValueError(f"method must be 'system' or 'bernoulli', got {method!r}")
|
|
140
|
+
pct = float(fraction) * 100.0
|
|
141
|
+
repeatable = f" REPEATABLE ({int(seed)})" if seed is not None else ""
|
|
142
|
+
where_sql = f" WHERE {' AND '.join(conds)}" if conds else ""
|
|
143
|
+
return f"SELECT {column} FROM {table} TABLESAMPLE {m.upper()} ({pct!r}){repeatable}{where_sql}"
|
|
144
|
+
|
|
145
|
+
# No TABLESAMPLE on these - a predicate on the table (full scan, but native).
|
|
146
|
+
if dialect in ("mysql", "mariadb"):
|
|
147
|
+
pred = f"rand({int(seed)}) < {float(fraction)!r}" if seed is not None else f"rand() < {float(fraction)!r}"
|
|
148
|
+
elif dialect == "sqlite":
|
|
149
|
+
if seed is not None:
|
|
150
|
+
raise SamplingNotSupported("sqlite has no seedable RNG; seeded table sampling is unsupported")
|
|
151
|
+
pred = f"(abs(random()) % 1000000) < {int(float(fraction) * 1_000_000)}"
|
|
152
|
+
else:
|
|
153
|
+
raise SamplingNotSupported(f"no table sampler for dialect {dialect!r}")
|
|
154
|
+
|
|
155
|
+
conds.append(pred)
|
|
156
|
+
return f"SELECT {column} FROM {table} WHERE {' AND '.join(conds)}"
|
|
157
|
+
|
|
158
|
+
|
|
159
|
+
def reservoir_sample(items: Iterable[str], k: int, rng: random.Random) -> list[str]:
|
|
160
|
+
"""Uniform sample of ``k`` items from a stream of unknown length (Algorithm L).
|
|
161
|
+
|
|
162
|
+
Without replacement, O(k) memory, O(k(1 + log(n/k))) expected time. Replaces
|
|
163
|
+
the old ``hash(line) % (i+1)`` reservoir, which was neither uniform nor
|
|
164
|
+
reproducible (it depended on PYTHONHASHSEED). Seed ``rng`` for reproducibility.
|
|
165
|
+
"""
|
|
166
|
+
if k <= 0:
|
|
167
|
+
return []
|
|
168
|
+
it = iter(items)
|
|
169
|
+
reservoir = list(islice(it, k))
|
|
170
|
+
if len(reservoir) < k:
|
|
171
|
+
return reservoir # fewer than k items - keep them all
|
|
172
|
+
|
|
173
|
+
def _u() -> float:
|
|
174
|
+
# Clamp away from 0.0 and 1.0 so log() and log1p(-w) never blow up.
|
|
175
|
+
return min(max(rng.random(), 1e-12), 1.0 - 1e-12)
|
|
176
|
+
|
|
177
|
+
w = math.exp(math.log(_u()) / k)
|
|
178
|
+
while True:
|
|
179
|
+
skip = math.floor(math.log(_u()) / math.log1p(-w))
|
|
180
|
+
# islice(it, skip, skip + 1) discards `skip` items and yields the next.
|
|
181
|
+
nxt = next(islice(it, skip, skip + 1), None)
|
|
182
|
+
if nxt is None:
|
|
183
|
+
return reservoir
|
|
184
|
+
reservoir[rng.randrange(k)] = nxt
|
|
185
|
+
w *= math.exp(math.log(_u()) / k)
|
|
186
|
+
|
|
187
|
+
|
|
188
|
+
def estimate_batch_rows(
|
|
189
|
+
sample_lines: list[str],
|
|
190
|
+
budget_bytes: int,
|
|
191
|
+
*,
|
|
192
|
+
floor: int = 1000,
|
|
193
|
+
ceil: int = 1_000_000,
|
|
194
|
+
align: int = 1000,
|
|
195
|
+
overhead: int = 3,
|
|
196
|
+
) -> int:
|
|
197
|
+
"""Rows that fit ``budget_bytes``, from the average byte size of a sample.
|
|
198
|
+
|
|
199
|
+
``overhead`` approximates Python's per-object memory multiplier (matching
|
|
200
|
+
MemoryMonitor). Result is clamped to ``[floor, ceil]`` and rounded down to a
|
|
201
|
+
multiple of ``align`` so it lines up with the cursor fetch batch.
|
|
202
|
+
"""
|
|
203
|
+
if not sample_lines or budget_bytes <= 0:
|
|
204
|
+
return floor
|
|
205
|
+
avg_bytes = sum(len(s.encode("utf-8")) for s in sample_lines) / len(sample_lines)
|
|
206
|
+
per_row = max(1.0, avg_bytes * overhead)
|
|
207
|
+
rows = int(budget_bytes / per_row)
|
|
208
|
+
if align > 1:
|
|
209
|
+
rows = (rows // align) * align
|
|
210
|
+
# Clamp LAST so alignment can never push the result outside [floor, ceil].
|
|
211
|
+
return max(floor, min(ceil, rows))
|
logreducer/sinks.py
ADDED
|
@@ -0,0 +1,81 @@
|
|
|
1
|
+
"""Output sinks for logreducer.
|
|
2
|
+
|
|
3
|
+
The reduction engine returns its result in memory (a ``list[str]``) - that is
|
|
4
|
+
the primary, dependency-free output. A *sink* is the abstraction for sending
|
|
5
|
+
those reduced lines somewhere else: a file, a Kafka topic, a database. Like
|
|
6
|
+
``Source`` on the input side, a sink is a thin seam - the package does not own
|
|
7
|
+
the connection or the delivery guarantees; an application can pass its own.
|
|
8
|
+
|
|
9
|
+
``Sink`` is a structural protocol: anything with a ``write(lines) -> int`` is a
|
|
10
|
+
sink. ``FileSink`` is built in; the Kafka producer sink lives behind the
|
|
11
|
+
``kafka`` optional extra (see the kafka submodule).
|
|
12
|
+
"""
|
|
13
|
+
|
|
14
|
+
from __future__ import annotations
|
|
15
|
+
|
|
16
|
+
import json
|
|
17
|
+
import os
|
|
18
|
+
from collections.abc import Iterable
|
|
19
|
+
from datetime import datetime
|
|
20
|
+
from pathlib import Path
|
|
21
|
+
from typing import Protocol, runtime_checkable
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
@runtime_checkable
|
|
25
|
+
class Sink(Protocol):
|
|
26
|
+
"""A destination for reduced log lines.
|
|
27
|
+
|
|
28
|
+
``write`` consumes an iterable of ``str`` lines and returns the number of
|
|
29
|
+
lines written. It may be called with a generator, so implementations should
|
|
30
|
+
iterate lazily rather than materialising the whole batch.
|
|
31
|
+
"""
|
|
32
|
+
|
|
33
|
+
def write(self, lines: Iterable[str]) -> int: ...
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
class FileSink:
|
|
37
|
+
"""Write reduced lines to a text file in one of three formats.
|
|
38
|
+
|
|
39
|
+
A standalone, format-aware writer (``line`` / ``json`` / ``jsonl``) that an
|
|
40
|
+
application can use without touching the reducer. It streams the ``line``
|
|
41
|
+
and ``jsonl`` formats a row at a time (constant memory); ``json`` buffers
|
|
42
|
+
the list because the format is a single document.
|
|
43
|
+
|
|
44
|
+
This is deliberately simpler than the reducer's own ``output_file`` path,
|
|
45
|
+
which additionally writes a ``.meta.json`` sidecar of run stats - that needs
|
|
46
|
+
the reducer's context, whereas a sink only sees the lines.
|
|
47
|
+
"""
|
|
48
|
+
|
|
49
|
+
def __init__(self, path: str | os.PathLike[str], *, output_format: str = "line") -> None:
|
|
50
|
+
self.path = os.fspath(path)
|
|
51
|
+
fmt = output_format.lower()
|
|
52
|
+
if fmt not in ("line", "json", "jsonl"):
|
|
53
|
+
raise ValueError(f"Unknown output_format {output_format!r} (use line, json or jsonl)")
|
|
54
|
+
self.output_format = fmt
|
|
55
|
+
|
|
56
|
+
def write(self, lines: Iterable[str]) -> int:
|
|
57
|
+
output_path = Path(self.path)
|
|
58
|
+
output_path.parent.mkdir(parents=True, exist_ok=True)
|
|
59
|
+
|
|
60
|
+
count = 0
|
|
61
|
+
if self.output_format == "json":
|
|
62
|
+
# A JSON document is a single value - the list has to be built.
|
|
63
|
+
materialised = list(lines)
|
|
64
|
+
count = len(materialised)
|
|
65
|
+
with open(output_path, "w", encoding="utf-8") as f:
|
|
66
|
+
json.dump({"lines": materialised, "timestamp": datetime.now().isoformat()}, f, indent=2)
|
|
67
|
+
elif self.output_format == "jsonl":
|
|
68
|
+
with open(output_path, "w", encoding="utf-8") as f:
|
|
69
|
+
for line in lines:
|
|
70
|
+
json.dump({"line": line}, f)
|
|
71
|
+
f.write("\n")
|
|
72
|
+
count += 1
|
|
73
|
+
else: # line
|
|
74
|
+
with open(output_path, "w", encoding="utf-8") as f:
|
|
75
|
+
for line in lines:
|
|
76
|
+
f.write(line + "\n")
|
|
77
|
+
count += 1
|
|
78
|
+
return count
|
|
79
|
+
|
|
80
|
+
def __repr__(self) -> str:
|
|
81
|
+
return f"FileSink({self.path!r}, output_format={self.output_format!r})"
|