logreducer 3.4.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
logreducer/config.py ADDED
@@ -0,0 +1,166 @@
1
+ """Configuration and tuning parameters for LogReducer."""
2
+
3
+ import os
4
+ import types
5
+ import typing
6
+ from dataclasses import dataclass, fields
7
+ from enum import Enum
8
+
9
+ import psutil
10
+ from loguru import logger
11
+
12
+
13
+ class ProcessingLevel(Enum):
14
+ """Processing level determines speed/quality tradeoff."""
15
+
16
+ STANDARD = "standard" # Fast, 99% reduction
17
+ ENHANCED = "enhanced" # Balanced, 99.5% reduction
18
+ MAXIMUM = "maximum" # Thorough, 99.9% reduction
19
+
20
+
21
+ class ProcessingMode(Enum):
22
+ """Processing mode determines reduction strategy."""
23
+
24
+ PATTERN = "pattern" # Pattern-based reduction (Drain3)
25
+ ANOMALY = "anomaly" # Anomaly detection focus
26
+ TEMPORAL = "temporal" # Time-based sampling
27
+ HYBRID = "hybrid" # Combined approach
28
+
29
+
30
+ class OutputFormat(Enum):
31
+ """Output format for reduced logs."""
32
+
33
+ LINE = "line" # Line-by-line text output (default)
34
+ JSON = "json" # JSON structured output
35
+ JSONL = "jsonl" # JSON Lines format (one JSON per line)
36
+
37
+
38
+ @dataclass
39
+ class BigDialConfig:
40
+ """Big dial tuning parameters."""
41
+
42
+ # Memory Control. The engine streams, so this cap mainly sizes the file
43
+ # read strategy (full/chunked/sampled), the reservoir, and the watchdog -
44
+ # measured reductions use tens of MB, so the default is deliberately low
45
+ # and container-friendly.
46
+ max_memory_gb: float = 1.0
47
+ dedup_cache_size: int = 100000
48
+
49
+ # Speed Control
50
+ hash_algorithm: str = "xxhash"
51
+
52
+ # Quality Control
53
+ drain_similarity: float = 0.4
54
+ fuzzy_threshold: float | None = 0.8
55
+ min_pattern_occurrences: int = 2
56
+ anomaly_contamination: float = 0.1
57
+ # Bound the Drain3 template store (LRU-evict beyond this many templates).
58
+ # None = unbounded (default; the store grows with distinct templates).
59
+ max_clusters: int | None = None
60
+ # Cap the rows fed to anomaly detection (reservoir-sampled). Bounds the
61
+ # TF-IDF matrix on a huge unique-line set, at the cost of anomaly recall
62
+ # (rare lines may be sampled out). None = no cap (use every unique line).
63
+ anomaly_max_rows: int | None = None
64
+
65
+ # Temporal Control
66
+ temporal_window_minutes: int = 60
67
+
68
+ # Sampling Control
69
+ max_patterns: int = 1000
70
+ examples_per_pattern: int = 3
71
+
72
+ # Logging Control
73
+ enable_logging: bool = False # Logging disabled by default
74
+ log_file: str | None = None # Path to log file (None = no file logging)
75
+ log_level: str = "INFO" # DEBUG, INFO, WARNING, ERROR
76
+ log_format: str = "rfc3339" # rfc3339 or simple
77
+
78
+ # Output Control
79
+ output_format: OutputFormat = OutputFormat.LINE # Default line-by-line
80
+ pretty_json: bool = False # Pretty print JSON output
81
+
82
+ def __post_init__(self) -> None:
83
+ # Never promise more memory than the host can give: clamp to 70% of
84
+ # what is currently available, and say so rather than silently mutate.
85
+ available_gb = psutil.virtual_memory().available / (1024**3)
86
+ if self.max_memory_gb > available_gb * 0.7:
87
+ clamped = available_gb * 0.7
88
+ logger.warning(
89
+ f"max_memory_gb={self.max_memory_gb:.1f} exceeds 70% of available RAM; clamped to {clamped:.1f} GB"
90
+ )
91
+ self.max_memory_gb = clamped
92
+
93
+ @classmethod
94
+ def from_env(cls, *prefixes: str) -> "BigDialConfig":
95
+ """Build a config from environment variables, cascade-aware.
96
+
97
+ ``from_env()`` reads ``LOGREDUCER_<FIELD>`` (upper-cased field names,
98
+ e.g. ``LOGREDUCER_MAX_MEMORY_GB=0.5``). Pass explicit prefixes to
99
+ cascade: ``from_env("DFE", "LOGREDUCER")`` reads ``DFE_<FIELD>`` first,
100
+ then falls back to ``LOGREDUCER_<FIELD>`` - the same
101
+ prefixed-overrides-bare convention as the host-app config cascades this
102
+ is designed to slot under. Unset fields keep their dataclass defaults,
103
+ so a host can drive only the knobs it cares about.
104
+ """
105
+ if not prefixes:
106
+ prefixes = ("LOGREDUCER",)
107
+ overrides: dict[str, object] = {}
108
+ for field in fields(cls):
109
+ for prefix in prefixes:
110
+ raw = os.environ.get(f"{prefix.rstrip('_')}_{field.name.upper()}")
111
+ if raw is not None:
112
+ overrides[field.name] = _coerce_env_value(raw, field.name)
113
+ break
114
+ return cls(**overrides) # type: ignore[arg-type]
115
+
116
+
117
+ def _coerce_env_value(raw: str, field_name: str) -> object:
118
+ """Coerce an env string to the annotated type of a BigDialConfig field."""
119
+ hints = typing.get_type_hints(BigDialConfig)
120
+ target = hints[field_name]
121
+ # Unwrap Optional[X]: 'none'/'null'/'' mean None, otherwise coerce to X.
122
+ if isinstance(target, types.UnionType):
123
+ args = [a for a in typing.get_args(target) if a is not type(None)]
124
+ if raw.strip().lower() in ("", "none", "null"):
125
+ return None
126
+ target = args[0]
127
+ if target is bool:
128
+ return raw.strip().lower() in ("1", "true", "yes", "on")
129
+ if target is int:
130
+ return int(raw)
131
+ if target is float:
132
+ return float(raw)
133
+ if target is OutputFormat:
134
+ return OutputFormat(raw.strip().lower())
135
+ return raw
136
+
137
+
138
+ def get_preset_config(level: ProcessingLevel) -> BigDialConfig:
139
+ """Get preset configuration for processing level."""
140
+ if level == ProcessingLevel.STANDARD:
141
+ return BigDialConfig(
142
+ max_memory_gb=0.5,
143
+ dedup_cache_size=50000,
144
+ drain_similarity=0.5,
145
+ fuzzy_threshold=None, # Disabled for speed
146
+ max_patterns=500,
147
+ examples_per_pattern=2,
148
+ )
149
+ elif level == ProcessingLevel.ENHANCED:
150
+ return BigDialConfig(
151
+ max_memory_gb=1.0,
152
+ dedup_cache_size=100000,
153
+ drain_similarity=0.4,
154
+ fuzzy_threshold=0.8,
155
+ max_patterns=1000,
156
+ examples_per_pattern=3,
157
+ )
158
+ else: # MAXIMUM
159
+ return BigDialConfig(
160
+ max_memory_gb=2.0,
161
+ dedup_cache_size=200000,
162
+ drain_similarity=0.3,
163
+ fuzzy_threshold=0.9,
164
+ max_patterns=2000,
165
+ examples_per_pattern=5,
166
+ )
logreducer/core.py ADDED
@@ -0,0 +1,493 @@
1
+ """
2
+ Core LogReducer implementation
3
+ """
4
+
5
+ import json
6
+ import os
7
+ import random
8
+ import time
9
+ from dataclasses import replace
10
+ from datetime import datetime
11
+ from enum import Enum
12
+ from pathlib import Path
13
+ from typing import Any
14
+
15
+ from .anomaly import AnomalyDetector
16
+ from .config import BigDialConfig, OutputFormat, ProcessingLevel, ProcessingMode, get_preset_config
17
+ from .logging_config import get_logger, setup_logging
18
+ from .memory import BoundedDeduplicator, MemoryMonitor
19
+ from .patterns import FuzzyDeduplicator, PatternExtractor
20
+ from .sampling import reservoir_sample
21
+ from .sinks import Sink
22
+ from .sources import FileSource, Source
23
+ from .temporal import TemporalProcessor
24
+
25
+
26
+ class LogReducer:
27
+ """
28
+ Main log reduction class
29
+
30
+ Example:
31
+ reducer = LogReducer(level="enhanced")
32
+ reduced = reducer.process_file("app.log")
33
+ """
34
+
35
+ def __init__(
36
+ self,
37
+ level: str | ProcessingLevel = "standard",
38
+ mode: str | ProcessingMode = "pattern",
39
+ max_memory_gb: float | None = None,
40
+ max_patterns: int | None = None,
41
+ config: BigDialConfig | None = None,
42
+ **kwargs: Any,
43
+ ) -> None:
44
+ """
45
+ Initialize LogReducer
46
+
47
+ Args:
48
+ level: Processing level (standard/enhanced/maximum)
49
+ mode: Processing mode (pattern/anomaly/temporal/hybrid)
50
+ max_memory_gb: Memory limit override
51
+ max_patterns: Maximum patterns override
52
+ config: A fully-built BigDialConfig to use INSTEAD of the level
53
+ preset - the injection seam for a host application that owns
54
+ its own config cascade (build the config however you like and
55
+ hand it over; kwargs still apply on top). Copied, so the
56
+ caller's object is never mutated. ``level`` then only selects
57
+ the fuzzy-dedup gate, not preset values.
58
+ **kwargs: Additional config overrides
59
+ """
60
+ # Parse level and mode
61
+ if isinstance(level, str):
62
+ level = ProcessingLevel(level.lower())
63
+ if isinstance(mode, str):
64
+ mode = ProcessingMode(mode.lower())
65
+
66
+ self.level = level
67
+ self.mode = mode
68
+
69
+ # Base config: an injected one wins over the level preset.
70
+ self.config = replace(config) if config is not None else get_preset_config(level)
71
+
72
+ # Apply overrides BEFORE setting up logging
73
+ if max_memory_gb:
74
+ self.config.max_memory_gb = max_memory_gb
75
+ if max_patterns:
76
+ self.config.max_patterns = max_patterns
77
+
78
+ for key, value in kwargs.items():
79
+ # Fail fast on typos: silently dropping an unknown override means a
80
+ # user's tuning quietly does nothing.
81
+ if not hasattr(self.config, key):
82
+ raise ValueError(f"Unknown config option {key!r} (see BigDialConfig for valid fields)")
83
+ # Special handling for output_format to convert string to enum
84
+ if key == "output_format" and isinstance(value, str):
85
+ value = OutputFormat(value.lower())
86
+ setattr(self.config, key, value)
87
+
88
+ # Setup logging based on config (after overrides applied)
89
+ setup_logging(
90
+ enable=self.config.enable_logging,
91
+ log_file=self.config.log_file,
92
+ log_level=self.config.log_level,
93
+ log_format=self.config.log_format,
94
+ )
95
+
96
+ # Get logger for this module
97
+ self.logger = get_logger("logreducer.core")
98
+
99
+ # Initialize components
100
+ self.memory_monitor = MemoryMonitor(self.config.max_memory_gb)
101
+
102
+ if mode in [ProcessingMode.ANOMALY, ProcessingMode.HYBRID]:
103
+ self.anomaly_detector: AnomalyDetector | None = AnomalyDetector(self.config.anomaly_contamination)
104
+ else:
105
+ self.anomaly_detector = None
106
+
107
+ # Stateful analysis components (dedup seen-set, Drain3 miners, fuzzy
108
+ # LSH) are (re)created per run by _reset_components(), so a reused
109
+ # reducer never carries one run's accumulated state into the next.
110
+ self.fuzzy_dedup: FuzzyDeduplicator | None = None
111
+ self.temporal_processor: TemporalProcessor | None = None
112
+ self._reset_components()
113
+
114
+ self.stats: dict[str, Any] = {}
115
+
116
+ def _reset_components(self) -> None:
117
+ """Recreate the stateful analysis components for a fresh, isolated run.
118
+
119
+ The deduplicator (seen-set), Drain3 miners (pattern AND per-window
120
+ temporal), and fuzzy-dedup LSH all accumulate per-line state.
121
+ Recreating them keeps a LogReducer instance reusable across
122
+ reduce()/process_file() calls, and lets a single run make independent
123
+ deduplication passes (hybrid mode) without the first pass poisoning
124
+ the second.
125
+ """
126
+ self.deduplicator = BoundedDeduplicator(self.config.dedup_cache_size, self.config.hash_algorithm)
127
+ self.pattern_extractor = PatternExtractor(self.config)
128
+ if self.config.fuzzy_threshold and self.level != ProcessingLevel.STANDARD:
129
+ self.fuzzy_dedup = FuzzyDeduplicator(self.config.fuzzy_threshold)
130
+ else:
131
+ self.fuzzy_dedup = None
132
+ if self.mode in [ProcessingMode.TEMPORAL, ProcessingMode.HYBRID]:
133
+ self.temporal_processor = TemporalProcessor(self.config.temporal_window_minutes)
134
+ else:
135
+ self.temporal_processor = None
136
+
137
+ def reduce(
138
+ self,
139
+ source: Source,
140
+ output_file: str | None = None,
141
+ return_metadata: bool = False,
142
+ sink: Sink | None = None,
143
+ ) -> list[str] | dict:
144
+ """Reduce a source of log lines to a representative sample.
145
+
146
+ This is the core entry point: it operates on an abstraction, not on IO.
147
+ The source is any re-iterable stream of str lines - a list, a
148
+ FileSource, or an app-provided iterable wrapping its own DB cursor or
149
+ Kafka consumer. The reducer never manages the connection or loading
150
+ path.
151
+
152
+ Args:
153
+ source: Re-iterable stream of str lines (see logreducer.sources).
154
+ Multi-pass modes (hybrid) require the source to be re-iterable.
155
+ output_file: Optional path to also write the result to, with a
156
+ format-aware ``.meta.json`` sidecar of run stats (CLI use).
157
+ return_metadata: Return a dict with lines + stats + config instead
158
+ of just the lines.
159
+ sink: Optional output abstraction (see logreducer.sinks). The
160
+ reduced lines are also handed to ``sink.write`` - a FileSink, a
161
+ KafkaSink, or any app-provided destination.
162
+
163
+ Returns:
164
+ The reduced lines in memory (list[str]), or a metadata dict.
165
+ """
166
+ # A Source must be re-iterable: reduce() counts lines in one pass, then
167
+ # re-reads the source to process it (hybrid re-reads again). A one-shot
168
+ # iterator (a bare generator) would be drained by the count and leave
169
+ # nothing to process - fail loudly rather than return an empty result.
170
+ if iter(source) is source:
171
+ raise TypeError(
172
+ "source must be re-iterable (a fresh iterator on each pass); got a "
173
+ "one-shot iterator/generator - wrap it, e.g. list(source)."
174
+ )
175
+
176
+ # Fresh analysis state per run: a reused reducer must not carry the
177
+ # previous run's dedup/miner/LSH state, which would drop every line as
178
+ # already-seen and silently return an empty result.
179
+ self._reset_components()
180
+
181
+ start_time = time.time()
182
+
183
+ # Count input lines for the reduction ratio (one pass over the source).
184
+ # Note: a size-sampled FileSource yields its sampled line count, so for
185
+ # very large files the ratio and input_lines stat are approximate.
186
+ input_lines = sum(1 for _ in source)
187
+
188
+ size_bytes = getattr(source, "size_bytes", None)
189
+ file_size_mb = size_bytes / (1024 * 1024) if size_bytes else None
190
+
191
+ if self.config.enable_logging:
192
+ where = f"{file_size_mb:.1f} MB" if file_size_mb is not None else f"{input_lines:,} lines"
193
+ self.logger.info(f"Reducing {source!r} ({where})")
194
+ self.logger.info(f"Mode: {self.mode.value}, Level: {self.level.value}")
195
+ self.logger.info(f"Memory limit: {self.config.max_memory_gb:.1f} GB")
196
+
197
+ # Process based on mode
198
+ if self.mode == ProcessingMode.PATTERN:
199
+ result_lines = self._process_pattern_mode(source)
200
+ elif self.mode == ProcessingMode.ANOMALY:
201
+ result_lines = self._process_anomaly_mode(source)
202
+ elif self.mode == ProcessingMode.TEMPORAL:
203
+ result_lines = self._process_temporal_mode(source)
204
+ else: # HYBRID
205
+ result_lines = self._process_hybrid_mode(source)
206
+
207
+ # Calculate stats
208
+ processing_time = time.time() - start_time
209
+ output_lines = len(result_lines)
210
+ reduction_percent = (1 - output_lines / max(input_lines, 1)) * 100 if input_lines > 0 else 0
211
+ input_label = getattr(source, "path", None)
212
+
213
+ self.stats = {
214
+ "input_file": str(input_label) if input_label is not None else repr(source),
215
+ "input_lines": input_lines,
216
+ "input_size_mb": file_size_mb,
217
+ "output_lines": output_lines,
218
+ "reduction_percent": reduction_percent,
219
+ "processing_time_seconds": processing_time,
220
+ "processing_rate_mb_per_sec": (
221
+ file_size_mb / max(processing_time, 0.001) if file_size_mb is not None else None
222
+ ),
223
+ "mode": self.mode.value,
224
+ "level": self.level.value,
225
+ "memory_limit_gb": self.config.max_memory_gb,
226
+ }
227
+
228
+ if output_file:
229
+ self._save_output(result_lines, output_file)
230
+
231
+ if sink is not None:
232
+ sink.write(result_lines)
233
+
234
+ self._print_summary()
235
+
236
+ if return_metadata:
237
+ return {
238
+ "lines": result_lines,
239
+ "stats": self.stats,
240
+ "config": self._config_as_dict(),
241
+ }
242
+ return result_lines
243
+
244
+ def process_file(
245
+ self,
246
+ input_file: str,
247
+ output_file: str | None = None,
248
+ return_metadata: bool = False,
249
+ ) -> list[str] | dict:
250
+ """Reduce a log file - convenience wrapper over reduce() + FileSource.
251
+
252
+ Args:
253
+ input_file: Path to the input log file.
254
+ output_file: Optional output file path.
255
+ return_metadata: Return a metadata dict instead of just the lines.
256
+
257
+ Returns:
258
+ The reduced lines (list[str]) or a metadata dict.
259
+ """
260
+ if not os.path.exists(input_file):
261
+ raise FileNotFoundError(f"File not found: {input_file}")
262
+ source = FileSource(input_file, max_memory_gb=self.config.max_memory_gb)
263
+ return self.reduce(source, output_file=output_file, return_metadata=return_metadata)
264
+
265
+ def _process_pattern_mode(self, source: Source) -> list[str]:
266
+ """Process using pattern extraction, streaming end to end.
267
+
268
+ Exact dedup -> optional fuzzy dedup -> Drain3 all run as generators, so
269
+ the unique lines are never collected into a list. Peak memory is the
270
+ bounded dedup cache plus the Drain3 template store (see max_clusters),
271
+ independent of how many unique lines the source has.
272
+ """
273
+ if self.config.enable_logging:
274
+ # One message, not phases: dedup and mining run interleaved as a
275
+ # single generator pipeline, so there is no phase boundary to report.
276
+ self.logger.info("Streaming pipeline: dedup -> fuzzy dedup -> pattern mining")
277
+
278
+ unique_stream = self.deduplicator.deduplicate_lines(source)
279
+ if self.fuzzy_dedup:
280
+ unique_stream = self.fuzzy_dedup.deduplicate_stream(unique_stream)
281
+ patterns = self.pattern_extractor.extract_patterns(unique_stream)
282
+
283
+ # Collect examples
284
+ result = []
285
+ for pattern in patterns[: self.config.max_patterns]:
286
+ result.extend(pattern.examples)
287
+
288
+ return result
289
+
290
+ def _cap_for_anomaly(self, unique_lines: list[str]) -> list[str]:
291
+ """Reservoir-cap unique lines to anomaly_max_rows to bound the ML matrix.
292
+
293
+ Anomaly detection (Isolation Forest over a TF-IDF matrix) is batch ML and
294
+ cannot stream, so a huge unique-line set is the one place a hard memory
295
+ cap needs an explicit sample. Fixed seed = reproducible across passes.
296
+ Trades anomaly recall (a rare line may be sampled out) for bounded memory;
297
+ off unless anomaly_max_rows is set.
298
+ """
299
+ cap = self.config.anomaly_max_rows
300
+ if cap and len(unique_lines) > cap:
301
+ return reservoir_sample(unique_lines, cap, random.Random(0))
302
+ return unique_lines
303
+
304
+ def _process_anomaly_mode(self, source: Source) -> list[str]:
305
+ """Process using anomaly detection"""
306
+ if not self.anomaly_detector:
307
+ if self.config.enable_logging:
308
+ self.logger.warning("Anomaly detector not available, falling back to pattern mode")
309
+ return self._process_pattern_mode(source)
310
+
311
+ if self.config.enable_logging:
312
+ self.logger.info("Phase 1/2: Reading and deduplicating")
313
+ unique_lines = self._cap_for_anomaly(list(self.deduplicator.deduplicate_lines(source)))
314
+
315
+ if self.config.enable_logging:
316
+ self.logger.info("Phase 2/2: Anomaly detection")
317
+ anomalous, normal = self.anomaly_detector.detect_anomalies(unique_lines)
318
+
319
+ # Keep all anomalies + sample of normal
320
+ result = anomalous[: self.config.max_patterns // 2]
321
+
322
+ # Add some normal lines for context. Seeded RNG so a re-run over the
323
+ # same input yields the same sample (consistent with the reproducibility
324
+ # guarantees elsewhere); there is no security requirement here.
325
+ normal_sample_size = min(len(normal), self.config.max_patterns // 4)
326
+ if normal_sample_size > 0:
327
+ result.extend(random.Random(0).sample(normal, normal_sample_size))
328
+
329
+ return result
330
+
331
+ def _process_temporal_mode(self, source: Source) -> list[str]:
332
+ """Process using temporal analysis"""
333
+ if not self.temporal_processor:
334
+ if self.config.enable_logging:
335
+ self.logger.warning("Temporal processor not available, falling back to pattern mode")
336
+ return self._process_pattern_mode(source)
337
+
338
+ if self.config.enable_logging:
339
+ self.logger.info("Phase 1/2: Reading lines")
340
+ lines = list(source)
341
+
342
+ if self.config.enable_logging:
343
+ self.logger.info("Phase 2/2: Temporal processing")
344
+ temporal_results = self.temporal_processor.process_temporal(lines)
345
+
346
+ # Collect examples from temporal patterns
347
+ result = []
348
+ for pattern in temporal_results.get("temporal_patterns", [])[: self.config.max_patterns]:
349
+ if "example" in pattern:
350
+ result.append(pattern["example"])
351
+
352
+ for pattern in temporal_results.get("timeless_patterns", [])[:100]:
353
+ if "example" in pattern:
354
+ result.append(pattern["example"])
355
+
356
+ return result
357
+
358
+ def _process_hybrid_mode(self, source: Source) -> list[str]:
359
+ """Process using combined approach"""
360
+ if self.config.enable_logging:
361
+ self.logger.info("Hybrid mode: combining pattern and anomaly detection")
362
+
363
+ # Get patterns
364
+ pattern_lines = self._process_pattern_mode(source)
365
+
366
+ # Get anomalies if available
367
+ if self.anomaly_detector:
368
+ # The pattern pass above consumed self.deduplicator's seen-set, so the
369
+ # anomaly pass needs a fresh deduplicator - otherwise every line reads
370
+ # as already-seen and no anomalies survive (hybrid -> pattern-only).
371
+ self.deduplicator = BoundedDeduplicator(self.config.dedup_cache_size, self.config.hash_algorithm)
372
+ unique_lines = self._cap_for_anomaly(list(self.deduplicator.deduplicate_lines(source)))
373
+ anomalous, _ = self.anomaly_detector.detect_anomalies(unique_lines)
374
+
375
+ # Combine, preferring anomalies
376
+ result = anomalous[: self.config.max_patterns // 2]
377
+ result.extend(pattern_lines[: self.config.max_patterns // 2])
378
+ else:
379
+ result = pattern_lines
380
+
381
+ # Remove duplicates while preserving order
382
+ seen = set()
383
+ final = []
384
+ for line in result:
385
+ if line not in seen:
386
+ seen.add(line)
387
+ final.append(line)
388
+
389
+ return final[: self.config.max_patterns]
390
+
391
+ def _config_as_dict(self) -> dict[str, Any]:
392
+ """Config as a JSON-serialisable dict.
393
+
394
+ Enum members (output_format, ...) become their ``.value`` string so the
395
+ result survives ``json.dumps`` and never leaks ``OutputFormat.LINE``-style
396
+ reprs into metadata output. Private (``_``-prefixed) attrs are dropped.
397
+ """
398
+ return {
399
+ k: (v.value if isinstance(v, Enum) else v) for k, v in vars(self.config).items() if not k.startswith("_")
400
+ }
401
+
402
+ def _save_output(self, lines: list[str], output_file: str) -> None:
403
+ """Save output to file in specified format"""
404
+ output_path = Path(output_file)
405
+ output_path.parent.mkdir(parents=True, exist_ok=True)
406
+
407
+ # Save in requested format
408
+ if self.config.output_format == OutputFormat.JSON:
409
+ # Full JSON format with metadata
410
+ output_data = {
411
+ "lines": lines,
412
+ "stats": self.stats,
413
+ "config": self._config_as_dict(),
414
+ "timestamp": datetime.now().isoformat(),
415
+ }
416
+ with open(output_path, "w", encoding="utf-8") as f:
417
+ json.dump(output_data, f, indent=2 if self.config.pretty_json else None)
418
+
419
+ if self.config.enable_logging:
420
+ self.logger.info(f"Output saved to {output_file}")
421
+
422
+ elif self.config.output_format == OutputFormat.JSONL:
423
+ # JSON Lines format - one JSON object per line
424
+ with open(output_path, "w", encoding="utf-8") as f:
425
+ for line in lines:
426
+ json.dump({"line": line, "timestamp": datetime.now().isoformat()}, f)
427
+ f.write("\n")
428
+
429
+ if self.config.enable_logging:
430
+ self.logger.info(f"Output saved to {output_file}")
431
+
432
+ else: # OutputFormat.LINE (default)
433
+ # Traditional line-by-line text output
434
+ with open(output_path, "w", encoding="utf-8") as f:
435
+ for line in lines:
436
+ f.write(line + "\n")
437
+
438
+ # Save metadata separately for line format
439
+ meta_file = output_path.with_suffix(".meta.json")
440
+ with open(meta_file, "w", encoding="utf-8") as f:
441
+ json.dump(
442
+ {
443
+ "stats": self.stats,
444
+ "config": self._config_as_dict(),
445
+ "timestamp": datetime.now().isoformat(),
446
+ },
447
+ f,
448
+ indent=2,
449
+ )
450
+
451
+ if self.config.enable_logging:
452
+ self.logger.info(f"Output saved to {output_file}")
453
+ self.logger.info(f"Metadata saved to {meta_file}")
454
+
455
+ def _print_summary(self) -> None:
456
+ """Log processing summary"""
457
+ self.logger.info("\n" + "=" * 60)
458
+ self.logger.info("LOG REDUCTION SUMMARY")
459
+ self.logger.info("=" * 60)
460
+ self.logger.info(f"Mode: {self.mode.value}")
461
+ self.logger.info(f"Level: {self.level.value}")
462
+ # input_size_mb / rate are None for non-file sources (a list, a DB
463
+ # cursor) - only a file has a byte size. Log them when known.
464
+ input_size_mb = self.stats.get("input_size_mb")
465
+ if input_size_mb is not None:
466
+ self.logger.info(f"Input: {input_size_mb:.1f} MB")
467
+ self.logger.info(f"Input lines: {self.stats['input_lines']:,}")
468
+ self.logger.info(f"Output: {self.stats['output_lines']} lines")
469
+ self.logger.info(f"Reduction: {self.stats['reduction_percent']:.1f}%")
470
+ self.logger.info(f"Time: {self.stats['processing_time_seconds']:.1f} seconds")
471
+ rate = self.stats.get("processing_rate_mb_per_sec")
472
+ if rate is not None:
473
+ self.logger.info(f"Rate: {rate:.1f} MB/sec")
474
+ self.logger.info("=" * 60)
475
+
476
+ def estimate_processing(self, file_path: str) -> dict:
477
+ """Estimate processing requirements before running"""
478
+ file_size = os.path.getsize(file_path)
479
+ file_size_gb = file_size / (1024**3)
480
+
481
+ strategy = self.memory_monitor.estimate_file_strategy(file_size)
482
+
483
+ return {
484
+ "file_size_gb": file_size_gb,
485
+ "memory_required_gb": min(file_size_gb * 0.3, self.config.max_memory_gb),
486
+ "strategy": strategy,
487
+ "estimated_time_seconds": file_size_gb * 30,
488
+ "will_sample": strategy == "sampled",
489
+ "estimated_output_lines": min(
490
+ int(file_size_gb * 1000),
491
+ self.config.max_patterns * self.config.examples_per_pattern,
492
+ ),
493
+ }