logreducer 3.4.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- logreducer/__init__.py +40 -0
- logreducer/anomaly.py +75 -0
- logreducer/cli.py +359 -0
- logreducer/clickhouse.py +171 -0
- logreducer/config.py +166 -0
- logreducer/core.py +493 -0
- logreducer/kafka.py +300 -0
- logreducer/logging_config.py +193 -0
- logreducer/memory.py +247 -0
- logreducer/patterns.py +154 -0
- logreducer/py.typed +1 -0
- logreducer/sampling.py +211 -0
- logreducer/sinks.py +81 -0
- logreducer/sources.py +78 -0
- logreducer/sql.py +209 -0
- logreducer/target.py +164 -0
- logreducer/temporal.py +163 -0
- logreducer-3.4.0.dist-info/METADATA +387 -0
- logreducer-3.4.0.dist-info/RECORD +23 -0
- logreducer-3.4.0.dist-info/WHEEL +4 -0
- logreducer-3.4.0.dist-info/entry_points.txt +3 -0
- logreducer-3.4.0.dist-info/licenses/LICENSE +201 -0
- logreducer-3.4.0.dist-info/licenses/NOTICE +53 -0
logreducer/config.py
ADDED
|
@@ -0,0 +1,166 @@
|
|
|
1
|
+
"""Configuration and tuning parameters for LogReducer."""
|
|
2
|
+
|
|
3
|
+
import os
|
|
4
|
+
import types
|
|
5
|
+
import typing
|
|
6
|
+
from dataclasses import dataclass, fields
|
|
7
|
+
from enum import Enum
|
|
8
|
+
|
|
9
|
+
import psutil
|
|
10
|
+
from loguru import logger
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
class ProcessingLevel(Enum):
|
|
14
|
+
"""Processing level determines speed/quality tradeoff."""
|
|
15
|
+
|
|
16
|
+
STANDARD = "standard" # Fast, 99% reduction
|
|
17
|
+
ENHANCED = "enhanced" # Balanced, 99.5% reduction
|
|
18
|
+
MAXIMUM = "maximum" # Thorough, 99.9% reduction
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
class ProcessingMode(Enum):
|
|
22
|
+
"""Processing mode determines reduction strategy."""
|
|
23
|
+
|
|
24
|
+
PATTERN = "pattern" # Pattern-based reduction (Drain3)
|
|
25
|
+
ANOMALY = "anomaly" # Anomaly detection focus
|
|
26
|
+
TEMPORAL = "temporal" # Time-based sampling
|
|
27
|
+
HYBRID = "hybrid" # Combined approach
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
class OutputFormat(Enum):
|
|
31
|
+
"""Output format for reduced logs."""
|
|
32
|
+
|
|
33
|
+
LINE = "line" # Line-by-line text output (default)
|
|
34
|
+
JSON = "json" # JSON structured output
|
|
35
|
+
JSONL = "jsonl" # JSON Lines format (one JSON per line)
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
@dataclass
|
|
39
|
+
class BigDialConfig:
|
|
40
|
+
"""Big dial tuning parameters."""
|
|
41
|
+
|
|
42
|
+
# Memory Control. The engine streams, so this cap mainly sizes the file
|
|
43
|
+
# read strategy (full/chunked/sampled), the reservoir, and the watchdog -
|
|
44
|
+
# measured reductions use tens of MB, so the default is deliberately low
|
|
45
|
+
# and container-friendly.
|
|
46
|
+
max_memory_gb: float = 1.0
|
|
47
|
+
dedup_cache_size: int = 100000
|
|
48
|
+
|
|
49
|
+
# Speed Control
|
|
50
|
+
hash_algorithm: str = "xxhash"
|
|
51
|
+
|
|
52
|
+
# Quality Control
|
|
53
|
+
drain_similarity: float = 0.4
|
|
54
|
+
fuzzy_threshold: float | None = 0.8
|
|
55
|
+
min_pattern_occurrences: int = 2
|
|
56
|
+
anomaly_contamination: float = 0.1
|
|
57
|
+
# Bound the Drain3 template store (LRU-evict beyond this many templates).
|
|
58
|
+
# None = unbounded (default; the store grows with distinct templates).
|
|
59
|
+
max_clusters: int | None = None
|
|
60
|
+
# Cap the rows fed to anomaly detection (reservoir-sampled). Bounds the
|
|
61
|
+
# TF-IDF matrix on a huge unique-line set, at the cost of anomaly recall
|
|
62
|
+
# (rare lines may be sampled out). None = no cap (use every unique line).
|
|
63
|
+
anomaly_max_rows: int | None = None
|
|
64
|
+
|
|
65
|
+
# Temporal Control
|
|
66
|
+
temporal_window_minutes: int = 60
|
|
67
|
+
|
|
68
|
+
# Sampling Control
|
|
69
|
+
max_patterns: int = 1000
|
|
70
|
+
examples_per_pattern: int = 3
|
|
71
|
+
|
|
72
|
+
# Logging Control
|
|
73
|
+
enable_logging: bool = False # Logging disabled by default
|
|
74
|
+
log_file: str | None = None # Path to log file (None = no file logging)
|
|
75
|
+
log_level: str = "INFO" # DEBUG, INFO, WARNING, ERROR
|
|
76
|
+
log_format: str = "rfc3339" # rfc3339 or simple
|
|
77
|
+
|
|
78
|
+
# Output Control
|
|
79
|
+
output_format: OutputFormat = OutputFormat.LINE # Default line-by-line
|
|
80
|
+
pretty_json: bool = False # Pretty print JSON output
|
|
81
|
+
|
|
82
|
+
def __post_init__(self) -> None:
|
|
83
|
+
# Never promise more memory than the host can give: clamp to 70% of
|
|
84
|
+
# what is currently available, and say so rather than silently mutate.
|
|
85
|
+
available_gb = psutil.virtual_memory().available / (1024**3)
|
|
86
|
+
if self.max_memory_gb > available_gb * 0.7:
|
|
87
|
+
clamped = available_gb * 0.7
|
|
88
|
+
logger.warning(
|
|
89
|
+
f"max_memory_gb={self.max_memory_gb:.1f} exceeds 70% of available RAM; clamped to {clamped:.1f} GB"
|
|
90
|
+
)
|
|
91
|
+
self.max_memory_gb = clamped
|
|
92
|
+
|
|
93
|
+
@classmethod
|
|
94
|
+
def from_env(cls, *prefixes: str) -> "BigDialConfig":
|
|
95
|
+
"""Build a config from environment variables, cascade-aware.
|
|
96
|
+
|
|
97
|
+
``from_env()`` reads ``LOGREDUCER_<FIELD>`` (upper-cased field names,
|
|
98
|
+
e.g. ``LOGREDUCER_MAX_MEMORY_GB=0.5``). Pass explicit prefixes to
|
|
99
|
+
cascade: ``from_env("DFE", "LOGREDUCER")`` reads ``DFE_<FIELD>`` first,
|
|
100
|
+
then falls back to ``LOGREDUCER_<FIELD>`` - the same
|
|
101
|
+
prefixed-overrides-bare convention as the host-app config cascades this
|
|
102
|
+
is designed to slot under. Unset fields keep their dataclass defaults,
|
|
103
|
+
so a host can drive only the knobs it cares about.
|
|
104
|
+
"""
|
|
105
|
+
if not prefixes:
|
|
106
|
+
prefixes = ("LOGREDUCER",)
|
|
107
|
+
overrides: dict[str, object] = {}
|
|
108
|
+
for field in fields(cls):
|
|
109
|
+
for prefix in prefixes:
|
|
110
|
+
raw = os.environ.get(f"{prefix.rstrip('_')}_{field.name.upper()}")
|
|
111
|
+
if raw is not None:
|
|
112
|
+
overrides[field.name] = _coerce_env_value(raw, field.name)
|
|
113
|
+
break
|
|
114
|
+
return cls(**overrides) # type: ignore[arg-type]
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
def _coerce_env_value(raw: str, field_name: str) -> object:
|
|
118
|
+
"""Coerce an env string to the annotated type of a BigDialConfig field."""
|
|
119
|
+
hints = typing.get_type_hints(BigDialConfig)
|
|
120
|
+
target = hints[field_name]
|
|
121
|
+
# Unwrap Optional[X]: 'none'/'null'/'' mean None, otherwise coerce to X.
|
|
122
|
+
if isinstance(target, types.UnionType):
|
|
123
|
+
args = [a for a in typing.get_args(target) if a is not type(None)]
|
|
124
|
+
if raw.strip().lower() in ("", "none", "null"):
|
|
125
|
+
return None
|
|
126
|
+
target = args[0]
|
|
127
|
+
if target is bool:
|
|
128
|
+
return raw.strip().lower() in ("1", "true", "yes", "on")
|
|
129
|
+
if target is int:
|
|
130
|
+
return int(raw)
|
|
131
|
+
if target is float:
|
|
132
|
+
return float(raw)
|
|
133
|
+
if target is OutputFormat:
|
|
134
|
+
return OutputFormat(raw.strip().lower())
|
|
135
|
+
return raw
|
|
136
|
+
|
|
137
|
+
|
|
138
|
+
def get_preset_config(level: ProcessingLevel) -> BigDialConfig:
|
|
139
|
+
"""Get preset configuration for processing level."""
|
|
140
|
+
if level == ProcessingLevel.STANDARD:
|
|
141
|
+
return BigDialConfig(
|
|
142
|
+
max_memory_gb=0.5,
|
|
143
|
+
dedup_cache_size=50000,
|
|
144
|
+
drain_similarity=0.5,
|
|
145
|
+
fuzzy_threshold=None, # Disabled for speed
|
|
146
|
+
max_patterns=500,
|
|
147
|
+
examples_per_pattern=2,
|
|
148
|
+
)
|
|
149
|
+
elif level == ProcessingLevel.ENHANCED:
|
|
150
|
+
return BigDialConfig(
|
|
151
|
+
max_memory_gb=1.0,
|
|
152
|
+
dedup_cache_size=100000,
|
|
153
|
+
drain_similarity=0.4,
|
|
154
|
+
fuzzy_threshold=0.8,
|
|
155
|
+
max_patterns=1000,
|
|
156
|
+
examples_per_pattern=3,
|
|
157
|
+
)
|
|
158
|
+
else: # MAXIMUM
|
|
159
|
+
return BigDialConfig(
|
|
160
|
+
max_memory_gb=2.0,
|
|
161
|
+
dedup_cache_size=200000,
|
|
162
|
+
drain_similarity=0.3,
|
|
163
|
+
fuzzy_threshold=0.9,
|
|
164
|
+
max_patterns=2000,
|
|
165
|
+
examples_per_pattern=5,
|
|
166
|
+
)
|
logreducer/core.py
ADDED
|
@@ -0,0 +1,493 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Core LogReducer implementation
|
|
3
|
+
"""
|
|
4
|
+
|
|
5
|
+
import json
|
|
6
|
+
import os
|
|
7
|
+
import random
|
|
8
|
+
import time
|
|
9
|
+
from dataclasses import replace
|
|
10
|
+
from datetime import datetime
|
|
11
|
+
from enum import Enum
|
|
12
|
+
from pathlib import Path
|
|
13
|
+
from typing import Any
|
|
14
|
+
|
|
15
|
+
from .anomaly import AnomalyDetector
|
|
16
|
+
from .config import BigDialConfig, OutputFormat, ProcessingLevel, ProcessingMode, get_preset_config
|
|
17
|
+
from .logging_config import get_logger, setup_logging
|
|
18
|
+
from .memory import BoundedDeduplicator, MemoryMonitor
|
|
19
|
+
from .patterns import FuzzyDeduplicator, PatternExtractor
|
|
20
|
+
from .sampling import reservoir_sample
|
|
21
|
+
from .sinks import Sink
|
|
22
|
+
from .sources import FileSource, Source
|
|
23
|
+
from .temporal import TemporalProcessor
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
class LogReducer:
|
|
27
|
+
"""
|
|
28
|
+
Main log reduction class
|
|
29
|
+
|
|
30
|
+
Example:
|
|
31
|
+
reducer = LogReducer(level="enhanced")
|
|
32
|
+
reduced = reducer.process_file("app.log")
|
|
33
|
+
"""
|
|
34
|
+
|
|
35
|
+
def __init__(
|
|
36
|
+
self,
|
|
37
|
+
level: str | ProcessingLevel = "standard",
|
|
38
|
+
mode: str | ProcessingMode = "pattern",
|
|
39
|
+
max_memory_gb: float | None = None,
|
|
40
|
+
max_patterns: int | None = None,
|
|
41
|
+
config: BigDialConfig | None = None,
|
|
42
|
+
**kwargs: Any,
|
|
43
|
+
) -> None:
|
|
44
|
+
"""
|
|
45
|
+
Initialize LogReducer
|
|
46
|
+
|
|
47
|
+
Args:
|
|
48
|
+
level: Processing level (standard/enhanced/maximum)
|
|
49
|
+
mode: Processing mode (pattern/anomaly/temporal/hybrid)
|
|
50
|
+
max_memory_gb: Memory limit override
|
|
51
|
+
max_patterns: Maximum patterns override
|
|
52
|
+
config: A fully-built BigDialConfig to use INSTEAD of the level
|
|
53
|
+
preset - the injection seam for a host application that owns
|
|
54
|
+
its own config cascade (build the config however you like and
|
|
55
|
+
hand it over; kwargs still apply on top). Copied, so the
|
|
56
|
+
caller's object is never mutated. ``level`` then only selects
|
|
57
|
+
the fuzzy-dedup gate, not preset values.
|
|
58
|
+
**kwargs: Additional config overrides
|
|
59
|
+
"""
|
|
60
|
+
# Parse level and mode
|
|
61
|
+
if isinstance(level, str):
|
|
62
|
+
level = ProcessingLevel(level.lower())
|
|
63
|
+
if isinstance(mode, str):
|
|
64
|
+
mode = ProcessingMode(mode.lower())
|
|
65
|
+
|
|
66
|
+
self.level = level
|
|
67
|
+
self.mode = mode
|
|
68
|
+
|
|
69
|
+
# Base config: an injected one wins over the level preset.
|
|
70
|
+
self.config = replace(config) if config is not None else get_preset_config(level)
|
|
71
|
+
|
|
72
|
+
# Apply overrides BEFORE setting up logging
|
|
73
|
+
if max_memory_gb:
|
|
74
|
+
self.config.max_memory_gb = max_memory_gb
|
|
75
|
+
if max_patterns:
|
|
76
|
+
self.config.max_patterns = max_patterns
|
|
77
|
+
|
|
78
|
+
for key, value in kwargs.items():
|
|
79
|
+
# Fail fast on typos: silently dropping an unknown override means a
|
|
80
|
+
# user's tuning quietly does nothing.
|
|
81
|
+
if not hasattr(self.config, key):
|
|
82
|
+
raise ValueError(f"Unknown config option {key!r} (see BigDialConfig for valid fields)")
|
|
83
|
+
# Special handling for output_format to convert string to enum
|
|
84
|
+
if key == "output_format" and isinstance(value, str):
|
|
85
|
+
value = OutputFormat(value.lower())
|
|
86
|
+
setattr(self.config, key, value)
|
|
87
|
+
|
|
88
|
+
# Setup logging based on config (after overrides applied)
|
|
89
|
+
setup_logging(
|
|
90
|
+
enable=self.config.enable_logging,
|
|
91
|
+
log_file=self.config.log_file,
|
|
92
|
+
log_level=self.config.log_level,
|
|
93
|
+
log_format=self.config.log_format,
|
|
94
|
+
)
|
|
95
|
+
|
|
96
|
+
# Get logger for this module
|
|
97
|
+
self.logger = get_logger("logreducer.core")
|
|
98
|
+
|
|
99
|
+
# Initialize components
|
|
100
|
+
self.memory_monitor = MemoryMonitor(self.config.max_memory_gb)
|
|
101
|
+
|
|
102
|
+
if mode in [ProcessingMode.ANOMALY, ProcessingMode.HYBRID]:
|
|
103
|
+
self.anomaly_detector: AnomalyDetector | None = AnomalyDetector(self.config.anomaly_contamination)
|
|
104
|
+
else:
|
|
105
|
+
self.anomaly_detector = None
|
|
106
|
+
|
|
107
|
+
# Stateful analysis components (dedup seen-set, Drain3 miners, fuzzy
|
|
108
|
+
# LSH) are (re)created per run by _reset_components(), so a reused
|
|
109
|
+
# reducer never carries one run's accumulated state into the next.
|
|
110
|
+
self.fuzzy_dedup: FuzzyDeduplicator | None = None
|
|
111
|
+
self.temporal_processor: TemporalProcessor | None = None
|
|
112
|
+
self._reset_components()
|
|
113
|
+
|
|
114
|
+
self.stats: dict[str, Any] = {}
|
|
115
|
+
|
|
116
|
+
def _reset_components(self) -> None:
|
|
117
|
+
"""Recreate the stateful analysis components for a fresh, isolated run.
|
|
118
|
+
|
|
119
|
+
The deduplicator (seen-set), Drain3 miners (pattern AND per-window
|
|
120
|
+
temporal), and fuzzy-dedup LSH all accumulate per-line state.
|
|
121
|
+
Recreating them keeps a LogReducer instance reusable across
|
|
122
|
+
reduce()/process_file() calls, and lets a single run make independent
|
|
123
|
+
deduplication passes (hybrid mode) without the first pass poisoning
|
|
124
|
+
the second.
|
|
125
|
+
"""
|
|
126
|
+
self.deduplicator = BoundedDeduplicator(self.config.dedup_cache_size, self.config.hash_algorithm)
|
|
127
|
+
self.pattern_extractor = PatternExtractor(self.config)
|
|
128
|
+
if self.config.fuzzy_threshold and self.level != ProcessingLevel.STANDARD:
|
|
129
|
+
self.fuzzy_dedup = FuzzyDeduplicator(self.config.fuzzy_threshold)
|
|
130
|
+
else:
|
|
131
|
+
self.fuzzy_dedup = None
|
|
132
|
+
if self.mode in [ProcessingMode.TEMPORAL, ProcessingMode.HYBRID]:
|
|
133
|
+
self.temporal_processor = TemporalProcessor(self.config.temporal_window_minutes)
|
|
134
|
+
else:
|
|
135
|
+
self.temporal_processor = None
|
|
136
|
+
|
|
137
|
+
def reduce(
|
|
138
|
+
self,
|
|
139
|
+
source: Source,
|
|
140
|
+
output_file: str | None = None,
|
|
141
|
+
return_metadata: bool = False,
|
|
142
|
+
sink: Sink | None = None,
|
|
143
|
+
) -> list[str] | dict:
|
|
144
|
+
"""Reduce a source of log lines to a representative sample.
|
|
145
|
+
|
|
146
|
+
This is the core entry point: it operates on an abstraction, not on IO.
|
|
147
|
+
The source is any re-iterable stream of str lines - a list, a
|
|
148
|
+
FileSource, or an app-provided iterable wrapping its own DB cursor or
|
|
149
|
+
Kafka consumer. The reducer never manages the connection or loading
|
|
150
|
+
path.
|
|
151
|
+
|
|
152
|
+
Args:
|
|
153
|
+
source: Re-iterable stream of str lines (see logreducer.sources).
|
|
154
|
+
Multi-pass modes (hybrid) require the source to be re-iterable.
|
|
155
|
+
output_file: Optional path to also write the result to, with a
|
|
156
|
+
format-aware ``.meta.json`` sidecar of run stats (CLI use).
|
|
157
|
+
return_metadata: Return a dict with lines + stats + config instead
|
|
158
|
+
of just the lines.
|
|
159
|
+
sink: Optional output abstraction (see logreducer.sinks). The
|
|
160
|
+
reduced lines are also handed to ``sink.write`` - a FileSink, a
|
|
161
|
+
KafkaSink, or any app-provided destination.
|
|
162
|
+
|
|
163
|
+
Returns:
|
|
164
|
+
The reduced lines in memory (list[str]), or a metadata dict.
|
|
165
|
+
"""
|
|
166
|
+
# A Source must be re-iterable: reduce() counts lines in one pass, then
|
|
167
|
+
# re-reads the source to process it (hybrid re-reads again). A one-shot
|
|
168
|
+
# iterator (a bare generator) would be drained by the count and leave
|
|
169
|
+
# nothing to process - fail loudly rather than return an empty result.
|
|
170
|
+
if iter(source) is source:
|
|
171
|
+
raise TypeError(
|
|
172
|
+
"source must be re-iterable (a fresh iterator on each pass); got a "
|
|
173
|
+
"one-shot iterator/generator - wrap it, e.g. list(source)."
|
|
174
|
+
)
|
|
175
|
+
|
|
176
|
+
# Fresh analysis state per run: a reused reducer must not carry the
|
|
177
|
+
# previous run's dedup/miner/LSH state, which would drop every line as
|
|
178
|
+
# already-seen and silently return an empty result.
|
|
179
|
+
self._reset_components()
|
|
180
|
+
|
|
181
|
+
start_time = time.time()
|
|
182
|
+
|
|
183
|
+
# Count input lines for the reduction ratio (one pass over the source).
|
|
184
|
+
# Note: a size-sampled FileSource yields its sampled line count, so for
|
|
185
|
+
# very large files the ratio and input_lines stat are approximate.
|
|
186
|
+
input_lines = sum(1 for _ in source)
|
|
187
|
+
|
|
188
|
+
size_bytes = getattr(source, "size_bytes", None)
|
|
189
|
+
file_size_mb = size_bytes / (1024 * 1024) if size_bytes else None
|
|
190
|
+
|
|
191
|
+
if self.config.enable_logging:
|
|
192
|
+
where = f"{file_size_mb:.1f} MB" if file_size_mb is not None else f"{input_lines:,} lines"
|
|
193
|
+
self.logger.info(f"Reducing {source!r} ({where})")
|
|
194
|
+
self.logger.info(f"Mode: {self.mode.value}, Level: {self.level.value}")
|
|
195
|
+
self.logger.info(f"Memory limit: {self.config.max_memory_gb:.1f} GB")
|
|
196
|
+
|
|
197
|
+
# Process based on mode
|
|
198
|
+
if self.mode == ProcessingMode.PATTERN:
|
|
199
|
+
result_lines = self._process_pattern_mode(source)
|
|
200
|
+
elif self.mode == ProcessingMode.ANOMALY:
|
|
201
|
+
result_lines = self._process_anomaly_mode(source)
|
|
202
|
+
elif self.mode == ProcessingMode.TEMPORAL:
|
|
203
|
+
result_lines = self._process_temporal_mode(source)
|
|
204
|
+
else: # HYBRID
|
|
205
|
+
result_lines = self._process_hybrid_mode(source)
|
|
206
|
+
|
|
207
|
+
# Calculate stats
|
|
208
|
+
processing_time = time.time() - start_time
|
|
209
|
+
output_lines = len(result_lines)
|
|
210
|
+
reduction_percent = (1 - output_lines / max(input_lines, 1)) * 100 if input_lines > 0 else 0
|
|
211
|
+
input_label = getattr(source, "path", None)
|
|
212
|
+
|
|
213
|
+
self.stats = {
|
|
214
|
+
"input_file": str(input_label) if input_label is not None else repr(source),
|
|
215
|
+
"input_lines": input_lines,
|
|
216
|
+
"input_size_mb": file_size_mb,
|
|
217
|
+
"output_lines": output_lines,
|
|
218
|
+
"reduction_percent": reduction_percent,
|
|
219
|
+
"processing_time_seconds": processing_time,
|
|
220
|
+
"processing_rate_mb_per_sec": (
|
|
221
|
+
file_size_mb / max(processing_time, 0.001) if file_size_mb is not None else None
|
|
222
|
+
),
|
|
223
|
+
"mode": self.mode.value,
|
|
224
|
+
"level": self.level.value,
|
|
225
|
+
"memory_limit_gb": self.config.max_memory_gb,
|
|
226
|
+
}
|
|
227
|
+
|
|
228
|
+
if output_file:
|
|
229
|
+
self._save_output(result_lines, output_file)
|
|
230
|
+
|
|
231
|
+
if sink is not None:
|
|
232
|
+
sink.write(result_lines)
|
|
233
|
+
|
|
234
|
+
self._print_summary()
|
|
235
|
+
|
|
236
|
+
if return_metadata:
|
|
237
|
+
return {
|
|
238
|
+
"lines": result_lines,
|
|
239
|
+
"stats": self.stats,
|
|
240
|
+
"config": self._config_as_dict(),
|
|
241
|
+
}
|
|
242
|
+
return result_lines
|
|
243
|
+
|
|
244
|
+
def process_file(
|
|
245
|
+
self,
|
|
246
|
+
input_file: str,
|
|
247
|
+
output_file: str | None = None,
|
|
248
|
+
return_metadata: bool = False,
|
|
249
|
+
) -> list[str] | dict:
|
|
250
|
+
"""Reduce a log file - convenience wrapper over reduce() + FileSource.
|
|
251
|
+
|
|
252
|
+
Args:
|
|
253
|
+
input_file: Path to the input log file.
|
|
254
|
+
output_file: Optional output file path.
|
|
255
|
+
return_metadata: Return a metadata dict instead of just the lines.
|
|
256
|
+
|
|
257
|
+
Returns:
|
|
258
|
+
The reduced lines (list[str]) or a metadata dict.
|
|
259
|
+
"""
|
|
260
|
+
if not os.path.exists(input_file):
|
|
261
|
+
raise FileNotFoundError(f"File not found: {input_file}")
|
|
262
|
+
source = FileSource(input_file, max_memory_gb=self.config.max_memory_gb)
|
|
263
|
+
return self.reduce(source, output_file=output_file, return_metadata=return_metadata)
|
|
264
|
+
|
|
265
|
+
def _process_pattern_mode(self, source: Source) -> list[str]:
|
|
266
|
+
"""Process using pattern extraction, streaming end to end.
|
|
267
|
+
|
|
268
|
+
Exact dedup -> optional fuzzy dedup -> Drain3 all run as generators, so
|
|
269
|
+
the unique lines are never collected into a list. Peak memory is the
|
|
270
|
+
bounded dedup cache plus the Drain3 template store (see max_clusters),
|
|
271
|
+
independent of how many unique lines the source has.
|
|
272
|
+
"""
|
|
273
|
+
if self.config.enable_logging:
|
|
274
|
+
# One message, not phases: dedup and mining run interleaved as a
|
|
275
|
+
# single generator pipeline, so there is no phase boundary to report.
|
|
276
|
+
self.logger.info("Streaming pipeline: dedup -> fuzzy dedup -> pattern mining")
|
|
277
|
+
|
|
278
|
+
unique_stream = self.deduplicator.deduplicate_lines(source)
|
|
279
|
+
if self.fuzzy_dedup:
|
|
280
|
+
unique_stream = self.fuzzy_dedup.deduplicate_stream(unique_stream)
|
|
281
|
+
patterns = self.pattern_extractor.extract_patterns(unique_stream)
|
|
282
|
+
|
|
283
|
+
# Collect examples
|
|
284
|
+
result = []
|
|
285
|
+
for pattern in patterns[: self.config.max_patterns]:
|
|
286
|
+
result.extend(pattern.examples)
|
|
287
|
+
|
|
288
|
+
return result
|
|
289
|
+
|
|
290
|
+
def _cap_for_anomaly(self, unique_lines: list[str]) -> list[str]:
|
|
291
|
+
"""Reservoir-cap unique lines to anomaly_max_rows to bound the ML matrix.
|
|
292
|
+
|
|
293
|
+
Anomaly detection (Isolation Forest over a TF-IDF matrix) is batch ML and
|
|
294
|
+
cannot stream, so a huge unique-line set is the one place a hard memory
|
|
295
|
+
cap needs an explicit sample. Fixed seed = reproducible across passes.
|
|
296
|
+
Trades anomaly recall (a rare line may be sampled out) for bounded memory;
|
|
297
|
+
off unless anomaly_max_rows is set.
|
|
298
|
+
"""
|
|
299
|
+
cap = self.config.anomaly_max_rows
|
|
300
|
+
if cap and len(unique_lines) > cap:
|
|
301
|
+
return reservoir_sample(unique_lines, cap, random.Random(0))
|
|
302
|
+
return unique_lines
|
|
303
|
+
|
|
304
|
+
def _process_anomaly_mode(self, source: Source) -> list[str]:
|
|
305
|
+
"""Process using anomaly detection"""
|
|
306
|
+
if not self.anomaly_detector:
|
|
307
|
+
if self.config.enable_logging:
|
|
308
|
+
self.logger.warning("Anomaly detector not available, falling back to pattern mode")
|
|
309
|
+
return self._process_pattern_mode(source)
|
|
310
|
+
|
|
311
|
+
if self.config.enable_logging:
|
|
312
|
+
self.logger.info("Phase 1/2: Reading and deduplicating")
|
|
313
|
+
unique_lines = self._cap_for_anomaly(list(self.deduplicator.deduplicate_lines(source)))
|
|
314
|
+
|
|
315
|
+
if self.config.enable_logging:
|
|
316
|
+
self.logger.info("Phase 2/2: Anomaly detection")
|
|
317
|
+
anomalous, normal = self.anomaly_detector.detect_anomalies(unique_lines)
|
|
318
|
+
|
|
319
|
+
# Keep all anomalies + sample of normal
|
|
320
|
+
result = anomalous[: self.config.max_patterns // 2]
|
|
321
|
+
|
|
322
|
+
# Add some normal lines for context. Seeded RNG so a re-run over the
|
|
323
|
+
# same input yields the same sample (consistent with the reproducibility
|
|
324
|
+
# guarantees elsewhere); there is no security requirement here.
|
|
325
|
+
normal_sample_size = min(len(normal), self.config.max_patterns // 4)
|
|
326
|
+
if normal_sample_size > 0:
|
|
327
|
+
result.extend(random.Random(0).sample(normal, normal_sample_size))
|
|
328
|
+
|
|
329
|
+
return result
|
|
330
|
+
|
|
331
|
+
def _process_temporal_mode(self, source: Source) -> list[str]:
|
|
332
|
+
"""Process using temporal analysis"""
|
|
333
|
+
if not self.temporal_processor:
|
|
334
|
+
if self.config.enable_logging:
|
|
335
|
+
self.logger.warning("Temporal processor not available, falling back to pattern mode")
|
|
336
|
+
return self._process_pattern_mode(source)
|
|
337
|
+
|
|
338
|
+
if self.config.enable_logging:
|
|
339
|
+
self.logger.info("Phase 1/2: Reading lines")
|
|
340
|
+
lines = list(source)
|
|
341
|
+
|
|
342
|
+
if self.config.enable_logging:
|
|
343
|
+
self.logger.info("Phase 2/2: Temporal processing")
|
|
344
|
+
temporal_results = self.temporal_processor.process_temporal(lines)
|
|
345
|
+
|
|
346
|
+
# Collect examples from temporal patterns
|
|
347
|
+
result = []
|
|
348
|
+
for pattern in temporal_results.get("temporal_patterns", [])[: self.config.max_patterns]:
|
|
349
|
+
if "example" in pattern:
|
|
350
|
+
result.append(pattern["example"])
|
|
351
|
+
|
|
352
|
+
for pattern in temporal_results.get("timeless_patterns", [])[:100]:
|
|
353
|
+
if "example" in pattern:
|
|
354
|
+
result.append(pattern["example"])
|
|
355
|
+
|
|
356
|
+
return result
|
|
357
|
+
|
|
358
|
+
def _process_hybrid_mode(self, source: Source) -> list[str]:
|
|
359
|
+
"""Process using combined approach"""
|
|
360
|
+
if self.config.enable_logging:
|
|
361
|
+
self.logger.info("Hybrid mode: combining pattern and anomaly detection")
|
|
362
|
+
|
|
363
|
+
# Get patterns
|
|
364
|
+
pattern_lines = self._process_pattern_mode(source)
|
|
365
|
+
|
|
366
|
+
# Get anomalies if available
|
|
367
|
+
if self.anomaly_detector:
|
|
368
|
+
# The pattern pass above consumed self.deduplicator's seen-set, so the
|
|
369
|
+
# anomaly pass needs a fresh deduplicator - otherwise every line reads
|
|
370
|
+
# as already-seen and no anomalies survive (hybrid -> pattern-only).
|
|
371
|
+
self.deduplicator = BoundedDeduplicator(self.config.dedup_cache_size, self.config.hash_algorithm)
|
|
372
|
+
unique_lines = self._cap_for_anomaly(list(self.deduplicator.deduplicate_lines(source)))
|
|
373
|
+
anomalous, _ = self.anomaly_detector.detect_anomalies(unique_lines)
|
|
374
|
+
|
|
375
|
+
# Combine, preferring anomalies
|
|
376
|
+
result = anomalous[: self.config.max_patterns // 2]
|
|
377
|
+
result.extend(pattern_lines[: self.config.max_patterns // 2])
|
|
378
|
+
else:
|
|
379
|
+
result = pattern_lines
|
|
380
|
+
|
|
381
|
+
# Remove duplicates while preserving order
|
|
382
|
+
seen = set()
|
|
383
|
+
final = []
|
|
384
|
+
for line in result:
|
|
385
|
+
if line not in seen:
|
|
386
|
+
seen.add(line)
|
|
387
|
+
final.append(line)
|
|
388
|
+
|
|
389
|
+
return final[: self.config.max_patterns]
|
|
390
|
+
|
|
391
|
+
def _config_as_dict(self) -> dict[str, Any]:
|
|
392
|
+
"""Config as a JSON-serialisable dict.
|
|
393
|
+
|
|
394
|
+
Enum members (output_format, ...) become their ``.value`` string so the
|
|
395
|
+
result survives ``json.dumps`` and never leaks ``OutputFormat.LINE``-style
|
|
396
|
+
reprs into metadata output. Private (``_``-prefixed) attrs are dropped.
|
|
397
|
+
"""
|
|
398
|
+
return {
|
|
399
|
+
k: (v.value if isinstance(v, Enum) else v) for k, v in vars(self.config).items() if not k.startswith("_")
|
|
400
|
+
}
|
|
401
|
+
|
|
402
|
+
def _save_output(self, lines: list[str], output_file: str) -> None:
|
|
403
|
+
"""Save output to file in specified format"""
|
|
404
|
+
output_path = Path(output_file)
|
|
405
|
+
output_path.parent.mkdir(parents=True, exist_ok=True)
|
|
406
|
+
|
|
407
|
+
# Save in requested format
|
|
408
|
+
if self.config.output_format == OutputFormat.JSON:
|
|
409
|
+
# Full JSON format with metadata
|
|
410
|
+
output_data = {
|
|
411
|
+
"lines": lines,
|
|
412
|
+
"stats": self.stats,
|
|
413
|
+
"config": self._config_as_dict(),
|
|
414
|
+
"timestamp": datetime.now().isoformat(),
|
|
415
|
+
}
|
|
416
|
+
with open(output_path, "w", encoding="utf-8") as f:
|
|
417
|
+
json.dump(output_data, f, indent=2 if self.config.pretty_json else None)
|
|
418
|
+
|
|
419
|
+
if self.config.enable_logging:
|
|
420
|
+
self.logger.info(f"Output saved to {output_file}")
|
|
421
|
+
|
|
422
|
+
elif self.config.output_format == OutputFormat.JSONL:
|
|
423
|
+
# JSON Lines format - one JSON object per line
|
|
424
|
+
with open(output_path, "w", encoding="utf-8") as f:
|
|
425
|
+
for line in lines:
|
|
426
|
+
json.dump({"line": line, "timestamp": datetime.now().isoformat()}, f)
|
|
427
|
+
f.write("\n")
|
|
428
|
+
|
|
429
|
+
if self.config.enable_logging:
|
|
430
|
+
self.logger.info(f"Output saved to {output_file}")
|
|
431
|
+
|
|
432
|
+
else: # OutputFormat.LINE (default)
|
|
433
|
+
# Traditional line-by-line text output
|
|
434
|
+
with open(output_path, "w", encoding="utf-8") as f:
|
|
435
|
+
for line in lines:
|
|
436
|
+
f.write(line + "\n")
|
|
437
|
+
|
|
438
|
+
# Save metadata separately for line format
|
|
439
|
+
meta_file = output_path.with_suffix(".meta.json")
|
|
440
|
+
with open(meta_file, "w", encoding="utf-8") as f:
|
|
441
|
+
json.dump(
|
|
442
|
+
{
|
|
443
|
+
"stats": self.stats,
|
|
444
|
+
"config": self._config_as_dict(),
|
|
445
|
+
"timestamp": datetime.now().isoformat(),
|
|
446
|
+
},
|
|
447
|
+
f,
|
|
448
|
+
indent=2,
|
|
449
|
+
)
|
|
450
|
+
|
|
451
|
+
if self.config.enable_logging:
|
|
452
|
+
self.logger.info(f"Output saved to {output_file}")
|
|
453
|
+
self.logger.info(f"Metadata saved to {meta_file}")
|
|
454
|
+
|
|
455
|
+
def _print_summary(self) -> None:
|
|
456
|
+
"""Log processing summary"""
|
|
457
|
+
self.logger.info("\n" + "=" * 60)
|
|
458
|
+
self.logger.info("LOG REDUCTION SUMMARY")
|
|
459
|
+
self.logger.info("=" * 60)
|
|
460
|
+
self.logger.info(f"Mode: {self.mode.value}")
|
|
461
|
+
self.logger.info(f"Level: {self.level.value}")
|
|
462
|
+
# input_size_mb / rate are None for non-file sources (a list, a DB
|
|
463
|
+
# cursor) - only a file has a byte size. Log them when known.
|
|
464
|
+
input_size_mb = self.stats.get("input_size_mb")
|
|
465
|
+
if input_size_mb is not None:
|
|
466
|
+
self.logger.info(f"Input: {input_size_mb:.1f} MB")
|
|
467
|
+
self.logger.info(f"Input lines: {self.stats['input_lines']:,}")
|
|
468
|
+
self.logger.info(f"Output: {self.stats['output_lines']} lines")
|
|
469
|
+
self.logger.info(f"Reduction: {self.stats['reduction_percent']:.1f}%")
|
|
470
|
+
self.logger.info(f"Time: {self.stats['processing_time_seconds']:.1f} seconds")
|
|
471
|
+
rate = self.stats.get("processing_rate_mb_per_sec")
|
|
472
|
+
if rate is not None:
|
|
473
|
+
self.logger.info(f"Rate: {rate:.1f} MB/sec")
|
|
474
|
+
self.logger.info("=" * 60)
|
|
475
|
+
|
|
476
|
+
def estimate_processing(self, file_path: str) -> dict:
|
|
477
|
+
"""Estimate processing requirements before running"""
|
|
478
|
+
file_size = os.path.getsize(file_path)
|
|
479
|
+
file_size_gb = file_size / (1024**3)
|
|
480
|
+
|
|
481
|
+
strategy = self.memory_monitor.estimate_file_strategy(file_size)
|
|
482
|
+
|
|
483
|
+
return {
|
|
484
|
+
"file_size_gb": file_size_gb,
|
|
485
|
+
"memory_required_gb": min(file_size_gb * 0.3, self.config.max_memory_gb),
|
|
486
|
+
"strategy": strategy,
|
|
487
|
+
"estimated_time_seconds": file_size_gb * 30,
|
|
488
|
+
"will_sample": strategy == "sampled",
|
|
489
|
+
"estimated_output_lines": min(
|
|
490
|
+
int(file_size_gb * 1000),
|
|
491
|
+
self.config.max_patterns * self.config.examples_per_pattern,
|
|
492
|
+
),
|
|
493
|
+
}
|