logreducer 3.4.0__tar.gz → 3.5.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (23) hide show
  1. {logreducer-3.4.0 → logreducer-3.5.0}/PKG-INFO +1 -1
  2. logreducer-3.5.0/pyproject.toml +300 -0
  3. logreducer-3.4.0/pyproject.toml → logreducer-3.5.0/pyproject.toml.orig +6 -6
  4. {logreducer-3.4.0 → logreducer-3.5.0}/src/logreducer/config.py +8 -2
  5. {logreducer-3.4.0 → logreducer-3.5.0}/src/logreducer/patterns.py +44 -0
  6. {logreducer-3.4.0 → logreducer-3.5.0}/LICENSE +0 -0
  7. {logreducer-3.4.0 → logreducer-3.5.0}/NOTICE +0 -0
  8. {logreducer-3.4.0 → logreducer-3.5.0}/README.md +0 -0
  9. {logreducer-3.4.0 → logreducer-3.5.0}/src/logreducer/__init__.py +0 -0
  10. {logreducer-3.4.0 → logreducer-3.5.0}/src/logreducer/anomaly.py +0 -0
  11. {logreducer-3.4.0 → logreducer-3.5.0}/src/logreducer/cli.py +0 -0
  12. {logreducer-3.4.0 → logreducer-3.5.0}/src/logreducer/clickhouse.py +0 -0
  13. {logreducer-3.4.0 → logreducer-3.5.0}/src/logreducer/core.py +0 -0
  14. {logreducer-3.4.0 → logreducer-3.5.0}/src/logreducer/kafka.py +0 -0
  15. {logreducer-3.4.0 → logreducer-3.5.0}/src/logreducer/logging_config.py +0 -0
  16. {logreducer-3.4.0 → logreducer-3.5.0}/src/logreducer/memory.py +0 -0
  17. {logreducer-3.4.0 → logreducer-3.5.0}/src/logreducer/py.typed +0 -0
  18. {logreducer-3.4.0 → logreducer-3.5.0}/src/logreducer/sampling.py +0 -0
  19. {logreducer-3.4.0 → logreducer-3.5.0}/src/logreducer/sinks.py +0 -0
  20. {logreducer-3.4.0 → logreducer-3.5.0}/src/logreducer/sources.py +0 -0
  21. {logreducer-3.4.0 → logreducer-3.5.0}/src/logreducer/sql.py +0 -0
  22. {logreducer-3.4.0 → logreducer-3.5.0}/src/logreducer/target.py +0 -0
  23. {logreducer-3.4.0 → logreducer-3.5.0}/src/logreducer/temporal.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: logreducer
3
- Version: 3.4.0
3
+ Version: 3.5.0
4
4
  Summary: Reduce GB-scale logs to a representative sample - streaming Python library and CLI with pattern mining and anomaly detection
5
5
  Keywords: log,logging,reduction,analysis,pattern-extraction,memory-efficient,streaming,anomaly-detection
6
6
  Author: HyperI Team
@@ -0,0 +1,300 @@
1
+ [build-system]
2
+ requires = ["uv_build>=0.8.0,<1"]
3
+ build-backend = "uv_build"
4
+
5
+ [project]
6
+ name = "logreducer"
7
+ version = "3.5.0"
8
+ description = "Reduce GB-scale logs to a representative sample - streaming Python library and CLI with pattern mining and anomaly detection"
9
+ readme = "README.md"
10
+ license = "Apache-2.0"
11
+ license-files = [
12
+ "LICENSE",
13
+ "NOTICE",
14
+ ]
15
+ requires-python = ">=3.12"
16
+ classifiers = [
17
+ "Development Status :: 5 - Production/Stable",
18
+ "Intended Audience :: Developers",
19
+ "Intended Audience :: System Administrators",
20
+ "Topic :: System :: Logging",
21
+ "Topic :: System :: Systems Administration",
22
+ "Topic :: Software Development :: Libraries :: Python Modules",
23
+ "Programming Language :: Python :: 3",
24
+ "Programming Language :: Python :: 3 :: Only",
25
+ "Operating System :: OS Independent",
26
+ ]
27
+ keywords = [
28
+ "log",
29
+ "logging",
30
+ "reduction",
31
+ "analysis",
32
+ "pattern-extraction",
33
+ "memory-efficient",
34
+ "streaming",
35
+ "anomaly-detection",
36
+ ]
37
+ dependencies = [
38
+ "drain3>=0.9.0",
39
+ "psutil>=7.2.0",
40
+ "loguru>=0.7.3",
41
+ "numpy>=2.0.0",
42
+ "scikit-learn>=1.5.0",
43
+ "typer>=0.25.0",
44
+ ]
45
+
46
+ [[project.authors]]
47
+ name = "HyperI Team"
48
+ email = "dev@hyperi.io"
49
+
50
+ [[project.maintainers]]
51
+ name = "HyperI Team"
52
+ email = "dev@hyperi.io"
53
+
54
+ [project.optional-dependencies]
55
+ enhanced = [
56
+ "xxhash>=3.0.0",
57
+ "datasketch>=1.5.0",
58
+ ]
59
+ sql = ["sqlalchemy>=2.0.30"]
60
+ clickhouse = ["clickhouse-connect>=1.0"]
61
+ kafka = ["confluent-kafka>=2.14.0"]
62
+
63
+ [project.urls]
64
+ Homepage = "https://github.com/hyperi-io/logreducer"
65
+ Documentation = "https://github.com/hyperi-io/logreducer#readme"
66
+ Repository = "https://github.com/hyperi-io/logreducer"
67
+ "Bug Reports" = "https://github.com/hyperi-io/logreducer/issues"
68
+ Changelog = "https://github.com/hyperi-io/logreducer/blob/main/CHANGELOG.md"
69
+
70
+ [project.scripts]
71
+ logreducer = "logreducer.cli:main"
72
+
73
+ [tool.uv]
74
+ index-strategy = "unsafe-best-match"
75
+ override-dependencies = []
76
+
77
+ [[tool.uv.index]]
78
+ name = "pypi"
79
+ url = "https://pypi.org/simple"
80
+ default = true
81
+
82
+ [tool.uv.build-backend]
83
+ module-name = "logreducer"
84
+ module-root = "src"
85
+
86
+ [tool.ruff]
87
+ target-version = "py312"
88
+ line-length = 120
89
+ indent-width = 4
90
+ src = ["src"]
91
+ extend-exclude = ["examples"]
92
+
93
+ [tool.ruff.lint]
94
+ select = [
95
+ "E",
96
+ "W",
97
+ "F",
98
+ "I",
99
+ "B",
100
+ "C4",
101
+ "UP",
102
+ "ARG",
103
+ "SIM",
104
+ "S",
105
+ "N",
106
+ "PT",
107
+ "RUF",
108
+ "PIE",
109
+ "T20",
110
+ ]
111
+ ignore = [
112
+ "ARG001",
113
+ "ARG002",
114
+ "B904",
115
+ "B008",
116
+ "B017",
117
+ "B007",
118
+ "B028",
119
+ "E402",
120
+ "E722",
121
+ "E501",
122
+ "E741",
123
+ "SIM105",
124
+ "SIM108",
125
+ "SIM102",
126
+ "F821",
127
+ "F601",
128
+ "TRY300",
129
+ "C901",
130
+ "UP035",
131
+ "W191",
132
+ "S104",
133
+ "S110",
134
+ "S112",
135
+ "S311",
136
+ "S603",
137
+ "S607",
138
+ "S608",
139
+ "T201",
140
+ "RUF012",
141
+ "RUF059",
142
+ "RUF001",
143
+ "RUF002",
144
+ "RUF005",
145
+ "RUF006",
146
+ "N806",
147
+ "N818",
148
+ "PIE810",
149
+ ]
150
+ fixable = ["ALL"]
151
+ unfixable = []
152
+ dummy-variable-rgx = "^(_+|(_+[a-zA-Z0-9_]*[a-zA-Z0-9]+?))$"
153
+
154
+ [tool.ruff.lint.per-file-ignores]
155
+ "tests/**" = [
156
+ "S101",
157
+ "S105",
158
+ "S106",
159
+ "S108",
160
+ "F541",
161
+ "RUF013",
162
+ "PT017",
163
+ "PT011",
164
+ "PT012",
165
+ "RUF043",
166
+ ]
167
+
168
+ [tool.ruff.lint.pydocstyle]
169
+ convention = "google"
170
+
171
+ [tool.ruff.format]
172
+ quote-style = "double"
173
+ indent-style = "space"
174
+ skip-magic-trailing-comma = false
175
+ line-ending = "auto"
176
+
177
+ [tool.mypy]
178
+ python_version = "3.12"
179
+ ignore_missing_imports = true
180
+ warn_return_any = true
181
+ warn_unused_configs = true
182
+ disallow_untyped_defs = true
183
+ disallow_incomplete_defs = true
184
+ check_untyped_defs = true
185
+ disallow_untyped_decorators = true
186
+ no_implicit_optional = true
187
+ warn_redundant_casts = true
188
+ warn_unused_ignores = true
189
+ warn_no_return = true
190
+ warn_unreachable = true
191
+ strict_equality = true
192
+
193
+ [tool.pytest.ini_options]
194
+ minversion = "8.0.0"
195
+ addopts = "-ra -q --strict-markers --strict-config"
196
+ testpaths = ["tests"]
197
+ python_files = [
198
+ "test_*.py",
199
+ "*_test.py",
200
+ ]
201
+ python_classes = ["Test*"]
202
+ python_functions = ["test_*"]
203
+ markers = [
204
+ """slow: marks tests as slow (deselect with '-m "not slow"')""",
205
+ "integration: marks tests as integration tests",
206
+ ]
207
+
208
+ [tool.coverage.run]
209
+ source = ["logreducer"]
210
+ omit = [
211
+ "*/tests/*",
212
+ "*/test_*",
213
+ "*/.venv/*",
214
+ "*/venv/*",
215
+ ]
216
+
217
+ [tool.coverage.report]
218
+ exclude_lines = [
219
+ "pragma: no cover",
220
+ "def __repr__",
221
+ "if self.debug:",
222
+ "if settings.DEBUG",
223
+ "raise AssertionError",
224
+ "raise NotImplementedError",
225
+ "if 0:",
226
+ "if __name__ == .__main__.:",
227
+ 'class .*\bProtocol\):',
228
+ '@(abc\.)?abstractmethod',
229
+ ]
230
+
231
+ [tool.bandit]
232
+ exclude_dirs = [
233
+ "tests",
234
+ "test",
235
+ ".venv",
236
+ "venv",
237
+ ".tox",
238
+ "build",
239
+ "dist",
240
+ "data/samples",
241
+ ]
242
+ skips = [
243
+ "B101",
244
+ "B603",
245
+ ]
246
+
247
+ [tool.bandit.assert_used]
248
+ skips = [
249
+ "*test*.py",
250
+ "*tests*.py",
251
+ ]
252
+
253
+ [tool.ty.environment]
254
+ python-version = "3.12"
255
+
256
+ [tool.ty.src]
257
+ exclude = [
258
+ "examples/",
259
+ "docs/",
260
+ "scripts/",
261
+ ]
262
+
263
+ [tool.vermin]
264
+ targets = ["3.12"]
265
+ no_tips = true
266
+
267
+ [tool.hatch.build.targets.sdist]
268
+ exclude = [
269
+ "/.github/copilot-instructions.md",
270
+ "/hyperi-ai",
271
+ "/.cursor",
272
+ "/CLAUDE.md",
273
+ "/CURSOR.md",
274
+ "/GEMINI.md",
275
+ "/.windsurf",
276
+ "/STATE.md",
277
+ "/ci",
278
+ "/.gemini",
279
+ "/.claude",
280
+ ]
281
+
282
+ [dependency-groups]
283
+ dev = [
284
+ "pytest>=9.0.0",
285
+ "pytest-cov>=7.0.0",
286
+ "ruff>=0.15.0",
287
+ "ty>=0.0.34",
288
+ "mypy>=2.0",
289
+ "bandit[toml]>=1.7.0",
290
+ "pip-audit>=2.6.0",
291
+ "vulture>=2.16",
292
+ "vermin>=1.6.0",
293
+ "sqlalchemy>=2.0.30",
294
+ "clickhouse-connect>=1.0",
295
+ "confluent-kafka>=2.14.0",
296
+ "testcontainers[clickhouse,kafka,postgres,mysql]>=4.14.0",
297
+ "psycopg[binary]>=3.2",
298
+ "pymysql>=1.1.0",
299
+ "python-dotenv>=1.0.0",
300
+ ]
@@ -6,7 +6,7 @@ build-backend = "uv_build"
6
6
 
7
7
  [project]
8
8
  name = "logreducer"
9
- version = "3.4.0"
9
+ version = "3.5.0"
10
10
  authors = [
11
11
  { name = "HyperI Team", email = "dev@hyperi.io" },
12
12
  ]
@@ -286,17 +286,17 @@ no_tips = true
286
286
 
287
287
  [tool.hatch.build.targets.sdist]
288
288
  exclude = [
289
+ "/.github/copilot-instructions.md",
290
+ "/hyperi-ai",
291
+ "/.cursor",
289
292
  "/CLAUDE.md",
293
+ "/CURSOR.md",
290
294
  "/GEMINI.md",
291
- "/.cursor",
292
295
  "/.windsurf",
293
296
  "/STATE.md",
294
- "/CURSOR.md",
295
- "/.claude",
296
- "/.github/copilot-instructions.md",
297
- "/hyperi-ai",
298
297
  "/ci",
299
298
  "/.gemini",
299
+ "/.claude",
300
300
  ]
301
301
 
302
302
  [dependency-groups]
@@ -61,6 +61,12 @@ class BigDialConfig:
61
61
  # TF-IDF matrix on a huge unique-line set, at the cost of anomaly recall
62
62
  # (rare lines may be sampled out). None = no cap (use every unique line).
63
63
  anomaly_max_rows: int | None = None
64
+ # Typed Drain3 masking: replace well-known value shapes (IPv4/IPv6, MAC,
65
+ # UUID, hex tokens, numbers) with typed template slots (<IP>, <NUM>, ...)
66
+ # before clustering, instead of the bare <*>. Opt-in: masking normalises
67
+ # lines before Drain3 sees them, so clustering can differ from the
68
+ # unmasked default - off keeps behaviour byte-identical.
69
+ typed_masking: bool = False
64
70
 
65
71
  # Temporal Control
66
72
  temporal_window_minutes: int = 60
@@ -104,14 +110,14 @@ class BigDialConfig:
104
110
  """
105
111
  if not prefixes:
106
112
  prefixes = ("LOGREDUCER",)
107
- overrides: dict[str, object] = {}
113
+ overrides: dict[str, typing.Any] = {}
108
114
  for field in fields(cls):
109
115
  for prefix in prefixes:
110
116
  raw = os.environ.get(f"{prefix.rstrip('_')}_{field.name.upper()}")
111
117
  if raw is not None:
112
118
  overrides[field.name] = _coerce_env_value(raw, field.name)
113
119
  break
114
- return cls(**overrides) # type: ignore[arg-type]
120
+ return cls(**overrides)
115
121
 
116
122
 
117
123
  def _coerce_env_value(raw: str, field_name: str) -> object:
@@ -5,6 +5,7 @@ from dataclasses import dataclass, field
5
5
  from typing import TYPE_CHECKING
6
6
 
7
7
  from drain3 import TemplateMiner
8
+ from drain3.masking import MaskingInstruction
8
9
  from drain3.template_miner_config import TemplateMinerConfig
9
10
  from loguru import logger
10
11
 
@@ -33,6 +34,45 @@ class LogPattern:
33
34
  metadata: dict = field(default_factory=dict)
34
35
 
35
36
 
37
+ def _typed_masking_instructions() -> list[MaskingInstruction]:
38
+ """Build the curated masking set for typed template slots.
39
+
40
+ Drain3 applies these to each line before clustering, so masked values
41
+ become literal typed tokens (``<IP>``, ``<NUM>``, ...) rather than
42
+ collapsing to the bare ``<*>`` wildcard. Instructions run sequentially
43
+ over already-masked text, so the specific shapes (IP, UUID, MAC, IPv6)
44
+ precede the greedy catch-alls (HEX, NUM).
45
+ """
46
+ # Zero-width token boundaries (drain3's documented idiom): mask a value
47
+ # delimited by non-alphanumerics, never a substring of a word.
48
+ start = r"(?:(?<=[^A-Za-z0-9])|^)"
49
+ end = r"(?:(?=[^A-Za-z0-9])|$)"
50
+ hex4 = r"[0-9A-Fa-f]{1,4}"
51
+ shapes = [
52
+ # IPv4 dotted quad.
53
+ (r"\d{1,3}(?:\.\d{1,3}){3}", "IP"),
54
+ # UUID before HEX, which would otherwise eat its 8/12-char runs.
55
+ (r"[0-9A-Fa-f]{8}-[0-9A-Fa-f]{4}-[0-9A-Fa-f]{4}-[0-9A-Fa-f]{4}-[0-9A-Fa-f]{12}", "UUID"),
56
+ # MAC (colon or dash separated) before IPv6, whose colon-hex shape overlaps.
57
+ (r"[0-9A-Fa-f]{2}(?:[:-][0-9A-Fa-f]{2}){5}", "MAC"),
58
+ # IPv6, best-effort: the full 8-group form, a '::'-compressed form
59
+ # (middle or trailing), then a leading-'::' form. Plain colon runs
60
+ # without '::' stay unmasked - that shape also matches timestamps.
61
+ (
62
+ rf"(?:{hex4}:){{7}}{hex4}"
63
+ rf"|(?:{hex4}:){{1,6}}:(?:{hex4}(?::{hex4}){{0,5}})?"
64
+ rf"|::(?:{hex4}(?::{hex4}){{0,6}})?",
65
+ "IPV6",
66
+ ),
67
+ # Hex token of >= 8 hex chars: 0x-prefixed, or containing at least
68
+ # one a-f letter so a long pure-decimal token stays NUM.
69
+ (r"0[xX][0-9A-Fa-f]{8,}|(?=\d*[A-Fa-f])[0-9A-Fa-f]{8,}", "HEX"),
70
+ # Integer (drain3's documented NUM example).
71
+ (r"[-+]?\d+", "NUM"),
72
+ ]
73
+ return [MaskingInstruction(f"{start}(?:{pattern}){end}", name) for pattern, name in shapes]
74
+
75
+
36
76
  class PatternExtractor:
37
77
  """Extract patterns using Drain3's online template miner."""
38
78
 
@@ -51,6 +91,10 @@ class PatternExtractor:
51
91
  # Bound the template store when configured: Drain3 LRU-evicts beyond
52
92
  # drain_max_clusters, keeping memory flat on high-cardinality logs.
53
93
  drain_config.drain_max_clusters = self.config.max_clusters
94
+ # Opt-in typed masking: pre-cluster masking of well-known value shapes
95
+ # so templates carry typed slots (<IP>, <NUM>, ...) instead of <*>.
96
+ if self.config.typed_masking:
97
+ drain_config.masking_instructions = _typed_masking_instructions()
54
98
  self.miner = TemplateMiner(config=drain_config)
55
99
 
56
100
  def extract_patterns(self, lines: Iterable[str]) -> list[LogPattern]:
File without changes
File without changes
File without changes