logometer 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- logometer/__init__.py +3 -0
- logometer/baseline.py +54 -0
- logometer/classifier.py +71 -0
- logometer/cli.py +447 -0
- logometer/detector.py +70 -0
- logometer/explain.py +185 -0
- logometer/pretty.py +52 -0
- logometer/timeparse.py +58 -0
- logometer/windower.py +247 -0
- logometer-0.1.0.dist-info/METADATA +294 -0
- logometer-0.1.0.dist-info/RECORD +15 -0
- logometer-0.1.0.dist-info/WHEEL +5 -0
- logometer-0.1.0.dist-info/entry_points.txt +2 -0
- logometer-0.1.0.dist-info/licenses/LICENSE +21 -0
- logometer-0.1.0.dist-info/top_level.txt +1 -0
logometer/__init__.py
ADDED
logometer/baseline.py
ADDED
|
@@ -0,0 +1,54 @@
|
|
|
1
|
+
"""
|
|
2
|
+
A small rolling baseline: keeps the last N observed values for a metric
|
|
3
|
+
(e.g. error count per window) and reports the mean/std of that history.
|
|
4
|
+
|
|
5
|
+
Deliberately simple — a moving window of raw values, not an
|
|
6
|
+
exponentially-weighted or Bayesian estimator. Easy to explain, easy to
|
|
7
|
+
reason about when it gets something wrong.
|
|
8
|
+
"""
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import statistics
|
|
12
|
+
from collections import deque
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
class RollingBaseline:
|
|
16
|
+
def __init__(self, history_size: int = 20):
|
|
17
|
+
"""Keep a fixed-size deque of recent metric samples (e.g. errors per window).
|
|
18
|
+
Older values drop off automatically when history_size is exceeded."""
|
|
19
|
+
self._values: deque[float] = deque(maxlen=history_size)
|
|
20
|
+
|
|
21
|
+
def update(self, value: float) -> None:
|
|
22
|
+
"""Append one observation; drop the oldest when history exceeds history_size.
|
|
23
|
+
Called after each non-spike window in AnomalyDetector.evaluate."""
|
|
24
|
+
self._values.append(value)
|
|
25
|
+
|
|
26
|
+
@property
|
|
27
|
+
def ready(self) -> bool:
|
|
28
|
+
"""True once there are enough samples to compare against (>= 3).
|
|
29
|
+
Until ready, deviation_score returns 0 and spikes are not scored."""
|
|
30
|
+
return len(self._values) >= 3
|
|
31
|
+
|
|
32
|
+
@property
|
|
33
|
+
def mean(self) -> float:
|
|
34
|
+
"""Arithmetic mean of stored samples, or 0.0 if history is empty.
|
|
35
|
+
Shown in CLI output as the approximate errors-per-window baseline."""
|
|
36
|
+
if not self._values:
|
|
37
|
+
return 0.0
|
|
38
|
+
return statistics.fmean(self._values)
|
|
39
|
+
|
|
40
|
+
@property
|
|
41
|
+
def stdev(self) -> float:
|
|
42
|
+
"""Population standard deviation of stored samples, or 0.0 if fewer than two.
|
|
43
|
+
Used with mean inside deviation_score for spike detection."""
|
|
44
|
+
if len(self._values) < 2:
|
|
45
|
+
return 0.0
|
|
46
|
+
return statistics.pstdev(self._values)
|
|
47
|
+
|
|
48
|
+
def deviation_score(self, value: float, min_stdev: float = 1.0) -> float:
|
|
49
|
+
"""Return how many std devs above the mean `value` is (0 if not ready).
|
|
50
|
+
Uses max(stdev, min_stdev) so flat histories do not over-flag single events."""
|
|
51
|
+
if not self.ready:
|
|
52
|
+
return 0.0
|
|
53
|
+
std = max(self.stdev, min_stdev)
|
|
54
|
+
return (value - self.mean) / std
|
logometer/classifier.py
ADDED
|
@@ -0,0 +1,71 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Classifies a single log line into a severity level, and produces a
|
|
3
|
+
"shape" fingerprint for error/warn lines so that structurally similar
|
|
4
|
+
messages (same error, different id/timestamp/number) collapse into the
|
|
5
|
+
same bucket.
|
|
6
|
+
|
|
7
|
+
This is intentionally simple regex/heuristic matching, not a trained
|
|
8
|
+
model — the goal is to be transparent and predictable, not clever.
|
|
9
|
+
"""
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
import re
|
|
13
|
+
from dataclasses import dataclass
|
|
14
|
+
|
|
15
|
+
# Checked in this order — first match wins. ERROR-ish tokens must be
|
|
16
|
+
# checked before WARN/INFO so a line like "ERROR: retrying after warning"
|
|
17
|
+
# is classified as ERROR, not WARN.
|
|
18
|
+
_LEVEL_PATTERNS = [
|
|
19
|
+
("ERROR", re.compile(r"\b(error|err|fatal|critical|exception|traceback|panic)\b", re.IGNORECASE)),
|
|
20
|
+
("WARN", re.compile(r"\b(warn|warning)\b", re.IGNORECASE)),
|
|
21
|
+
("INFO", re.compile(r"\b(info|notice)\b", re.IGNORECASE)),
|
|
22
|
+
("DEBUG", re.compile(r"\b(debug|trace)\b", re.IGNORECASE)),
|
|
23
|
+
]
|
|
24
|
+
|
|
25
|
+
# Patterns stripped out (in order) when building a fingerprint, so that
|
|
26
|
+
# two error lines differing only in a request id / timestamp / number
|
|
27
|
+
# are recognized as "the same shape".
|
|
28
|
+
_FINGERPRINT_STRIPS = [
|
|
29
|
+
re.compile(r"\b[0-9a-fA-F]{8}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{12}\b"), # UUID
|
|
30
|
+
re.compile(r"\b0x[0-9a-fA-F]+\b"), # hex addresses
|
|
31
|
+
re.compile(r"\b\d{4}-\d{2}-\d{2}[T ]\d{2}:\d{2}:\d{2}(?:[.,]\d+)?(?:Z|[+-]\d{2}:?\d{2})?\b"), # ISO timestamp (Python logging uses a comma before millis)
|
|
32
|
+
re.compile(r'"[^"]*"'), # quoted strings
|
|
33
|
+
re.compile(r"'[^']*'"), # single-quoted strings
|
|
34
|
+
re.compile(r"\d+"), # numbers, including ones glued to units/words (e.g. "313ms", "Errno104")
|
|
35
|
+
]
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
@dataclass(frozen=True)
|
|
39
|
+
class ClassifiedLine:
|
|
40
|
+
raw: str
|
|
41
|
+
level: str # ERROR / WARN / INFO / DEBUG / UNKNOWN
|
|
42
|
+
shape: str | None # only set for ERROR / WARN lines; None otherwise
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def classify_level(line: str) -> str:
|
|
46
|
+
"""Tag a log line with ERROR/WARN/INFO/DEBUG via regex; UNKNOWN if no match.
|
|
47
|
+
Patterns are checked in priority order so ERROR wins over WARN on the same line."""
|
|
48
|
+
for level, pattern in _LEVEL_PATTERNS:
|
|
49
|
+
if pattern.search(line):
|
|
50
|
+
return level
|
|
51
|
+
return "UNKNOWN"
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def fingerprint(line: str) -> str:
|
|
55
|
+
"""Normalize a line to an error/warn shape by stripping ids, numbers, and quotes.
|
|
56
|
+
Similar messages with different volatile tokens collapse to the same string."""
|
|
57
|
+
text = line.strip()
|
|
58
|
+
for pattern in _FINGERPRINT_STRIPS:
|
|
59
|
+
text = pattern.sub("<x>", text)
|
|
60
|
+
# collapse repeated whitespace and cap length so pathologically long
|
|
61
|
+
# lines don't blow up memory in the "seen shapes" set
|
|
62
|
+
text = re.sub(r"\s+", " ", text).strip()
|
|
63
|
+
return text[:200]
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
def classify_line(line: str) -> ClassifiedLine:
|
|
67
|
+
"""Classify severity and attach a shape fingerprint for ERROR/WARN lines.
|
|
68
|
+
INFO/DEBUG/UNKNOWN lines get shape=None."""
|
|
69
|
+
level = classify_level(line)
|
|
70
|
+
shape = fingerprint(line) if level in ("ERROR", "WARN") else None
|
|
71
|
+
return ClassifiedLine(raw=line, level=level, shape=shape)
|
logometer/cli.py
ADDED
|
@@ -0,0 +1,447 @@
|
|
|
1
|
+
"""
|
|
2
|
+
logometer CLI — Day 1 scope: `logometer tail <file>`.
|
|
3
|
+
|
|
4
|
+
Pure stdlib (argparse, no typer/rich dependency) so the tool runs with
|
|
5
|
+
nothing but Python installed. --explain (LLM-powered) and prettier
|
|
6
|
+
rich-based output are Day 2 additions.
|
|
7
|
+
"""
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import argparse
|
|
11
|
+
import json
|
|
12
|
+
import os
|
|
13
|
+
import re
|
|
14
|
+
import signal
|
|
15
|
+
import sys
|
|
16
|
+
import threading
|
|
17
|
+
import time
|
|
18
|
+
from queue import Empty, Queue
|
|
19
|
+
from typing import Iterator, Pattern, TextIO
|
|
20
|
+
|
|
21
|
+
from . import __version__
|
|
22
|
+
from .detector import AnomalyDetector, AnomalyVerdict
|
|
23
|
+
from .explain import ExplainError, explain_window
|
|
24
|
+
from .pretty import HAS_RICH
|
|
25
|
+
from .windower import Window, WindowAggregator
|
|
26
|
+
|
|
27
|
+
_RED = "\033[91m"
|
|
28
|
+
_DIM = "\033[2m"
|
|
29
|
+
_YELLOW = "\033[93m"
|
|
30
|
+
_CYAN = "\033[96m"
|
|
31
|
+
_RESET = "\033[0m"
|
|
32
|
+
|
|
33
|
+
_POLL_INTERVAL_SECONDS = 0.5
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def _supports_color(stream: TextIO) -> bool:
|
|
37
|
+
return hasattr(stream, "isatty") and stream.isatty()
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def _flush(stream: TextIO) -> None:
|
|
41
|
+
"""Push buffered output out now. Never raises — a closed or exotic
|
|
42
|
+
stream must not take down a tailing run."""
|
|
43
|
+
try:
|
|
44
|
+
stream.flush()
|
|
45
|
+
except (AttributeError, ValueError, OSError):
|
|
46
|
+
pass
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def _iter_stdin() -> Iterator[str | None]:
|
|
50
|
+
"""Yield lines from stdin, plus a None "idle tick" whenever nothing
|
|
51
|
+
arrives for a poll interval, so a window can still close when the
|
|
52
|
+
upstream producer goes quiet.
|
|
53
|
+
|
|
54
|
+
A reader thread is used rather than select(): Python buffers stdin
|
|
55
|
+
internally, so select() on the file descriptor can report "no data"
|
|
56
|
+
while whole lines already sit in that buffer, waiting.
|
|
57
|
+
"""
|
|
58
|
+
queue: Queue[str | None] = Queue()
|
|
59
|
+
|
|
60
|
+
def reader() -> None:
|
|
61
|
+
try:
|
|
62
|
+
for line in sys.stdin:
|
|
63
|
+
queue.put(line)
|
|
64
|
+
finally:
|
|
65
|
+
queue.put(None) # sentinel: end of input
|
|
66
|
+
|
|
67
|
+
threading.Thread(target=reader, daemon=True).start()
|
|
68
|
+
|
|
69
|
+
while True:
|
|
70
|
+
try:
|
|
71
|
+
item = queue.get(timeout=_POLL_INTERVAL_SECONDS)
|
|
72
|
+
except Empty:
|
|
73
|
+
yield None # nothing upstream right now — let windows close
|
|
74
|
+
continue
|
|
75
|
+
if item is None:
|
|
76
|
+
return
|
|
77
|
+
yield item
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
def _iter_file_replay(path: str) -> Iterator[str]:
|
|
81
|
+
with open(path, "r", errors="replace") as f:
|
|
82
|
+
for line in f:
|
|
83
|
+
yield line
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
def _file_identity(path: str) -> tuple[int, int] | None:
|
|
87
|
+
"""(device, inode) for a path, or None if it isn't there right now."""
|
|
88
|
+
try:
|
|
89
|
+
st = os.stat(path)
|
|
90
|
+
except OSError:
|
|
91
|
+
return None
|
|
92
|
+
return (st.st_dev, st.st_ino)
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
def _iter_file_live(path: str, poll_interval: float = _POLL_INTERVAL_SECONDS) -> Iterator[str | None]:
|
|
96
|
+
"""Classic tail -f polling loop: seek to EOF, then poll for appended
|
|
97
|
+
lines, yielding None while idle so windows can close on time.
|
|
98
|
+
|
|
99
|
+
Also survives log rotation. Holding one file handle forever means
|
|
100
|
+
that after logrotate moves the file aside, we keep reading a
|
|
101
|
+
now-orphaned inode and silently see nothing ever again — so each
|
|
102
|
+
idle poll checks whether the path points somewhere new (rotated) or
|
|
103
|
+
has shrunk (truncated in place) and reopens accordingly.
|
|
104
|
+
|
|
105
|
+
Opened in binary mode on purpose: it gives an exact byte position, which
|
|
106
|
+
is what makes the truncation check reliable, and it lets a half-written
|
|
107
|
+
line (one whose newline hasn't been flushed yet) be held back instead of
|
|
108
|
+
handed over as though it were complete.
|
|
109
|
+
"""
|
|
110
|
+
f = open(path, "rb")
|
|
111
|
+
try:
|
|
112
|
+
f.seek(0, 2) # seek to end
|
|
113
|
+
identity = _file_identity(path)
|
|
114
|
+
|
|
115
|
+
while True:
|
|
116
|
+
chunk = f.readline()
|
|
117
|
+
if chunk.endswith(b"\n"):
|
|
118
|
+
yield chunk.decode("utf-8", errors="replace")
|
|
119
|
+
continue
|
|
120
|
+
if chunk:
|
|
121
|
+
# partial line — rewind and wait for the writer to finish it
|
|
122
|
+
f.seek(-len(chunk), os.SEEK_CUR)
|
|
123
|
+
|
|
124
|
+
time.sleep(poll_interval)
|
|
125
|
+
|
|
126
|
+
current = _file_identity(path)
|
|
127
|
+
if current is None:
|
|
128
|
+
yield None # mid-rotation; the path will be back shortly
|
|
129
|
+
continue
|
|
130
|
+
if current != identity:
|
|
131
|
+
f.close()
|
|
132
|
+
f = open(path, "rb") # new file: read it from the top
|
|
133
|
+
identity = current
|
|
134
|
+
elif os.stat(path).st_size < f.tell():
|
|
135
|
+
# truncated in place (copytruncate). Caveat: if the file is
|
|
136
|
+
# emptied and then refilled past our read offset inside one
|
|
137
|
+
# poll interval, the shrink is never observable by size and
|
|
138
|
+
# we resume mid-file — the same blind spot GNU tail has.
|
|
139
|
+
f.seek(0)
|
|
140
|
+
|
|
141
|
+
yield None
|
|
142
|
+
finally:
|
|
143
|
+
f.close()
|
|
144
|
+
|
|
145
|
+
|
|
146
|
+
def _shape_levels(window: Window) -> dict[str, str]:
|
|
147
|
+
"""Map each shape in a window back to the level of the line that produced it,
|
|
148
|
+
so a WARN-triggered anomaly isn't reported as an 'error signature'."""
|
|
149
|
+
return {l.shape: l.level for l in window.lines if l.shape}
|
|
150
|
+
|
|
151
|
+
|
|
152
|
+
def _signature_label(level: str) -> str:
|
|
153
|
+
return "warning" if level == "WARN" else "error"
|
|
154
|
+
|
|
155
|
+
|
|
156
|
+
def _format_plain(window: Window, verdict: AnomalyVerdict, color: bool, explanation: str | None = None) -> str:
|
|
157
|
+
label = f"[{window.start_label} \u2013 {window.end_label}]"
|
|
158
|
+
if verdict.is_anomaly:
|
|
159
|
+
marker = f"{_RED}\u26a0 ANOMALY{_RESET}" if color else "! ANOMALY"
|
|
160
|
+
# A new WARN shape can flag a window that holds zero errors \u2014
|
|
161
|
+
# show the warn count too, or the header reads "errors: 0" with
|
|
162
|
+
# no hint of what actually tripped the detector.
|
|
163
|
+
counts = f"errors: {window.error_count}"
|
|
164
|
+
if window.warn_count:
|
|
165
|
+
counts += f" warns: {window.warn_count}"
|
|
166
|
+
header = (
|
|
167
|
+
f" {label} {marker} {counts} "
|
|
168
|
+
f"baseline: ~{verdict.baseline_mean:.1f}"
|
|
169
|
+
+ (f" score: {verdict.error_score:.1f}x" if verdict.error_score else "")
|
|
170
|
+
)
|
|
171
|
+
lines = [header]
|
|
172
|
+
levels = _shape_levels(window)
|
|
173
|
+
for shape in verdict.new_shapes:
|
|
174
|
+
kind = _signature_label(levels.get(shape, "ERROR"))
|
|
175
|
+
note = f" New {kind} signature detected: {shape!r}"
|
|
176
|
+
lines.append(f"{_YELLOW}{note}{_RESET}" if color else note)
|
|
177
|
+
if not verdict.new_shapes and not explanation:
|
|
178
|
+
lines.append(" (error rate spike \u2014 no brand-new error signature)")
|
|
179
|
+
if explanation:
|
|
180
|
+
note = f" Explanation: {explanation}"
|
|
181
|
+
lines.append(f"{_CYAN}{note}{_RESET}" if color else note)
|
|
182
|
+
return "\n".join(lines)
|
|
183
|
+
else:
|
|
184
|
+
text = f" {label} ok errors: {window.error_count} baseline: ~{verdict.baseline_mean:.1f}"
|
|
185
|
+
return f"{_DIM}{text}{_RESET}" if color else text
|
|
186
|
+
|
|
187
|
+
|
|
188
|
+
def _format_json(window: Window, verdict: AnomalyVerdict, explanation: str | None = None) -> str:
|
|
189
|
+
payload = {
|
|
190
|
+
"window_index": window.index,
|
|
191
|
+
"start": window.start_label,
|
|
192
|
+
"end": window.end_label,
|
|
193
|
+
"error_count": window.error_count,
|
|
194
|
+
"warn_count": window.warn_count,
|
|
195
|
+
"baseline_mean": round(verdict.baseline_mean, 3),
|
|
196
|
+
"error_score": round(verdict.error_score, 3),
|
|
197
|
+
"is_anomaly": verdict.is_anomaly,
|
|
198
|
+
"new_shapes": verdict.new_shapes,
|
|
199
|
+
"explanation": explanation,
|
|
200
|
+
}
|
|
201
|
+
return json.dumps(payload)
|
|
202
|
+
|
|
203
|
+
|
|
204
|
+
def run_tail(
|
|
205
|
+
source: Iterator[str | None],
|
|
206
|
+
window_seconds: float,
|
|
207
|
+
sensitivity: str,
|
|
208
|
+
output_format: str,
|
|
209
|
+
quiet: bool,
|
|
210
|
+
explain: bool = False,
|
|
211
|
+
explain_provider: str = "anthropic",
|
|
212
|
+
explain_model: str | None = None,
|
|
213
|
+
use_rich: bool = False,
|
|
214
|
+
ignore: list[Pattern[str]] | None = None,
|
|
215
|
+
out: TextIO = sys.stdout,
|
|
216
|
+
) -> int:
|
|
217
|
+
"""Core loop shared by file/stdin, replay/live. Returns an exit code."""
|
|
218
|
+
aggregator = WindowAggregator(window_seconds=window_seconds)
|
|
219
|
+
detector = AnomalyDetector(sensitivity=sensitivity)
|
|
220
|
+
color = output_format == "plain" and _supports_color(out)
|
|
221
|
+
anomaly_count = 0
|
|
222
|
+
window_count = 0
|
|
223
|
+
explain_warned = False # only print an --explain setup problem once, not per-anomaly
|
|
224
|
+
|
|
225
|
+
console = None
|
|
226
|
+
if use_rich and output_format == "plain" and HAS_RICH and _supports_color(out):
|
|
227
|
+
from rich.console import Console
|
|
228
|
+
console = Console(file=out)
|
|
229
|
+
|
|
230
|
+
def get_explanation(window: Window) -> str | None:
|
|
231
|
+
nonlocal explain_warned
|
|
232
|
+
if not explain:
|
|
233
|
+
return None
|
|
234
|
+
try:
|
|
235
|
+
raw_lines = [l.raw for l in window.lines if l.level in ("ERROR", "WARN")]
|
|
236
|
+
return explain_window(raw_lines or [l.raw for l in window.lines], provider=explain_provider, model=explain_model)
|
|
237
|
+
except ExplainError as e:
|
|
238
|
+
if not explain_warned:
|
|
239
|
+
print(f"[logometer] --explain unavailable: {e}", file=sys.stderr)
|
|
240
|
+
explain_warned = True
|
|
241
|
+
return None
|
|
242
|
+
|
|
243
|
+
def handle(window: Window) -> None:
|
|
244
|
+
nonlocal anomaly_count, window_count
|
|
245
|
+
if not window.lines:
|
|
246
|
+
return
|
|
247
|
+
window_count += 1
|
|
248
|
+
verdict = detector.evaluate(window)
|
|
249
|
+
if verdict.is_anomaly:
|
|
250
|
+
anomaly_count += 1
|
|
251
|
+
if quiet and not verdict.is_anomaly:
|
|
252
|
+
return
|
|
253
|
+
|
|
254
|
+
explanation = get_explanation(window) if verdict.is_anomaly else None
|
|
255
|
+
|
|
256
|
+
if output_format == "json":
|
|
257
|
+
print(_format_json(window, verdict, explanation), file=out)
|
|
258
|
+
elif console is not None:
|
|
259
|
+
from .pretty import render_window
|
|
260
|
+
render_window(window, verdict, explanation, console)
|
|
261
|
+
else:
|
|
262
|
+
print(_format_plain(window, verdict, color, explanation), file=out)
|
|
263
|
+
print(file=out)
|
|
264
|
+
|
|
265
|
+
# Python block-buffers stdout when it isn't a terminal, so without
|
|
266
|
+
# this an alert can sit in memory for minutes when the output is
|
|
267
|
+
# redirected to a file or piped into something else — the exact
|
|
268
|
+
# setups a monitoring tool actually runs in.
|
|
269
|
+
_flush(out)
|
|
270
|
+
|
|
271
|
+
try:
|
|
272
|
+
for raw_line in source:
|
|
273
|
+
if raw_line is None:
|
|
274
|
+
# idle tick from a live source: close anything overdue
|
|
275
|
+
for window in aggregator.tick():
|
|
276
|
+
handle(window)
|
|
277
|
+
continue
|
|
278
|
+
if ignore and any(p.search(raw_line) for p in ignore):
|
|
279
|
+
continue
|
|
280
|
+
for window in aggregator.feed(raw_line):
|
|
281
|
+
handle(window)
|
|
282
|
+
except KeyboardInterrupt:
|
|
283
|
+
pass
|
|
284
|
+
finally:
|
|
285
|
+
for window in aggregator.flush():
|
|
286
|
+
handle(window)
|
|
287
|
+
|
|
288
|
+
if output_format != "json":
|
|
289
|
+
summary = f"-- {window_count} window(s) processed, {anomaly_count} anomaly(ies) flagged --"
|
|
290
|
+
if console is not None:
|
|
291
|
+
console.print(summary, style="dim")
|
|
292
|
+
else:
|
|
293
|
+
print(summary, file=out)
|
|
294
|
+
_flush(out)
|
|
295
|
+
|
|
296
|
+
return 0
|
|
297
|
+
|
|
298
|
+
|
|
299
|
+
def build_parser() -> argparse.ArgumentParser:
|
|
300
|
+
parser = argparse.ArgumentParser(
|
|
301
|
+
prog="logometer",
|
|
302
|
+
description="Tail your logs. Catch anomalies. Skip the 2am grep.",
|
|
303
|
+
)
|
|
304
|
+
parser.add_argument("--version", action="version", version=f"logometer {__version__}")
|
|
305
|
+
|
|
306
|
+
subparsers = parser.add_subparsers(dest="command", required=True)
|
|
307
|
+
|
|
308
|
+
tail_parser = subparsers.add_parser("tail", help="Tail a log file (or stdin) and flag anomalous windows")
|
|
309
|
+
tail_parser.add_argument(
|
|
310
|
+
"file",
|
|
311
|
+
help="Path to a log file, or '-' to read from stdin",
|
|
312
|
+
)
|
|
313
|
+
tail_parser.add_argument(
|
|
314
|
+
"--window",
|
|
315
|
+
type=float,
|
|
316
|
+
default=10.0,
|
|
317
|
+
metavar="SECONDS",
|
|
318
|
+
help="Window size in seconds when timestamps are parseable (default: 10)",
|
|
319
|
+
)
|
|
320
|
+
tail_parser.add_argument(
|
|
321
|
+
"--sensitivity",
|
|
322
|
+
choices=["low", "medium", "high"],
|
|
323
|
+
default="medium",
|
|
324
|
+
help="How aggressively to flag deviations (default: medium)",
|
|
325
|
+
)
|
|
326
|
+
tail_parser.add_argument(
|
|
327
|
+
"--format",
|
|
328
|
+
choices=["plain", "json"],
|
|
329
|
+
default="plain",
|
|
330
|
+
dest="output_format",
|
|
331
|
+
help="Output format (default: plain)",
|
|
332
|
+
)
|
|
333
|
+
tail_parser.add_argument(
|
|
334
|
+
"--replay",
|
|
335
|
+
action="store_true",
|
|
336
|
+
help="Read a file from start to end and exit, instead of live-tailing it",
|
|
337
|
+
)
|
|
338
|
+
tail_parser.add_argument(
|
|
339
|
+
"--quiet",
|
|
340
|
+
action="store_true",
|
|
341
|
+
help="Only print anomalous windows, suppress 'ok' windows",
|
|
342
|
+
)
|
|
343
|
+
tail_parser.add_argument(
|
|
344
|
+
"--ignore",
|
|
345
|
+
action="append",
|
|
346
|
+
default=None,
|
|
347
|
+
metavar="REGEX",
|
|
348
|
+
dest="ignore",
|
|
349
|
+
help="Drop lines matching this regex before any analysis — use it to mute "
|
|
350
|
+
"known-noisy messages that would otherwise dominate the baseline. "
|
|
351
|
+
"Repeatable: --ignore 'healthcheck' --ignore 'DeprecationWarning'",
|
|
352
|
+
)
|
|
353
|
+
tail_parser.add_argument(
|
|
354
|
+
"--explain",
|
|
355
|
+
action="store_true",
|
|
356
|
+
help="Ask an LLM for a one-sentence explanation of each anomaly "
|
|
357
|
+
"(requires ANTHROPIC_API_KEY, OPENAI_API_KEY or OPENROUTER_API_KEY; fails soft if unset)",
|
|
358
|
+
)
|
|
359
|
+
tail_parser.add_argument(
|
|
360
|
+
"--explain-provider",
|
|
361
|
+
choices=["anthropic", "openai", "openrouter"],
|
|
362
|
+
default="anthropic",
|
|
363
|
+
help="Which API to use for --explain (default: anthropic)",
|
|
364
|
+
)
|
|
365
|
+
tail_parser.add_argument(
|
|
366
|
+
"--explain-model",
|
|
367
|
+
default=None,
|
|
368
|
+
metavar="MODEL",
|
|
369
|
+
help="Override the default model used for --explain",
|
|
370
|
+
)
|
|
371
|
+
tail_parser.add_argument(
|
|
372
|
+
"--no-pretty",
|
|
373
|
+
action="store_true",
|
|
374
|
+
help="Disable rich-styled output even if the 'rich' package is installed",
|
|
375
|
+
)
|
|
376
|
+
|
|
377
|
+
return parser
|
|
378
|
+
|
|
379
|
+
|
|
380
|
+
def _raise_keyboard_interrupt(signum, frame):
|
|
381
|
+
raise KeyboardInterrupt()
|
|
382
|
+
|
|
383
|
+
|
|
384
|
+
def main(argv: list[str] | None = None) -> int:
|
|
385
|
+
# Treat SIGTERM the same as Ctrl-C (SIGINT) so a live-tailing process
|
|
386
|
+
# shuts down cleanly (flushing its in-progress window) whether it's
|
|
387
|
+
# interrupted at the terminal or killed by a process manager/`timeout`.
|
|
388
|
+
signal.signal(signal.SIGTERM, _raise_keyboard_interrupt)
|
|
389
|
+
|
|
390
|
+
parser = build_parser()
|
|
391
|
+
args = parser.parse_args(argv)
|
|
392
|
+
|
|
393
|
+
if args.command == "tail":
|
|
394
|
+
ignore = []
|
|
395
|
+
for raw_pattern in args.ignore or []:
|
|
396
|
+
try:
|
|
397
|
+
ignore.append(re.compile(raw_pattern))
|
|
398
|
+
except re.error as e:
|
|
399
|
+
parser.error(f"invalid --ignore regex {raw_pattern!r}: {e}")
|
|
400
|
+
|
|
401
|
+
if args.file == "-":
|
|
402
|
+
source = _iter_stdin()
|
|
403
|
+
else:
|
|
404
|
+
# Check readability up front so a typo'd path produces one clear
|
|
405
|
+
# line instead of a traceback out of the middle of a generator.
|
|
406
|
+
if not os.path.exists(args.file):
|
|
407
|
+
print(f"[logometer] no such file: {args.file}", file=sys.stderr)
|
|
408
|
+
return 2
|
|
409
|
+
if os.path.isdir(args.file):
|
|
410
|
+
print(f"[logometer] {args.file} is a directory, not a log file", file=sys.stderr)
|
|
411
|
+
return 2
|
|
412
|
+
try:
|
|
413
|
+
open(args.file, "rb").close()
|
|
414
|
+
except OSError as e:
|
|
415
|
+
print(f"[logometer] cannot read {args.file}: {e.strerror}", file=sys.stderr)
|
|
416
|
+
return 2
|
|
417
|
+
source = _iter_file_replay(args.file) if args.replay else _iter_file_live(args.file)
|
|
418
|
+
|
|
419
|
+
try:
|
|
420
|
+
return run_tail(
|
|
421
|
+
source=source,
|
|
422
|
+
window_seconds=args.window,
|
|
423
|
+
sensitivity=args.sensitivity,
|
|
424
|
+
output_format=args.output_format,
|
|
425
|
+
quiet=args.quiet,
|
|
426
|
+
explain=args.explain,
|
|
427
|
+
explain_provider=args.explain_provider,
|
|
428
|
+
explain_model=args.explain_model,
|
|
429
|
+
use_rich=not args.no_pretty,
|
|
430
|
+
ignore=ignore,
|
|
431
|
+
)
|
|
432
|
+
except OSError as e:
|
|
433
|
+
# e.g. the file is deleted for good while we're live-tailing it
|
|
434
|
+
print(f"[logometer] stopped reading {args.file}: {e}", file=sys.stderr)
|
|
435
|
+
return 1
|
|
436
|
+
|
|
437
|
+
parser.error(f"unknown command {args.command!r}")
|
|
438
|
+
return 2
|
|
439
|
+
|
|
440
|
+
|
|
441
|
+
if __name__ == "__main__":
|
|
442
|
+
try:
|
|
443
|
+
sys.exit(main())
|
|
444
|
+
except BrokenPipeError:
|
|
445
|
+
# e.g. piped into `head` — exit quietly instead of a traceback
|
|
446
|
+
sys.stderr.close()
|
|
447
|
+
sys.exit(0)
|
logometer/detector.py
ADDED
|
@@ -0,0 +1,70 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Turns a stream of Windows into a stream of (Window, AnomalyVerdict) pairs.
|
|
3
|
+
|
|
4
|
+
A window is flagged anomalous if either:
|
|
5
|
+
(a) its error count deviates from the rolling baseline by more than
|
|
6
|
+
`sensitivity` standard deviations, or
|
|
7
|
+
(b) it contains an error "shape" that has never been seen before in
|
|
8
|
+
this run (and we're past the warm-up period).
|
|
9
|
+
"""
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
from dataclasses import dataclass, field
|
|
13
|
+
|
|
14
|
+
from .baseline import RollingBaseline
|
|
15
|
+
from .windower import Window
|
|
16
|
+
|
|
17
|
+
SENSITIVITY_MULTIPLIERS = {
|
|
18
|
+
"low": 3.0, # only flag big, obvious spikes
|
|
19
|
+
"medium": 2.0, # default
|
|
20
|
+
"high": 1.2, # flag smaller deviations too — noisier
|
|
21
|
+
}
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
@dataclass
|
|
25
|
+
class AnomalyVerdict:
|
|
26
|
+
is_anomaly: bool
|
|
27
|
+
error_score: float # std-devs above baseline (0 if not ready/not anomalous)
|
|
28
|
+
new_shapes: list[str] = field(default_factory=list)
|
|
29
|
+
baseline_mean: float = 0.0
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
class AnomalyDetector:
|
|
33
|
+
def __init__(self, sensitivity: str = "medium", history_size: int = 20):
|
|
34
|
+
"""Configure spike threshold (low/medium/high) and rolling error-count baseline.
|
|
35
|
+
Also initializes the set of error shapes seen so far in this run."""
|
|
36
|
+
if sensitivity not in SENSITIVITY_MULTIPLIERS:
|
|
37
|
+
raise ValueError(f"unknown sensitivity {sensitivity!r}, expected one of {list(SENSITIVITY_MULTIPLIERS)}")
|
|
38
|
+
self.multiplier = SENSITIVITY_MULTIPLIERS[sensitivity]
|
|
39
|
+
self._error_baseline = RollingBaseline(history_size=history_size)
|
|
40
|
+
self._seen_shapes: set[str] = set()
|
|
41
|
+
self._warmed_up = False # true once we've built at least one baseline observation
|
|
42
|
+
|
|
43
|
+
def evaluate(self, window: Window) -> AnomalyVerdict:
|
|
44
|
+
"""Score one window for error spikes and unseen error shapes; update baseline state.
|
|
45
|
+
Volume spikes are omitted from baseline updates; new shapes still learn normal volume."""
|
|
46
|
+
error_count = window.error_count
|
|
47
|
+
score = self._error_baseline.deviation_score(error_count)
|
|
48
|
+
score_anomaly = self._error_baseline.ready and score >= self.multiplier
|
|
49
|
+
|
|
50
|
+
new_shapes = []
|
|
51
|
+
if self._warmed_up:
|
|
52
|
+
for shape in window.shapes:
|
|
53
|
+
if shape not in self._seen_shapes:
|
|
54
|
+
new_shapes.append(shape)
|
|
55
|
+
|
|
56
|
+
for shape in window.shapes:
|
|
57
|
+
self._seen_shapes.add(shape)
|
|
58
|
+
|
|
59
|
+
verdict = AnomalyVerdict(
|
|
60
|
+
is_anomaly=bool(score_anomaly or new_shapes),
|
|
61
|
+
error_score=score if score != float("inf") else 999.0,
|
|
62
|
+
new_shapes=new_shapes,
|
|
63
|
+
baseline_mean=self._error_baseline.mean,
|
|
64
|
+
)
|
|
65
|
+
|
|
66
|
+
if not score_anomaly:
|
|
67
|
+
self._error_baseline.update(error_count)
|
|
68
|
+
self._warmed_up = True
|
|
69
|
+
|
|
70
|
+
return verdict
|