logometer 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
logometer/explain.py ADDED
@@ -0,0 +1,185 @@
1
+ """
2
+ Optional "--explain" support: sends an anomalous window's lines to an
3
+ LLM and asks for a one-sentence plain-English summary of the likely
4
+ cause.
5
+
6
+ Providers: Anthropic, OpenAI, and OpenRouter (OpenAI-compatible, gives
7
+ access to many vendors' models with one key).
8
+
9
+ Deliberately built on stdlib `urllib` rather than the `anthropic` or
10
+ `openai` SDKs, so `--explain` needs nothing beyond an API key — no
11
+ extra pip install, no risk of drifting out of sync with an SDK's API
12
+ surface. This is a deviation from the original plan (which sketched
13
+ "Anthropic or OpenAI SDK"); seemed like the better trade for a tool
14
+ whose whole pitch is "works with nothing installed."
15
+
16
+ This module must NEVER raise anything except ExplainError — callers
17
+ (cli.py) rely on that to fail soft (print a note, keep tailing) rather
18
+ than crashing the whole run because an anomaly happened to coincide
19
+ with a flaky network.
20
+ """
21
+ from __future__ import annotations
22
+
23
+ import json
24
+ import os
25
+ import urllib.error
26
+ import urllib.request
27
+
28
+ _ANTHROPIC_URL = "https://api.anthropic.com/v1/messages"
29
+ _ANTHROPIC_MODEL = "claude-haiku-4-5-20251001" # fast + cheap, all we need for a one-liner
30
+
31
+ _OPENAI_URL = "https://api.openai.com/v1/chat/completions"
32
+ _OPENAI_MODEL = "gpt-4o-mini"
33
+
34
+ # OpenRouter speaks the OpenAI chat-completions format, so it shares
35
+ # that code path; model ids are namespaced, e.g. "openai/gpt-4o-mini".
36
+ _OPENROUTER_URL = "https://openrouter.ai/api/v1/chat/completions"
37
+ _OPENROUTER_MODEL = "anthropic/claude-haiku-4.5"
38
+
39
+ _TIMEOUT_SECONDS = 15
40
+ _MAX_LINES_SENT = 30 # cap what we send: a window shouldn't need more context than this to explain
41
+
42
+ _SYSTEM_PROMPT = (
43
+ "You are looking at a burst of anomalous lines from an application log. "
44
+ "In ONE short sentence (under 25 words), state the most likely underlying cause. "
45
+ "Be concrete and specific to what's in the lines. No preamble, no hedging, no markdown."
46
+ )
47
+
48
+
49
+ class ExplainError(Exception):
50
+ """Raised for any failure in getting an explanation — missing API
51
+ key, network failure, bad response, unknown provider. Callers
52
+ should catch this and degrade gracefully rather than crash."""
53
+
54
+
55
+ def _build_prompt(lines: list[str]) -> str:
56
+ sample = lines[:_MAX_LINES_SENT]
57
+ joined = "\n".join(sample)
58
+ if len(lines) > _MAX_LINES_SENT:
59
+ joined += f"\n... ({len(lines) - _MAX_LINES_SENT} more similar lines omitted)"
60
+ return f"Log lines:\n{joined}"
61
+
62
+
63
+ def _call_anthropic(lines: list[str], model: str) -> str:
64
+ api_key = os.environ.get("ANTHROPIC_API_KEY")
65
+ if not api_key:
66
+ raise ExplainError(
67
+ "ANTHROPIC_API_KEY is not set. Set it to enable --explain, "
68
+ "e.g. export ANTHROPIC_API_KEY=your-key-here"
69
+ )
70
+
71
+ body = json.dumps({
72
+ "model": model,
73
+ "max_tokens": 100,
74
+ "system": _SYSTEM_PROMPT,
75
+ "messages": [{"role": "user", "content": _build_prompt(lines)}],
76
+ }).encode("utf-8")
77
+
78
+ request = urllib.request.Request(
79
+ _ANTHROPIC_URL,
80
+ data=body,
81
+ method="POST",
82
+ headers={
83
+ "Content-Type": "application/json",
84
+ "x-api-key": api_key,
85
+ "anthropic-version": "2023-06-01",
86
+ },
87
+ )
88
+ try:
89
+ with urllib.request.urlopen(request, timeout=_TIMEOUT_SECONDS) as response:
90
+ payload = json.loads(response.read().decode("utf-8"))
91
+ except urllib.error.HTTPError as e:
92
+ detail = e.read().decode("utf-8", errors="replace")[:300]
93
+ raise ExplainError(f"Anthropic API returned {e.code}: {detail}") from e
94
+ except urllib.error.URLError as e:
95
+ raise ExplainError(f"could not reach Anthropic API: {e.reason}") from e
96
+ except TimeoutError as e:
97
+ raise ExplainError("Anthropic API request timed out") from e
98
+
99
+ try:
100
+ blocks = payload["content"]
101
+ text = "".join(b.get("text", "") for b in blocks if b.get("type") == "text").strip()
102
+ except (KeyError, TypeError) as e:
103
+ raise ExplainError(f"unexpected response shape from Anthropic API: {payload!r}") from e
104
+
105
+ if not text:
106
+ raise ExplainError("Anthropic API returned an empty explanation")
107
+ return text
108
+
109
+
110
+ def _call_openai_compatible(
111
+ lines: list[str], model: str, *, url: str, key_env: str, provider: str, label: str
112
+ ) -> str:
113
+ api_key = os.environ.get(key_env)
114
+ if not api_key:
115
+ raise ExplainError(
116
+ f"{key_env} is not set. Set it to enable --explain with --explain-provider {provider}, "
117
+ f"e.g. export {key_env}=your-key-here"
118
+ )
119
+
120
+ body = json.dumps({
121
+ "model": model,
122
+ "max_tokens": 100,
123
+ "messages": [
124
+ {"role": "system", "content": _SYSTEM_PROMPT},
125
+ {"role": "user", "content": _build_prompt(lines)},
126
+ ],
127
+ }).encode("utf-8")
128
+
129
+ request = urllib.request.Request(
130
+ url,
131
+ data=body,
132
+ method="POST",
133
+ headers={
134
+ "Content-Type": "application/json",
135
+ "Authorization": f"Bearer {api_key}",
136
+ },
137
+ )
138
+ try:
139
+ with urllib.request.urlopen(request, timeout=_TIMEOUT_SECONDS) as response:
140
+ payload = json.loads(response.read().decode("utf-8"))
141
+ except urllib.error.HTTPError as e:
142
+ detail = e.read().decode("utf-8", errors="replace")[:300]
143
+ raise ExplainError(f"{label} API returned {e.code}: {detail}") from e
144
+ except urllib.error.URLError as e:
145
+ raise ExplainError(f"could not reach {label} API: {e.reason}") from e
146
+ except TimeoutError as e:
147
+ raise ExplainError(f"{label} API request timed out") from e
148
+
149
+ try:
150
+ text = payload["choices"][0]["message"]["content"].strip()
151
+ except (KeyError, IndexError, TypeError, AttributeError) as e:
152
+ raise ExplainError(f"unexpected response shape from {label} API: {payload!r}") from e
153
+
154
+ if not text:
155
+ raise ExplainError(f"{label} API returned an empty explanation")
156
+ return text
157
+
158
+
159
+ def _call_openai(lines: list[str], model: str) -> str:
160
+ return _call_openai_compatible(
161
+ lines, model, url=_OPENAI_URL, key_env="OPENAI_API_KEY", provider="openai", label="OpenAI"
162
+ )
163
+
164
+
165
+ def _call_openrouter(lines: list[str], model: str) -> str:
166
+ return _call_openai_compatible(
167
+ lines, model, url=_OPENROUTER_URL, key_env="OPENROUTER_API_KEY", provider="openrouter", label="OpenRouter"
168
+ )
169
+
170
+
171
+ _PROVIDERS = {
172
+ "anthropic": (_call_anthropic, _ANTHROPIC_MODEL),
173
+ "openai": (_call_openai, _OPENAI_MODEL),
174
+ "openrouter": (_call_openrouter, _OPENROUTER_MODEL),
175
+ }
176
+
177
+
178
+ def explain_window(lines: list[str], provider: str = "anthropic", model: str | None = None) -> str:
179
+ """Get a one-sentence explanation for a burst of anomalous log
180
+ lines. Raises ExplainError on any failure — callers should catch
181
+ this and continue without the explanation rather than crash."""
182
+ if provider not in _PROVIDERS:
183
+ raise ExplainError(f"unknown --explain-provider {provider!r}, expected one of {list(_PROVIDERS)}")
184
+ fn, default_model = _PROVIDERS[provider]
185
+ return fn(lines, model or default_model)
logometer/pretty.py ADDED
@@ -0,0 +1,52 @@
1
+ """
2
+ Nicer terminal rendering using `rich`, used automatically when it's
3
+ installed and we're writing to a real terminal. Falls back to the
4
+ plain ANSI formatting in cli.py otherwise — this module is never
5
+ required, only opportunistic.
6
+ """
7
+ from __future__ import annotations
8
+
9
+ try:
10
+ from rich.console import Console
11
+ from rich.panel import Panel
12
+ from rich.text import Text
13
+ HAS_RICH = True
14
+ except ImportError:
15
+ HAS_RICH = False
16
+
17
+
18
+ def render_window(window, verdict, explanation: str | None, console: "Console") -> None:
19
+ """Render one window's verdict with rich. Only call this if
20
+ HAS_RICH is True."""
21
+ label = f"{window.start_label} \u2013 {window.end_label}"
22
+
23
+ if not verdict.is_anomaly:
24
+ text = Text(f" [{label}] ok errors: {window.error_count} baseline: ~{verdict.baseline_mean:.1f}")
25
+ text.stylize("dim")
26
+ console.print(text)
27
+ return
28
+
29
+ body = Text()
30
+ body.append(f"errors: {window.error_count} ", style="bold")
31
+ if window.warn_count:
32
+ # a first-seen WARN shape can flag a window with zero errors
33
+ body.append(f"warns: {window.warn_count} ", style="bold")
34
+ body.append(f"baseline: ~{verdict.baseline_mean:.1f}")
35
+ if verdict.error_score:
36
+ body.append(f" score: {verdict.error_score:.1f}x", style="bold red" if verdict.error_score >= 3 else "yellow")
37
+
38
+ if verdict.new_shapes:
39
+ body.append("\n")
40
+ levels = {l.shape: l.level for l in window.lines if l.shape}
41
+ for shape in verdict.new_shapes:
42
+ kind = "warning" if levels.get(shape) == "WARN" else "error"
43
+ body.append(f"\n New {kind} signature: ", style="yellow")
44
+ body.append(shape, style="italic")
45
+ elif not explanation:
46
+ body.append("\n\n (error rate spike \u2014 no brand-new error signature)", style="dim")
47
+
48
+ if explanation:
49
+ body.append("\n\n ")
50
+ body.append(explanation, style="cyan")
51
+
52
+ console.print(Panel(body, title=f"\u26a0 ANOMALY [{label}]", border_style="red", expand=False))
logometer/timeparse.py ADDED
@@ -0,0 +1,58 @@
1
+ """
2
+ Best-effort extraction of a timestamp from the start of a log line.
3
+
4
+ We only need this to decide how to bucket lines into windows. If a
5
+ line's timestamp can't be parsed, the caller falls back to a
6
+ line-count-based window instead of a time-based one — see windower.py.
7
+ """
8
+ from __future__ import annotations
9
+
10
+ import re
11
+ from datetime import datetime
12
+ from typing import Optional
13
+
14
+ # ISO 8601, e.g. "2026-09-17T12:03:10.123Z" or "2026-09-17 12:03:10"
15
+ _ISO_RE = re.compile(
16
+ r"(\d{4}-\d{2}-\d{2}[T ]\d{2}:\d{2}:\d{2}(?:\.\d+)?)(Z|[+-]\d{2}:?\d{2})?"
17
+ )
18
+
19
+ # Syslog-style, e.g. "Sep 17 12:03:10"
20
+ _SYSLOG_RE = re.compile(
21
+ r"\b([A-Z][a-z]{2}\s+\d{1,2}\s+\d{2}:\d{2}:\d{2})\b"
22
+ )
23
+
24
+ _SYSLOG_MONTHS = {
25
+ "Jan": 1, "Feb": 2, "Mar": 3, "Apr": 4, "May": 5, "Jun": 6,
26
+ "Jul": 7, "Aug": 8, "Sep": 9, "Oct": 10, "Nov": 11, "Dec": 12,
27
+ }
28
+
29
+
30
+ def parse_timestamp(line: str, assumed_year: Optional[int] = None) -> Optional[datetime]:
31
+ """Best-effort ISO or syslog timestamp extraction from a log line.
32
+ Returns None when no recognizable time is found (caller may use line-based windows)."""
33
+ m = _ISO_RE.search(line)
34
+ if m:
35
+ raw = m.group(1)
36
+ try:
37
+ # normalize a couple of common variants Python's fromisoformat
38
+ # is picky about
39
+ raw = raw.replace(" ", "T") if "T" not in raw else raw
40
+ return datetime.fromisoformat(raw)
41
+ except ValueError:
42
+ pass
43
+
44
+ m = _SYSLOG_RE.search(line)
45
+ if m:
46
+ try:
47
+ month_str, day_str, time_str = m.group(1).split(None, 2)
48
+ # handle "Sep 1 12:03:10" (double space) already collapsed by split
49
+ month = _SYSLOG_MONTHS.get(month_str)
50
+ if month is None:
51
+ return None
52
+ year = assumed_year or datetime.now().year
53
+ hh, mm, ss = (int(x) for x in time_str.split(":"))
54
+ return datetime(year, month, int(day_str), hh, mm, ss)
55
+ except (ValueError, IndexError):
56
+ return None
57
+
58
+ return None
logometer/windower.py ADDED
@@ -0,0 +1,247 @@
1
+ """
2
+ Groups a stream of classified log lines into fixed-size windows.
3
+
4
+ Two modes, chosen automatically:
5
+ - "timed": if timestamps can be parsed from the log lines, windows are
6
+ bucketed by wall-clock/log-clock time (--window seconds).
7
+ - "count": if no timestamps are found (first N lines all fail to
8
+ parse), windows are bucketed by a fixed number of lines instead.
9
+ This keeps the tool useful on logs with no recognizable timestamp
10
+ format, at the cost of windows not corresponding to a fixed time span.
11
+ """
12
+ from __future__ import annotations
13
+
14
+ import math
15
+ import time
16
+ from dataclasses import dataclass, field
17
+ from datetime import datetime
18
+ from typing import Iterator, Optional
19
+
20
+ from .classifier import ClassifiedLine, classify_line
21
+ from .timeparse import parse_timestamp
22
+
23
+ # how many leading lines we sample to decide timed vs. count mode
24
+ _PROBE_LINES = 20
25
+ # fraction of probe lines that must have a parseable timestamp to use timed mode
26
+ _PROBE_THRESHOLD = 0.5
27
+ # fallback window size (in lines) when no timestamps are parseable
28
+ _DEFAULT_LINES_PER_WINDOW = 50
29
+
30
+
31
+ @dataclass
32
+ class Window:
33
+ index: int
34
+ start_label: str # human-readable label for the window (time range or line range)
35
+ end_label: str
36
+ lines: list[ClassifiedLine] = field(default_factory=list)
37
+
38
+ @property
39
+ def error_count(self) -> int:
40
+ """Number of lines in this window classified as ERROR.
41
+ Fed to RollingBaseline as the primary spike metric."""
42
+ return sum(1 for l in self.lines if l.level == "ERROR")
43
+
44
+ @property
45
+ def warn_count(self) -> int:
46
+ """Number of lines in this window classified as WARN.
47
+ Included in JSON output; not used for baseline spike scoring."""
48
+ return sum(1 for l in self.lines if l.level == "WARN")
49
+
50
+ @property
51
+ def shapes(self) -> set[str]:
52
+ """Distinct error/warn fingerprints present in this window.
53
+ Compared against AnomalyDetector._seen_shapes for new-signature anomalies."""
54
+ return {l.shape for l in self.lines if l.shape}
55
+
56
+
57
+ class WindowAggregator:
58
+ """Feed classified lines in; get completed Window objects out.
59
+
60
+ Usage:
61
+ agg = WindowAggregator(window_seconds=10)
62
+ for line in source:
63
+ for window in agg.feed(line):
64
+ handle(window)
65
+ for window in agg.flush():
66
+ handle(window)
67
+ """
68
+
69
+ def __init__(self, window_seconds: float = 10.0, lines_per_window: int = _DEFAULT_LINES_PER_WINDOW):
70
+ """Create an aggregator; timed vs count mode is chosen after probing first lines.
71
+ window_seconds applies in timed mode; lines_per_window in count mode."""
72
+ self.window_seconds = window_seconds
73
+ self.lines_per_window = lines_per_window
74
+
75
+ self._mode: Optional[str] = None # "timed" or "count", decided after probing
76
+ self._probe_buffer: list[str] = []
77
+ self._probe_timestamps: list[Optional[datetime]] = []
78
+ self._probe_started_at: Optional[float] = None
79
+
80
+ self._current: Optional[Window] = None
81
+ self._current_bucket_key = None
82
+ self._window_index = 0
83
+ self._first_timestamp: Optional[datetime] = None
84
+ # real-time clock (not log time) for the in-progress window, so
85
+ # tick() can close it on schedule when the log goes silent
86
+ self._current_opened_at: Optional[float] = None
87
+
88
+ # -- mode detection -----------------------------------------------------
89
+
90
+ def _decide_mode(self) -> None:
91
+ """Pick timed windows if enough probe lines had timestamps, else count-based.
92
+ Sets _first_timestamp when entering timed mode."""
93
+ parseable = sum(1 for t in self._probe_timestamps if t is not None)
94
+ if self._probe_timestamps and parseable / len(self._probe_timestamps) >= _PROBE_THRESHOLD:
95
+ self._mode = "timed"
96
+ for t in self._probe_timestamps:
97
+ if t is not None:
98
+ self._first_timestamp = t
99
+ break
100
+ else:
101
+ self._mode = "count"
102
+
103
+ # -- public API -----------------------------------------------------
104
+
105
+ def feed(self, raw_line: str) -> Iterator[Window]:
106
+ """Ingest one raw log line; yield any windows that closed (often zero or one).
107
+ Buffers an initial probe batch before choosing timed vs count mode."""
108
+ raw_line = raw_line.rstrip("\n")
109
+ if not raw_line:
110
+ return
111
+
112
+ if self._mode is None:
113
+ if self._probe_started_at is None:
114
+ self._probe_started_at = time.monotonic()
115
+ self._probe_buffer.append(raw_line)
116
+ self._probe_timestamps.append(parse_timestamp(raw_line))
117
+ if len(self._probe_buffer) < _PROBE_LINES:
118
+ return # still probing, hold lines until mode is decided
119
+ self._decide_mode()
120
+ # replay the buffered probe lines now that we know the mode
121
+ buffered = self._probe_buffer
122
+ self._probe_buffer = []
123
+ for buffered_line in buffered:
124
+ yield from self._feed_decided(buffered_line)
125
+ return
126
+
127
+ yield from self._feed_decided(raw_line)
128
+
129
+ def _feed_decided(self, raw_line: str) -> Iterator[Window]:
130
+ """Classify a line, assign it to the current bucket, yield completed windows.
131
+ Requires _mode to already be timed or count."""
132
+ classified = classify_line(raw_line)
133
+
134
+ if self._mode == "timed":
135
+ ts = parse_timestamp(raw_line) or self._last_seen_timestamp()
136
+ bucket_key = math.floor((ts - self._first_timestamp).total_seconds() / self.window_seconds)
137
+ self._last_ts = ts
138
+ else:
139
+ bucket_key = None # computed below from line count
140
+
141
+ if self._current is None:
142
+ self._current = self._new_window(bucket_key)
143
+ self._current_bucket_key = bucket_key
144
+
145
+ if self._mode == "count":
146
+ if len(self._current.lines) >= self.lines_per_window:
147
+ yield self._current
148
+ self._window_index += 1
149
+ self._current = self._new_window(None)
150
+ else:
151
+ if bucket_key != self._current_bucket_key:
152
+ yield self._current
153
+ self._window_index += 1
154
+ self._current = self._new_window(bucket_key)
155
+ self._current_bucket_key = bucket_key
156
+
157
+ self._current.lines.append(classified)
158
+
159
+ def _last_seen_timestamp(self) -> datetime:
160
+ """Fallback clock when a line in timed mode has no parseable timestamp.
161
+ Reuses last seen time, else log start, else wall clock."""
162
+ return getattr(self, "_last_ts", self._first_timestamp or datetime.now())
163
+
164
+ def _new_window(self, bucket_key) -> Window:
165
+ """Allocate the next window with human-readable start/end labels.
166
+ Timed labels are HH:MM:SS ranges; count mode uses line number ranges."""
167
+ if self._mode == "timed" and bucket_key is not None:
168
+ start = self._first_timestamp.timestamp() + bucket_key * self.window_seconds
169
+ end = start + self.window_seconds
170
+ start_label = datetime.fromtimestamp(start).strftime("%H:%M:%S")
171
+ end_label = datetime.fromtimestamp(end).strftime("%H:%M:%S")
172
+ else:
173
+ start_n = self._window_index * self.lines_per_window + 1
174
+ end_n = start_n + self.lines_per_window - 1
175
+ start_label = f"line {start_n}"
176
+ end_label = f"line {end_n}"
177
+ self._current_opened_at = time.monotonic()
178
+ return Window(index=self._window_index, start_label=start_label, end_label=end_label)
179
+
180
+ def tick(self, now: Optional[float] = None) -> list[Window]:
181
+ """Close anything whose time is up, without needing a new line.
182
+
183
+ Feeding alone only closes a window when the *next* line arrives
184
+ in a later bucket. On a live tail that means a service which
185
+ errors and then dies never gets reported at all: the burst that
186
+ matters most stays buffered forever, because nothing follows it.
187
+ Callers poll this while idle so windows close on schedule.
188
+
189
+ Two things can be overdue:
190
+ - the in-progress window, once `window_seconds` of real time
191
+ have passed since it opened
192
+ - the probe batch itself, on a log too quiet to produce 20
193
+ lines — otherwise a low-volume service reports nothing ever
194
+
195
+ Real (monotonic) time is used deliberately, not log time: a log
196
+ whose clock is skewed or in another timezone would otherwise
197
+ close every window the moment it went idle.
198
+ """
199
+ now = time.monotonic() if now is None else now
200
+ closed: list[Window] = []
201
+
202
+ if self._mode is None:
203
+ waited = self._probe_started_at is not None and now - self._probe_started_at >= self.window_seconds
204
+ if not (self._probe_buffer and waited):
205
+ return []
206
+ self._decide_mode()
207
+ buffered = self._probe_buffer
208
+ self._probe_buffer = []
209
+ for buffered_line in buffered:
210
+ closed.extend(self._feed_decided(buffered_line))
211
+
212
+ if (
213
+ self._current is not None
214
+ and self._current.lines
215
+ and self._current_opened_at is not None
216
+ and now - self._current_opened_at >= self.window_seconds
217
+ ):
218
+ closed.append(self._current)
219
+ self._current = None
220
+ self._current_opened_at = None
221
+ self._window_index += 1
222
+
223
+ return closed
224
+
225
+ def flush(self) -> list[Window]:
226
+ """Return every window still pending at EOF, oldest first.
227
+
228
+ Usually that's just the one in-progress window. But on a file
229
+ shorter than the probe batch, mode is decided here and the
230
+ buffered lines are replayed now — which can close several
231
+ windows at once. All of them must come back: an earlier version
232
+ kept only the last one, so a short log silently lost every
233
+ window but its final one (and with them the baseline history
234
+ the detector needs to flag anything at all).
235
+ """
236
+ pending: list[Window] = []
237
+ if self._mode is None and self._probe_buffer:
238
+ # never reached probe threshold (short file) — decide now
239
+ self._decide_mode()
240
+ buffered = self._probe_buffer
241
+ self._probe_buffer = []
242
+ for buffered_line in buffered:
243
+ pending.extend(self._feed_decided(buffered_line))
244
+ if self._current and self._current.lines:
245
+ pending.append(self._current)
246
+ self._current = None
247
+ return pending