logometer 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- logometer/__init__.py +3 -0
- logometer/baseline.py +54 -0
- logometer/classifier.py +71 -0
- logometer/cli.py +447 -0
- logometer/detector.py +70 -0
- logometer/explain.py +185 -0
- logometer/pretty.py +52 -0
- logometer/timeparse.py +58 -0
- logometer/windower.py +247 -0
- logometer-0.1.0.dist-info/METADATA +294 -0
- logometer-0.1.0.dist-info/RECORD +15 -0
- logometer-0.1.0.dist-info/WHEEL +5 -0
- logometer-0.1.0.dist-info/entry_points.txt +2 -0
- logometer-0.1.0.dist-info/licenses/LICENSE +21 -0
- logometer-0.1.0.dist-info/top_level.txt +1 -0
logometer/explain.py
ADDED
|
@@ -0,0 +1,185 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Optional "--explain" support: sends an anomalous window's lines to an
|
|
3
|
+
LLM and asks for a one-sentence plain-English summary of the likely
|
|
4
|
+
cause.
|
|
5
|
+
|
|
6
|
+
Providers: Anthropic, OpenAI, and OpenRouter (OpenAI-compatible, gives
|
|
7
|
+
access to many vendors' models with one key).
|
|
8
|
+
|
|
9
|
+
Deliberately built on stdlib `urllib` rather than the `anthropic` or
|
|
10
|
+
`openai` SDKs, so `--explain` needs nothing beyond an API key — no
|
|
11
|
+
extra pip install, no risk of drifting out of sync with an SDK's API
|
|
12
|
+
surface. This is a deviation from the original plan (which sketched
|
|
13
|
+
"Anthropic or OpenAI SDK"); seemed like the better trade for a tool
|
|
14
|
+
whose whole pitch is "works with nothing installed."
|
|
15
|
+
|
|
16
|
+
This module must NEVER raise anything except ExplainError — callers
|
|
17
|
+
(cli.py) rely on that to fail soft (print a note, keep tailing) rather
|
|
18
|
+
than crashing the whole run because an anomaly happened to coincide
|
|
19
|
+
with a flaky network.
|
|
20
|
+
"""
|
|
21
|
+
from __future__ import annotations
|
|
22
|
+
|
|
23
|
+
import json
|
|
24
|
+
import os
|
|
25
|
+
import urllib.error
|
|
26
|
+
import urllib.request
|
|
27
|
+
|
|
28
|
+
_ANTHROPIC_URL = "https://api.anthropic.com/v1/messages"
|
|
29
|
+
_ANTHROPIC_MODEL = "claude-haiku-4-5-20251001" # fast + cheap, all we need for a one-liner
|
|
30
|
+
|
|
31
|
+
_OPENAI_URL = "https://api.openai.com/v1/chat/completions"
|
|
32
|
+
_OPENAI_MODEL = "gpt-4o-mini"
|
|
33
|
+
|
|
34
|
+
# OpenRouter speaks the OpenAI chat-completions format, so it shares
|
|
35
|
+
# that code path; model ids are namespaced, e.g. "openai/gpt-4o-mini".
|
|
36
|
+
_OPENROUTER_URL = "https://openrouter.ai/api/v1/chat/completions"
|
|
37
|
+
_OPENROUTER_MODEL = "anthropic/claude-haiku-4.5"
|
|
38
|
+
|
|
39
|
+
_TIMEOUT_SECONDS = 15
|
|
40
|
+
_MAX_LINES_SENT = 30 # cap what we send: a window shouldn't need more context than this to explain
|
|
41
|
+
|
|
42
|
+
_SYSTEM_PROMPT = (
|
|
43
|
+
"You are looking at a burst of anomalous lines from an application log. "
|
|
44
|
+
"In ONE short sentence (under 25 words), state the most likely underlying cause. "
|
|
45
|
+
"Be concrete and specific to what's in the lines. No preamble, no hedging, no markdown."
|
|
46
|
+
)
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
class ExplainError(Exception):
|
|
50
|
+
"""Raised for any failure in getting an explanation — missing API
|
|
51
|
+
key, network failure, bad response, unknown provider. Callers
|
|
52
|
+
should catch this and degrade gracefully rather than crash."""
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def _build_prompt(lines: list[str]) -> str:
|
|
56
|
+
sample = lines[:_MAX_LINES_SENT]
|
|
57
|
+
joined = "\n".join(sample)
|
|
58
|
+
if len(lines) > _MAX_LINES_SENT:
|
|
59
|
+
joined += f"\n... ({len(lines) - _MAX_LINES_SENT} more similar lines omitted)"
|
|
60
|
+
return f"Log lines:\n{joined}"
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
def _call_anthropic(lines: list[str], model: str) -> str:
|
|
64
|
+
api_key = os.environ.get("ANTHROPIC_API_KEY")
|
|
65
|
+
if not api_key:
|
|
66
|
+
raise ExplainError(
|
|
67
|
+
"ANTHROPIC_API_KEY is not set. Set it to enable --explain, "
|
|
68
|
+
"e.g. export ANTHROPIC_API_KEY=your-key-here"
|
|
69
|
+
)
|
|
70
|
+
|
|
71
|
+
body = json.dumps({
|
|
72
|
+
"model": model,
|
|
73
|
+
"max_tokens": 100,
|
|
74
|
+
"system": _SYSTEM_PROMPT,
|
|
75
|
+
"messages": [{"role": "user", "content": _build_prompt(lines)}],
|
|
76
|
+
}).encode("utf-8")
|
|
77
|
+
|
|
78
|
+
request = urllib.request.Request(
|
|
79
|
+
_ANTHROPIC_URL,
|
|
80
|
+
data=body,
|
|
81
|
+
method="POST",
|
|
82
|
+
headers={
|
|
83
|
+
"Content-Type": "application/json",
|
|
84
|
+
"x-api-key": api_key,
|
|
85
|
+
"anthropic-version": "2023-06-01",
|
|
86
|
+
},
|
|
87
|
+
)
|
|
88
|
+
try:
|
|
89
|
+
with urllib.request.urlopen(request, timeout=_TIMEOUT_SECONDS) as response:
|
|
90
|
+
payload = json.loads(response.read().decode("utf-8"))
|
|
91
|
+
except urllib.error.HTTPError as e:
|
|
92
|
+
detail = e.read().decode("utf-8", errors="replace")[:300]
|
|
93
|
+
raise ExplainError(f"Anthropic API returned {e.code}: {detail}") from e
|
|
94
|
+
except urllib.error.URLError as e:
|
|
95
|
+
raise ExplainError(f"could not reach Anthropic API: {e.reason}") from e
|
|
96
|
+
except TimeoutError as e:
|
|
97
|
+
raise ExplainError("Anthropic API request timed out") from e
|
|
98
|
+
|
|
99
|
+
try:
|
|
100
|
+
blocks = payload["content"]
|
|
101
|
+
text = "".join(b.get("text", "") for b in blocks if b.get("type") == "text").strip()
|
|
102
|
+
except (KeyError, TypeError) as e:
|
|
103
|
+
raise ExplainError(f"unexpected response shape from Anthropic API: {payload!r}") from e
|
|
104
|
+
|
|
105
|
+
if not text:
|
|
106
|
+
raise ExplainError("Anthropic API returned an empty explanation")
|
|
107
|
+
return text
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
def _call_openai_compatible(
|
|
111
|
+
lines: list[str], model: str, *, url: str, key_env: str, provider: str, label: str
|
|
112
|
+
) -> str:
|
|
113
|
+
api_key = os.environ.get(key_env)
|
|
114
|
+
if not api_key:
|
|
115
|
+
raise ExplainError(
|
|
116
|
+
f"{key_env} is not set. Set it to enable --explain with --explain-provider {provider}, "
|
|
117
|
+
f"e.g. export {key_env}=your-key-here"
|
|
118
|
+
)
|
|
119
|
+
|
|
120
|
+
body = json.dumps({
|
|
121
|
+
"model": model,
|
|
122
|
+
"max_tokens": 100,
|
|
123
|
+
"messages": [
|
|
124
|
+
{"role": "system", "content": _SYSTEM_PROMPT},
|
|
125
|
+
{"role": "user", "content": _build_prompt(lines)},
|
|
126
|
+
],
|
|
127
|
+
}).encode("utf-8")
|
|
128
|
+
|
|
129
|
+
request = urllib.request.Request(
|
|
130
|
+
url,
|
|
131
|
+
data=body,
|
|
132
|
+
method="POST",
|
|
133
|
+
headers={
|
|
134
|
+
"Content-Type": "application/json",
|
|
135
|
+
"Authorization": f"Bearer {api_key}",
|
|
136
|
+
},
|
|
137
|
+
)
|
|
138
|
+
try:
|
|
139
|
+
with urllib.request.urlopen(request, timeout=_TIMEOUT_SECONDS) as response:
|
|
140
|
+
payload = json.loads(response.read().decode("utf-8"))
|
|
141
|
+
except urllib.error.HTTPError as e:
|
|
142
|
+
detail = e.read().decode("utf-8", errors="replace")[:300]
|
|
143
|
+
raise ExplainError(f"{label} API returned {e.code}: {detail}") from e
|
|
144
|
+
except urllib.error.URLError as e:
|
|
145
|
+
raise ExplainError(f"could not reach {label} API: {e.reason}") from e
|
|
146
|
+
except TimeoutError as e:
|
|
147
|
+
raise ExplainError(f"{label} API request timed out") from e
|
|
148
|
+
|
|
149
|
+
try:
|
|
150
|
+
text = payload["choices"][0]["message"]["content"].strip()
|
|
151
|
+
except (KeyError, IndexError, TypeError, AttributeError) as e:
|
|
152
|
+
raise ExplainError(f"unexpected response shape from {label} API: {payload!r}") from e
|
|
153
|
+
|
|
154
|
+
if not text:
|
|
155
|
+
raise ExplainError(f"{label} API returned an empty explanation")
|
|
156
|
+
return text
|
|
157
|
+
|
|
158
|
+
|
|
159
|
+
def _call_openai(lines: list[str], model: str) -> str:
|
|
160
|
+
return _call_openai_compatible(
|
|
161
|
+
lines, model, url=_OPENAI_URL, key_env="OPENAI_API_KEY", provider="openai", label="OpenAI"
|
|
162
|
+
)
|
|
163
|
+
|
|
164
|
+
|
|
165
|
+
def _call_openrouter(lines: list[str], model: str) -> str:
|
|
166
|
+
return _call_openai_compatible(
|
|
167
|
+
lines, model, url=_OPENROUTER_URL, key_env="OPENROUTER_API_KEY", provider="openrouter", label="OpenRouter"
|
|
168
|
+
)
|
|
169
|
+
|
|
170
|
+
|
|
171
|
+
_PROVIDERS = {
|
|
172
|
+
"anthropic": (_call_anthropic, _ANTHROPIC_MODEL),
|
|
173
|
+
"openai": (_call_openai, _OPENAI_MODEL),
|
|
174
|
+
"openrouter": (_call_openrouter, _OPENROUTER_MODEL),
|
|
175
|
+
}
|
|
176
|
+
|
|
177
|
+
|
|
178
|
+
def explain_window(lines: list[str], provider: str = "anthropic", model: str | None = None) -> str:
|
|
179
|
+
"""Get a one-sentence explanation for a burst of anomalous log
|
|
180
|
+
lines. Raises ExplainError on any failure — callers should catch
|
|
181
|
+
this and continue without the explanation rather than crash."""
|
|
182
|
+
if provider not in _PROVIDERS:
|
|
183
|
+
raise ExplainError(f"unknown --explain-provider {provider!r}, expected one of {list(_PROVIDERS)}")
|
|
184
|
+
fn, default_model = _PROVIDERS[provider]
|
|
185
|
+
return fn(lines, model or default_model)
|
logometer/pretty.py
ADDED
|
@@ -0,0 +1,52 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Nicer terminal rendering using `rich`, used automatically when it's
|
|
3
|
+
installed and we're writing to a real terminal. Falls back to the
|
|
4
|
+
plain ANSI formatting in cli.py otherwise — this module is never
|
|
5
|
+
required, only opportunistic.
|
|
6
|
+
"""
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
try:
|
|
10
|
+
from rich.console import Console
|
|
11
|
+
from rich.panel import Panel
|
|
12
|
+
from rich.text import Text
|
|
13
|
+
HAS_RICH = True
|
|
14
|
+
except ImportError:
|
|
15
|
+
HAS_RICH = False
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def render_window(window, verdict, explanation: str | None, console: "Console") -> None:
|
|
19
|
+
"""Render one window's verdict with rich. Only call this if
|
|
20
|
+
HAS_RICH is True."""
|
|
21
|
+
label = f"{window.start_label} \u2013 {window.end_label}"
|
|
22
|
+
|
|
23
|
+
if not verdict.is_anomaly:
|
|
24
|
+
text = Text(f" [{label}] ok errors: {window.error_count} baseline: ~{verdict.baseline_mean:.1f}")
|
|
25
|
+
text.stylize("dim")
|
|
26
|
+
console.print(text)
|
|
27
|
+
return
|
|
28
|
+
|
|
29
|
+
body = Text()
|
|
30
|
+
body.append(f"errors: {window.error_count} ", style="bold")
|
|
31
|
+
if window.warn_count:
|
|
32
|
+
# a first-seen WARN shape can flag a window with zero errors
|
|
33
|
+
body.append(f"warns: {window.warn_count} ", style="bold")
|
|
34
|
+
body.append(f"baseline: ~{verdict.baseline_mean:.1f}")
|
|
35
|
+
if verdict.error_score:
|
|
36
|
+
body.append(f" score: {verdict.error_score:.1f}x", style="bold red" if verdict.error_score >= 3 else "yellow")
|
|
37
|
+
|
|
38
|
+
if verdict.new_shapes:
|
|
39
|
+
body.append("\n")
|
|
40
|
+
levels = {l.shape: l.level for l in window.lines if l.shape}
|
|
41
|
+
for shape in verdict.new_shapes:
|
|
42
|
+
kind = "warning" if levels.get(shape) == "WARN" else "error"
|
|
43
|
+
body.append(f"\n New {kind} signature: ", style="yellow")
|
|
44
|
+
body.append(shape, style="italic")
|
|
45
|
+
elif not explanation:
|
|
46
|
+
body.append("\n\n (error rate spike \u2014 no brand-new error signature)", style="dim")
|
|
47
|
+
|
|
48
|
+
if explanation:
|
|
49
|
+
body.append("\n\n ")
|
|
50
|
+
body.append(explanation, style="cyan")
|
|
51
|
+
|
|
52
|
+
console.print(Panel(body, title=f"\u26a0 ANOMALY [{label}]", border_style="red", expand=False))
|
logometer/timeparse.py
ADDED
|
@@ -0,0 +1,58 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Best-effort extraction of a timestamp from the start of a log line.
|
|
3
|
+
|
|
4
|
+
We only need this to decide how to bucket lines into windows. If a
|
|
5
|
+
line's timestamp can't be parsed, the caller falls back to a
|
|
6
|
+
line-count-based window instead of a time-based one — see windower.py.
|
|
7
|
+
"""
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import re
|
|
11
|
+
from datetime import datetime
|
|
12
|
+
from typing import Optional
|
|
13
|
+
|
|
14
|
+
# ISO 8601, e.g. "2026-09-17T12:03:10.123Z" or "2026-09-17 12:03:10"
|
|
15
|
+
_ISO_RE = re.compile(
|
|
16
|
+
r"(\d{4}-\d{2}-\d{2}[T ]\d{2}:\d{2}:\d{2}(?:\.\d+)?)(Z|[+-]\d{2}:?\d{2})?"
|
|
17
|
+
)
|
|
18
|
+
|
|
19
|
+
# Syslog-style, e.g. "Sep 17 12:03:10"
|
|
20
|
+
_SYSLOG_RE = re.compile(
|
|
21
|
+
r"\b([A-Z][a-z]{2}\s+\d{1,2}\s+\d{2}:\d{2}:\d{2})\b"
|
|
22
|
+
)
|
|
23
|
+
|
|
24
|
+
_SYSLOG_MONTHS = {
|
|
25
|
+
"Jan": 1, "Feb": 2, "Mar": 3, "Apr": 4, "May": 5, "Jun": 6,
|
|
26
|
+
"Jul": 7, "Aug": 8, "Sep": 9, "Oct": 10, "Nov": 11, "Dec": 12,
|
|
27
|
+
}
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def parse_timestamp(line: str, assumed_year: Optional[int] = None) -> Optional[datetime]:
|
|
31
|
+
"""Best-effort ISO or syslog timestamp extraction from a log line.
|
|
32
|
+
Returns None when no recognizable time is found (caller may use line-based windows)."""
|
|
33
|
+
m = _ISO_RE.search(line)
|
|
34
|
+
if m:
|
|
35
|
+
raw = m.group(1)
|
|
36
|
+
try:
|
|
37
|
+
# normalize a couple of common variants Python's fromisoformat
|
|
38
|
+
# is picky about
|
|
39
|
+
raw = raw.replace(" ", "T") if "T" not in raw else raw
|
|
40
|
+
return datetime.fromisoformat(raw)
|
|
41
|
+
except ValueError:
|
|
42
|
+
pass
|
|
43
|
+
|
|
44
|
+
m = _SYSLOG_RE.search(line)
|
|
45
|
+
if m:
|
|
46
|
+
try:
|
|
47
|
+
month_str, day_str, time_str = m.group(1).split(None, 2)
|
|
48
|
+
# handle "Sep 1 12:03:10" (double space) already collapsed by split
|
|
49
|
+
month = _SYSLOG_MONTHS.get(month_str)
|
|
50
|
+
if month is None:
|
|
51
|
+
return None
|
|
52
|
+
year = assumed_year or datetime.now().year
|
|
53
|
+
hh, mm, ss = (int(x) for x in time_str.split(":"))
|
|
54
|
+
return datetime(year, month, int(day_str), hh, mm, ss)
|
|
55
|
+
except (ValueError, IndexError):
|
|
56
|
+
return None
|
|
57
|
+
|
|
58
|
+
return None
|
logometer/windower.py
ADDED
|
@@ -0,0 +1,247 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Groups a stream of classified log lines into fixed-size windows.
|
|
3
|
+
|
|
4
|
+
Two modes, chosen automatically:
|
|
5
|
+
- "timed": if timestamps can be parsed from the log lines, windows are
|
|
6
|
+
bucketed by wall-clock/log-clock time (--window seconds).
|
|
7
|
+
- "count": if no timestamps are found (first N lines all fail to
|
|
8
|
+
parse), windows are bucketed by a fixed number of lines instead.
|
|
9
|
+
This keeps the tool useful on logs with no recognizable timestamp
|
|
10
|
+
format, at the cost of windows not corresponding to a fixed time span.
|
|
11
|
+
"""
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
import math
|
|
15
|
+
import time
|
|
16
|
+
from dataclasses import dataclass, field
|
|
17
|
+
from datetime import datetime
|
|
18
|
+
from typing import Iterator, Optional
|
|
19
|
+
|
|
20
|
+
from .classifier import ClassifiedLine, classify_line
|
|
21
|
+
from .timeparse import parse_timestamp
|
|
22
|
+
|
|
23
|
+
# how many leading lines we sample to decide timed vs. count mode
|
|
24
|
+
_PROBE_LINES = 20
|
|
25
|
+
# fraction of probe lines that must have a parseable timestamp to use timed mode
|
|
26
|
+
_PROBE_THRESHOLD = 0.5
|
|
27
|
+
# fallback window size (in lines) when no timestamps are parseable
|
|
28
|
+
_DEFAULT_LINES_PER_WINDOW = 50
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
@dataclass
|
|
32
|
+
class Window:
|
|
33
|
+
index: int
|
|
34
|
+
start_label: str # human-readable label for the window (time range or line range)
|
|
35
|
+
end_label: str
|
|
36
|
+
lines: list[ClassifiedLine] = field(default_factory=list)
|
|
37
|
+
|
|
38
|
+
@property
|
|
39
|
+
def error_count(self) -> int:
|
|
40
|
+
"""Number of lines in this window classified as ERROR.
|
|
41
|
+
Fed to RollingBaseline as the primary spike metric."""
|
|
42
|
+
return sum(1 for l in self.lines if l.level == "ERROR")
|
|
43
|
+
|
|
44
|
+
@property
|
|
45
|
+
def warn_count(self) -> int:
|
|
46
|
+
"""Number of lines in this window classified as WARN.
|
|
47
|
+
Included in JSON output; not used for baseline spike scoring."""
|
|
48
|
+
return sum(1 for l in self.lines if l.level == "WARN")
|
|
49
|
+
|
|
50
|
+
@property
|
|
51
|
+
def shapes(self) -> set[str]:
|
|
52
|
+
"""Distinct error/warn fingerprints present in this window.
|
|
53
|
+
Compared against AnomalyDetector._seen_shapes for new-signature anomalies."""
|
|
54
|
+
return {l.shape for l in self.lines if l.shape}
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
class WindowAggregator:
|
|
58
|
+
"""Feed classified lines in; get completed Window objects out.
|
|
59
|
+
|
|
60
|
+
Usage:
|
|
61
|
+
agg = WindowAggregator(window_seconds=10)
|
|
62
|
+
for line in source:
|
|
63
|
+
for window in agg.feed(line):
|
|
64
|
+
handle(window)
|
|
65
|
+
for window in agg.flush():
|
|
66
|
+
handle(window)
|
|
67
|
+
"""
|
|
68
|
+
|
|
69
|
+
def __init__(self, window_seconds: float = 10.0, lines_per_window: int = _DEFAULT_LINES_PER_WINDOW):
|
|
70
|
+
"""Create an aggregator; timed vs count mode is chosen after probing first lines.
|
|
71
|
+
window_seconds applies in timed mode; lines_per_window in count mode."""
|
|
72
|
+
self.window_seconds = window_seconds
|
|
73
|
+
self.lines_per_window = lines_per_window
|
|
74
|
+
|
|
75
|
+
self._mode: Optional[str] = None # "timed" or "count", decided after probing
|
|
76
|
+
self._probe_buffer: list[str] = []
|
|
77
|
+
self._probe_timestamps: list[Optional[datetime]] = []
|
|
78
|
+
self._probe_started_at: Optional[float] = None
|
|
79
|
+
|
|
80
|
+
self._current: Optional[Window] = None
|
|
81
|
+
self._current_bucket_key = None
|
|
82
|
+
self._window_index = 0
|
|
83
|
+
self._first_timestamp: Optional[datetime] = None
|
|
84
|
+
# real-time clock (not log time) for the in-progress window, so
|
|
85
|
+
# tick() can close it on schedule when the log goes silent
|
|
86
|
+
self._current_opened_at: Optional[float] = None
|
|
87
|
+
|
|
88
|
+
# -- mode detection -----------------------------------------------------
|
|
89
|
+
|
|
90
|
+
def _decide_mode(self) -> None:
|
|
91
|
+
"""Pick timed windows if enough probe lines had timestamps, else count-based.
|
|
92
|
+
Sets _first_timestamp when entering timed mode."""
|
|
93
|
+
parseable = sum(1 for t in self._probe_timestamps if t is not None)
|
|
94
|
+
if self._probe_timestamps and parseable / len(self._probe_timestamps) >= _PROBE_THRESHOLD:
|
|
95
|
+
self._mode = "timed"
|
|
96
|
+
for t in self._probe_timestamps:
|
|
97
|
+
if t is not None:
|
|
98
|
+
self._first_timestamp = t
|
|
99
|
+
break
|
|
100
|
+
else:
|
|
101
|
+
self._mode = "count"
|
|
102
|
+
|
|
103
|
+
# -- public API -----------------------------------------------------
|
|
104
|
+
|
|
105
|
+
def feed(self, raw_line: str) -> Iterator[Window]:
|
|
106
|
+
"""Ingest one raw log line; yield any windows that closed (often zero or one).
|
|
107
|
+
Buffers an initial probe batch before choosing timed vs count mode."""
|
|
108
|
+
raw_line = raw_line.rstrip("\n")
|
|
109
|
+
if not raw_line:
|
|
110
|
+
return
|
|
111
|
+
|
|
112
|
+
if self._mode is None:
|
|
113
|
+
if self._probe_started_at is None:
|
|
114
|
+
self._probe_started_at = time.monotonic()
|
|
115
|
+
self._probe_buffer.append(raw_line)
|
|
116
|
+
self._probe_timestamps.append(parse_timestamp(raw_line))
|
|
117
|
+
if len(self._probe_buffer) < _PROBE_LINES:
|
|
118
|
+
return # still probing, hold lines until mode is decided
|
|
119
|
+
self._decide_mode()
|
|
120
|
+
# replay the buffered probe lines now that we know the mode
|
|
121
|
+
buffered = self._probe_buffer
|
|
122
|
+
self._probe_buffer = []
|
|
123
|
+
for buffered_line in buffered:
|
|
124
|
+
yield from self._feed_decided(buffered_line)
|
|
125
|
+
return
|
|
126
|
+
|
|
127
|
+
yield from self._feed_decided(raw_line)
|
|
128
|
+
|
|
129
|
+
def _feed_decided(self, raw_line: str) -> Iterator[Window]:
|
|
130
|
+
"""Classify a line, assign it to the current bucket, yield completed windows.
|
|
131
|
+
Requires _mode to already be timed or count."""
|
|
132
|
+
classified = classify_line(raw_line)
|
|
133
|
+
|
|
134
|
+
if self._mode == "timed":
|
|
135
|
+
ts = parse_timestamp(raw_line) or self._last_seen_timestamp()
|
|
136
|
+
bucket_key = math.floor((ts - self._first_timestamp).total_seconds() / self.window_seconds)
|
|
137
|
+
self._last_ts = ts
|
|
138
|
+
else:
|
|
139
|
+
bucket_key = None # computed below from line count
|
|
140
|
+
|
|
141
|
+
if self._current is None:
|
|
142
|
+
self._current = self._new_window(bucket_key)
|
|
143
|
+
self._current_bucket_key = bucket_key
|
|
144
|
+
|
|
145
|
+
if self._mode == "count":
|
|
146
|
+
if len(self._current.lines) >= self.lines_per_window:
|
|
147
|
+
yield self._current
|
|
148
|
+
self._window_index += 1
|
|
149
|
+
self._current = self._new_window(None)
|
|
150
|
+
else:
|
|
151
|
+
if bucket_key != self._current_bucket_key:
|
|
152
|
+
yield self._current
|
|
153
|
+
self._window_index += 1
|
|
154
|
+
self._current = self._new_window(bucket_key)
|
|
155
|
+
self._current_bucket_key = bucket_key
|
|
156
|
+
|
|
157
|
+
self._current.lines.append(classified)
|
|
158
|
+
|
|
159
|
+
def _last_seen_timestamp(self) -> datetime:
|
|
160
|
+
"""Fallback clock when a line in timed mode has no parseable timestamp.
|
|
161
|
+
Reuses last seen time, else log start, else wall clock."""
|
|
162
|
+
return getattr(self, "_last_ts", self._first_timestamp or datetime.now())
|
|
163
|
+
|
|
164
|
+
def _new_window(self, bucket_key) -> Window:
|
|
165
|
+
"""Allocate the next window with human-readable start/end labels.
|
|
166
|
+
Timed labels are HH:MM:SS ranges; count mode uses line number ranges."""
|
|
167
|
+
if self._mode == "timed" and bucket_key is not None:
|
|
168
|
+
start = self._first_timestamp.timestamp() + bucket_key * self.window_seconds
|
|
169
|
+
end = start + self.window_seconds
|
|
170
|
+
start_label = datetime.fromtimestamp(start).strftime("%H:%M:%S")
|
|
171
|
+
end_label = datetime.fromtimestamp(end).strftime("%H:%M:%S")
|
|
172
|
+
else:
|
|
173
|
+
start_n = self._window_index * self.lines_per_window + 1
|
|
174
|
+
end_n = start_n + self.lines_per_window - 1
|
|
175
|
+
start_label = f"line {start_n}"
|
|
176
|
+
end_label = f"line {end_n}"
|
|
177
|
+
self._current_opened_at = time.monotonic()
|
|
178
|
+
return Window(index=self._window_index, start_label=start_label, end_label=end_label)
|
|
179
|
+
|
|
180
|
+
def tick(self, now: Optional[float] = None) -> list[Window]:
|
|
181
|
+
"""Close anything whose time is up, without needing a new line.
|
|
182
|
+
|
|
183
|
+
Feeding alone only closes a window when the *next* line arrives
|
|
184
|
+
in a later bucket. On a live tail that means a service which
|
|
185
|
+
errors and then dies never gets reported at all: the burst that
|
|
186
|
+
matters most stays buffered forever, because nothing follows it.
|
|
187
|
+
Callers poll this while idle so windows close on schedule.
|
|
188
|
+
|
|
189
|
+
Two things can be overdue:
|
|
190
|
+
- the in-progress window, once `window_seconds` of real time
|
|
191
|
+
have passed since it opened
|
|
192
|
+
- the probe batch itself, on a log too quiet to produce 20
|
|
193
|
+
lines — otherwise a low-volume service reports nothing ever
|
|
194
|
+
|
|
195
|
+
Real (monotonic) time is used deliberately, not log time: a log
|
|
196
|
+
whose clock is skewed or in another timezone would otherwise
|
|
197
|
+
close every window the moment it went idle.
|
|
198
|
+
"""
|
|
199
|
+
now = time.monotonic() if now is None else now
|
|
200
|
+
closed: list[Window] = []
|
|
201
|
+
|
|
202
|
+
if self._mode is None:
|
|
203
|
+
waited = self._probe_started_at is not None and now - self._probe_started_at >= self.window_seconds
|
|
204
|
+
if not (self._probe_buffer and waited):
|
|
205
|
+
return []
|
|
206
|
+
self._decide_mode()
|
|
207
|
+
buffered = self._probe_buffer
|
|
208
|
+
self._probe_buffer = []
|
|
209
|
+
for buffered_line in buffered:
|
|
210
|
+
closed.extend(self._feed_decided(buffered_line))
|
|
211
|
+
|
|
212
|
+
if (
|
|
213
|
+
self._current is not None
|
|
214
|
+
and self._current.lines
|
|
215
|
+
and self._current_opened_at is not None
|
|
216
|
+
and now - self._current_opened_at >= self.window_seconds
|
|
217
|
+
):
|
|
218
|
+
closed.append(self._current)
|
|
219
|
+
self._current = None
|
|
220
|
+
self._current_opened_at = None
|
|
221
|
+
self._window_index += 1
|
|
222
|
+
|
|
223
|
+
return closed
|
|
224
|
+
|
|
225
|
+
def flush(self) -> list[Window]:
|
|
226
|
+
"""Return every window still pending at EOF, oldest first.
|
|
227
|
+
|
|
228
|
+
Usually that's just the one in-progress window. But on a file
|
|
229
|
+
shorter than the probe batch, mode is decided here and the
|
|
230
|
+
buffered lines are replayed now — which can close several
|
|
231
|
+
windows at once. All of them must come back: an earlier version
|
|
232
|
+
kept only the last one, so a short log silently lost every
|
|
233
|
+
window but its final one (and with them the baseline history
|
|
234
|
+
the detector needs to flag anything at all).
|
|
235
|
+
"""
|
|
236
|
+
pending: list[Window] = []
|
|
237
|
+
if self._mode is None and self._probe_buffer:
|
|
238
|
+
# never reached probe threshold (short file) — decide now
|
|
239
|
+
self._decide_mode()
|
|
240
|
+
buffered = self._probe_buffer
|
|
241
|
+
self._probe_buffer = []
|
|
242
|
+
for buffered_line in buffered:
|
|
243
|
+
pending.extend(self._feed_decided(buffered_line))
|
|
244
|
+
if self._current and self._current.lines:
|
|
245
|
+
pending.append(self._current)
|
|
246
|
+
self._current = None
|
|
247
|
+
return pending
|