stackdoctor 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
File without changes
@@ -0,0 +1,3 @@
1
+ from .server import main
2
+
3
+ main()
@@ -0,0 +1,29 @@
1
+ """Checks. Every check is a plain sync function returning a dict.
2
+
3
+ Checks may include an `events` list: notable things with a timestamp, which
4
+ diagnose() merges into one cross-system timeline.
5
+ """
6
+
7
+ from __future__ import annotations
8
+
9
+ import datetime as dt
10
+
11
+
12
+ def utcnow() -> dt.datetime:
13
+ return dt.datetime.now(dt.timezone.utc)
14
+
15
+
16
+ def to_utc(ts: dt.datetime | float | None) -> dt.datetime | None:
17
+ if ts is None:
18
+ return None
19
+ if isinstance(ts, (int, float)):
20
+ return dt.datetime.fromtimestamp(ts, dt.timezone.utc)
21
+ if ts.tzinfo is None:
22
+ ts = ts.astimezone() # naive = local time
23
+ return ts.astimezone(dt.timezone.utc)
24
+
25
+
26
+ def event(ts, source: str, kind: str, detail: str, severity: str = "info", **extra) -> dict:
27
+ """A timeline event. `kind` is a stable keyword used by the cause → effect rules."""
28
+ return {"ts": to_utc(ts) or utcnow(), "source": source, "kind": kind,
29
+ "severity": severity, "detail": detail, **extra}
@@ -0,0 +1,319 @@
1
+ """Celery checks without Flower: the inspect API plus reading the broker directly.
2
+
3
+ Only ping/active/reserved/active_queues/stats/query_task are used. Nothing here
4
+ sends, revokes, retries or shuts down tasks or workers.
5
+ """
6
+
7
+ from __future__ import annotations
8
+
9
+ import contextlib
10
+ import datetime as dt
11
+ import importlib
12
+ import json
13
+ import os
14
+ import re
15
+ import sys
16
+ import threading
17
+ import time
18
+
19
+ from celery import Celery
20
+
21
+ from ..config import get_config, skipped
22
+ from ..safety import preview
23
+ from . import event, to_utc
24
+ from .redis import client as redis_client, glob_escape
25
+
26
+ SOURCE = "celery"
27
+ _TASK_ID_RE = re.compile(r"^[A-Za-z0-9_.:-]{1,128}$")
28
+ _META_PREFIX = "celery-task-meta-"
29
+ _HISTORY: dict[str, list[tuple[float, int]]] = {} # queue -> recent (time, length) samples
30
+
31
+
32
+ _app = None
33
+ _app_lock = threading.Lock() # diagnose() calls checks from several threads at once
34
+
35
+
36
+ def get_app():
37
+ global _app
38
+ with _app_lock:
39
+ if _app is None:
40
+ _app = _load_app()
41
+ return _app
42
+
43
+
44
+ def _load_app():
45
+ """Load CELERY_APP (`pkg.module:app` or `pkg.module`), else a bare app on the broker."""
46
+ cfg = get_config()
47
+ if cfg.celery_app:
48
+ if os.getcwd() not in sys.path:
49
+ sys.path.insert(0, os.getcwd())
50
+ mod_name, _, attr = cfg.celery_app.partition(":")
51
+ # Importing user code must never print to stdout: that is the MCP channel.
52
+ with contextlib.redirect_stdout(sys.stderr):
53
+ module = importlib.import_module(mod_name)
54
+ if attr:
55
+ return getattr(module, attr)
56
+ for name in ("app", "celery", "celery_app"):
57
+ if hasattr(module, name):
58
+ return getattr(module, name)
59
+ raise ImportError(f"No Celery app found in {mod_name} (use module:attr)")
60
+ if cfg.celery_broker_url:
61
+ return Celery("stackdoctor", broker=cfg.celery_broker_url,
62
+ backend=cfg.celery_result_backend)
63
+ return None
64
+
65
+
66
+ def _not_configured():
67
+ cfg = get_config()
68
+ if not (cfg.celery_app or cfg.celery_broker_url):
69
+ return skipped("CELERY_APP / CELERY_BROKER_URL not set")
70
+ return None
71
+
72
+
73
+ def _inspect(app, destination=None):
74
+ timeout = get_config().celery_inspect_timeout_s
75
+ return app.control.inspect(timeout=timeout, destination=destination,
76
+ limit=len(destination) if destination else None)
77
+
78
+
79
+ def _task_summary(t: dict) -> dict:
80
+ started = t.get("time_start")
81
+ return {
82
+ "id": t.get("id"), "name": t.get("name"),
83
+ "args": preview(t.get("args"), 120), "kwargs": preview(t.get("kwargs"), 120),
84
+ "started": to_utc(started),
85
+ "runtime_s": round(time.time() - started, 1) if started else None,
86
+ "queue": (t.get("delivery_info") or {}).get("routing_key"),
87
+ }
88
+
89
+
90
+ def workers() -> dict:
91
+ if (s := _not_configured()):
92
+ return s
93
+ cfg, app = get_config(), get_app()
94
+ ping = _inspect(app).ping() or {}
95
+ events = []
96
+ if not ping:
97
+ events.append(event(None, SOURCE, "no_workers",
98
+ "No Celery workers replied to ping", "critical"))
99
+ return {"alive_count": 0, "workers": [], "events": events}
100
+
101
+ names = sorted(ping)
102
+ insp = _inspect(app, destination=names)
103
+ active, reserved = insp.active() or {}, insp.reserved() or {}
104
+ queues, stats = insp.active_queues() or {}, insp.stats() or {}
105
+
106
+ out = []
107
+ for name in names:
108
+ st = stats.get(name, {})
109
+ concurrency = (st.get("pool") or {}).get("max-concurrency")
110
+ act = [_task_summary(t) for t in active.get(name, [])]
111
+ res = [_task_summary(t) for t in reserved.get(name, [])]
112
+ out.append({
113
+ "name": name, "alive": True,
114
+ "queues": [q.get("name") for q in queues.get(name, [])],
115
+ "concurrency": concurrency,
116
+ "active_count": len(act), "reserved_count": len(res),
117
+ "active": act[:20], "reserved": res[:10],
118
+ "processed_total": sum((st.get("total") or {}).values()),
119
+ })
120
+ for t in act:
121
+ if t["runtime_s"] and t["runtime_s"] >= cfg.long_task_s:
122
+ events.append(event(t["started"], SOURCE, "task_long_running",
123
+ f"{t['name']}[{t['id']}] running {t['runtime_s']}s on {name}",
124
+ "warning", task_id=t["id"]))
125
+ if concurrency and len(act) >= concurrency:
126
+ events.append(event(None, SOURCE, "workers_saturated",
127
+ f"{name}: all {concurrency} slots busy, {len(res)} reserved", "warning"))
128
+
129
+ if cfg.expected_workers and len(names) < cfg.expected_workers:
130
+ events.append(event(None, SOURCE, "worker_missing",
131
+ f"Only {len(names)} of {cfg.expected_workers} expected workers replied",
132
+ "critical"))
133
+ return {"alive_count": len(names), "workers": out, "events": events}
134
+
135
+
136
+ def _queue_names(app, given: list[str] | None) -> list[str]:
137
+ if given:
138
+ return given
139
+ names = set(get_config().celery_queues)
140
+ names.add(app.conf.task_default_queue or "celery")
141
+ for q in app.conf.task_queues or []:
142
+ names.add(getattr(q, "name", q))
143
+ for route in (app.conf.task_routes or {}).values() if isinstance(app.conf.task_routes, dict) else []:
144
+ if isinstance(route, dict) and route.get("queue"):
145
+ names.add(route["queue"])
146
+ return sorted(names)
147
+
148
+
149
+ def _redis_queue_lengths(app, url: str, names: list[str]) -> dict:
150
+ opts = app.conf.broker_transport_options or {}
151
+ steps = opts.get("priority_steps") or [0, 3, 6, 9]
152
+ sep = opts.get("sep", "\x06\x16")
153
+ prefix = opts.get("global_keyprefix", "")
154
+ r = redis_client(url)
155
+ result = {}
156
+ for q in names:
157
+ by_priority = {}
158
+ for step in steps:
159
+ key = prefix + (f"{q}{sep}{step}" if step else q)
160
+ n = r.cmd("LLEN", key)
161
+ if n:
162
+ by_priority[str(step)] = n
163
+ # Steps we don't know about (custom priority_steps on the producer side).
164
+ extra, _ = r.scan_iter(glob_escape(prefix + q + sep) + "*", 50)
165
+ for key in extra:
166
+ step = key.rsplit(sep, 1)[-1]
167
+ if step.isdigit() and int(step) not in steps:
168
+ by_priority[step] = r.cmd("LLEN", key)
169
+ result[q] = {"messages": sum(by_priority.values()),
170
+ "by_priority": by_priority if len(by_priority) > 1 else None}
171
+ unacked = r.cmd("HLEN", prefix + "unacked")
172
+ return {"queues": result, "unacked_total": unacked}
173
+
174
+
175
+ def _amqp_queue_lengths(app, names: list[str]) -> dict:
176
+ result = {}
177
+ for q in names:
178
+ with app.connection_for_read() as conn:
179
+ try:
180
+ # passive=True only checks the queue; it never creates anything.
181
+ _, messages, consumers = conn.default_channel.queue_declare(queue=q, passive=True)
182
+ result[q] = {"messages": messages, "consumers": consumers}
183
+ except conn.channel_errors:
184
+ result[q] = {"error": "queue does not exist"}
185
+ return {"queues": result}
186
+
187
+
188
+ def queue_lengths(queues: list[str] | None = None) -> dict:
189
+ if (s := _not_configured()):
190
+ return s
191
+ cfg, app = get_config(), get_app()
192
+ url = app.conf.broker_url or cfg.celery_broker_url or ""
193
+ names = _queue_names(app, queues)
194
+ if url.startswith(("redis://", "rediss://", "unix://")):
195
+ out = _redis_queue_lengths(app, url, names)
196
+ elif url.startswith(("amqp://", "amqps://", "pyamqp://")):
197
+ out = _amqp_queue_lengths(app, names)
198
+ else:
199
+ return {"error": f"Queue lengths are supported for Redis and RabbitMQ brokers, not {url.split(':')[0]}"}
200
+
201
+ now, events = time.time(), []
202
+ for q, info in out["queues"].items():
203
+ n = info.get("messages")
204
+ if n is None:
205
+ continue
206
+ hist = _HISTORY.setdefault(q, [])
207
+ hist.append((now, n))
208
+ del hist[:-10]
209
+ older = [(t, m) for t, m in hist if now - t >= 5]
210
+ if older:
211
+ t0, m0 = older[-1]
212
+ info["change_since_last_check"] = {"seconds_ago": round(now - t0), "delta": n - m0}
213
+ if n >= cfg.queue_threshold:
214
+ events.append(event(None, SOURCE, "queue_backlog",
215
+ f"Queue '{q}' has {n} waiting messages (threshold {cfg.queue_threshold})",
216
+ "warning", queue=q, messages=n))
217
+ if info.get("consumers") == 0 and n:
218
+ events.append(event(None, SOURCE, "queue_no_consumer",
219
+ f"Queue '{q}' has {n} messages and no consumers", "critical", queue=q))
220
+ out["events"] = events
221
+ return out
222
+
223
+
224
+ def _backend(app) -> tuple[str | None, str | None]:
225
+ """Return (redis_url, None) if results are readable, else (None, reason)."""
226
+ cfg = get_config()
227
+ url = cfg.celery_result_backend or app.conf.result_backend
228
+ if not url:
229
+ return None, "no result backend is configured (CELERY_RESULT_BACKEND / app.conf.result_backend)"
230
+ if not str(url).startswith(("redis://", "rediss://")):
231
+ return None, f"the result backend is '{str(url).split(':')[0]}'; stackdoctor only reads Redis result backends"
232
+ return url, None
233
+
234
+
235
+ def failed_tasks(limit: int = 20) -> dict:
236
+ if (s := _not_configured()):
237
+ return s
238
+ app = get_app()
239
+ url, reason = _backend(app)
240
+ if reason:
241
+ return {"visible": False, "reason": f"Task failures are not visible because {reason}. "
242
+ "Check logs for task errors instead."}
243
+ r = redis_client(url)
244
+ keys, complete = r.scan_iter(_META_PREFIX + "*", 2000)
245
+ if not keys:
246
+ why = ("task_ignore_result is on" if app.conf.task_ignore_result
247
+ else f"no results are stored: results may have expired (result_expires="
248
+ f"{app.conf.result_expires}) or tasks use ignore_result=True")
249
+ return {"visible": False, "reason": f"Task failures are not visible because {why}. "
250
+ "Check logs for task errors instead."}
251
+ failures = []
252
+ for i in range(0, len(keys), 200):
253
+ for raw in r.cmd("MGET", *keys[i:i + 200]):
254
+ try:
255
+ meta = json.loads(raw) if raw else None
256
+ except ValueError:
257
+ continue # pickle or other serializer
258
+ if meta and meta.get("status") == "FAILURE":
259
+ failures.append(meta)
260
+ failures.sort(key=lambda m: m.get("date_done") or "", reverse=True)
261
+ events, out = [], []
262
+ for m in failures[: max(1, min(int(limit), 100))]:
263
+ res = m.get("result") or {}
264
+ tb_tail = (m.get("traceback") or "").strip().splitlines()[-3:]
265
+ item = {"task_id": m.get("task_id"), "name": m.get("name"),
266
+ "date_done": m.get("date_done"),
267
+ "exception": preview(f"{res.get('exc_type')}: {res.get('exc_message')}", 300)
268
+ if isinstance(res, dict) else preview(res, 300),
269
+ "traceback_tail": [preview(l, 200) for l in tb_tail]}
270
+ out.append(item)
271
+ events.append(event(_parse_iso(m.get("date_done")), SOURCE, "task_failed",
272
+ f"{item['name'] or 'task'}[{item['task_id']}] failed: {item['exception']}",
273
+ "warning", task_id=item["task_id"]))
274
+ return {"visible": True, "results_scanned": len(keys), "scan_complete": complete,
275
+ "failed_count": len(failures), "failed": out, "events": events}
276
+
277
+
278
+ def _parse_iso(value):
279
+ try:
280
+ return dt.datetime.fromisoformat(value) if value else None
281
+ except ValueError:
282
+ return None
283
+
284
+
285
+ def task_details(task_id: str) -> dict:
286
+ if (s := _not_configured()):
287
+ return s
288
+ if not _TASK_ID_RE.match(task_id or ""):
289
+ return {"error": "Invalid task id"}
290
+ app = get_app()
291
+ out: dict = {"task_id": task_id}
292
+ url, reason = _backend(app)
293
+ if url:
294
+ raw = redis_client(url).cmd("GET", _META_PREFIX + task_id)
295
+ if raw:
296
+ try:
297
+ meta = json.loads(raw)
298
+ out["result_backend"] = {
299
+ "status": meta.get("status"), "name": meta.get("name"),
300
+ "date_done": meta.get("date_done"), "retries": meta.get("retries"),
301
+ "worker": meta.get("worker"),
302
+ "args": preview(meta.get("args"), 200), "kwargs": preview(meta.get("kwargs"), 200),
303
+ "result": preview(meta.get("result"), 300),
304
+ "traceback_tail": [preview(l, 200) for l in
305
+ (meta.get("traceback") or "").strip().splitlines()[-5:]],
306
+ }
307
+ except ValueError:
308
+ out["result_backend"] = {"note": "result is not JSON-encoded"}
309
+ else:
310
+ out["result_backend"] = {"note": "no stored result (pending, expired, or ignore_result)"}
311
+ else:
312
+ out["result_backend"] = {"note": f"not readable: {reason}"}
313
+ # Ask workers whether they currently hold the task (active/reserved/scheduled).
314
+ found = _inspect(app).query_task(task_id) or {}
315
+ out["on_workers"] = {w: {tid: {"state": state, "name": info.get("name"),
316
+ "args": preview(info.get("args"), 120)}
317
+ for tid, (state, info) in tasks.items()}
318
+ for w, tasks in found.items() if tasks}
319
+ return out
@@ -0,0 +1,235 @@
1
+ """Log checks for files and docker containers listed in LOG_SOURCES.
2
+
3
+ Only configured sources can be read, so the AI can't use this to read arbitrary files.
4
+ """
5
+
6
+ from __future__ import annotations
7
+
8
+ import datetime as dt
9
+ import os
10
+ import re
11
+ import shutil
12
+ import subprocess
13
+ from collections import defaultdict
14
+
15
+ from ..config import get_config, skipped
16
+ from . import event, to_utc, utcnow
17
+
18
+ SOURCE = "logs"
19
+ MAX_LINE = 2000
20
+ MAX_FILE_BYTES = 5_000_000
21
+
22
+ _DOCKER_TS = re.compile(r"^(\d{4}-\d\d-\d\dT\d\d:\d\d:\d\d)(?:\.(\d+))?Z\s?")
23
+ _GENERIC_TS = re.compile(
24
+ r"(\d{4}-\d\d-\d\d)[T ](\d\d:\d\d:\d\d)(?:[.,](\d+))?\s*(Z|UTC|[+-]\d\d:?\d\d)?")
25
+
26
+ # (kind, pattern, severity) — first match wins, so specific rules come first.
27
+ RULES = [
28
+ ("worker_lost", r"WorkerLostError|worker lost|exited with 'signal 9|exited prematurely", "critical"),
29
+ ("worker_shutdown", r"\b(?:warm|cold) shutdown\b", "warning"), # Celery: "worker: Warm shutdown (MainProcess)"
30
+ ("task_timeout", r"SoftTimeLimitExceeded|TimeLimitExceeded|(?:hard|soft) time limit", "warning"),
31
+ ("task_failed", r"Task [\w.]+\[[\w-]+\] raised unexpected|Task [\w.]+\[[\w-]+\] failed", "warning"),
32
+ ("db_lock_wait", r"still waiting for \w*Lock", "warning"),
33
+ ("db_lock_error", r"lock timeout|deadlock detected|could not obtain lock|LockNotAvailable", "warning"),
34
+ ("db_timeout", r"statement timeout|QueryCanceled|canceling statement", "warning"),
35
+ ("db_connection_error", r"could not connect to server|connection to server .* failed|too many clients|"
36
+ r"remaining connection slots|server closed the connection", "warning"),
37
+ ("redis_error", r"OOM command not allowed|MISCONF|Error \d+ connecting to|redis\.exceptions\.\w*Error", "warning"),
38
+ ("timeout", r"\b504\b|Gateway Time-?out|ReadTimeout|timed out", "warning"),
39
+ ("error", r"\b(?:ERROR|CRITICAL|FATAL|PANIC)\b|Traceback \(most recent call last\)", "warning"),
40
+ ]
41
+ _RULES = [(k, re.compile(p, re.IGNORECASE), s) for k, p, s in RULES]
42
+ PROBLEM_KINDS = {k for k, _, _ in RULES}
43
+
44
+
45
+ # ---------------------------------------------------------------------------
46
+ # Sources
47
+ # ---------------------------------------------------------------------------
48
+
49
+ def _parse_source(spec: str) -> tuple[str, str]:
50
+ """'docker:api' / 'file:/x.log' / '/var/log/x.log' / 'api' -> (kind, target)."""
51
+ if spec.startswith("docker:"):
52
+ return "docker", spec[7:]
53
+ if spec.startswith("file:"):
54
+ return "file", spec[5:]
55
+ looks_like_path = any(c in spec for c in "/\\") or spec.endswith((".log", ".txt")) or os.path.exists(spec)
56
+ return ("file", os.path.expanduser(spec)) if looks_like_path else ("docker", spec)
57
+
58
+
59
+ def _label(spec: str) -> str:
60
+ kind, target = _parse_source(spec)
61
+ return os.path.basename(target) if kind == "file" else f"docker:{target}"
62
+
63
+
64
+ def _resolve(source: str) -> tuple[str, str, str] | None:
65
+ """Match a user-provided name against configured sources. Returns (label, kind, target)."""
66
+ for spec in get_config().log_sources:
67
+ kind, target = _parse_source(spec)
68
+ if source in (spec, target, os.path.basename(target)):
69
+ return spec, kind, target
70
+ return None
71
+
72
+
73
+ def _parse_ts(line: str) -> tuple[dt.datetime | None, str]:
74
+ """Return (timestamp, line without docker's timestamp prefix)."""
75
+ m = _DOCKER_TS.match(line)
76
+ if m:
77
+ frac = (m.group(2) or "0")[:6].ljust(6, "0")
78
+ ts = dt.datetime.fromisoformat(f"{m.group(1)}.{frac}+00:00")
79
+ return ts, line[m.end():]
80
+ m = _GENERIC_TS.search(line[:80])
81
+ if m:
82
+ date, clock, frac, tz = m.groups()
83
+ frac = (frac or "0")[:6].ljust(6, "0")
84
+ tz = "+00:00" if tz in ("Z", "UTC") else (tz if tz and ":" in tz else (f"{tz[:3]}:{tz[3:]}" if tz else ""))
85
+ try:
86
+ return to_utc(dt.datetime.fromisoformat(f"{date}T{clock}.{frac}{tz}")), line
87
+ except ValueError:
88
+ pass
89
+ return None, line
90
+
91
+
92
+ def _read_file(path: str, max_lines: int) -> list[str]:
93
+ with open(path, "rb") as f:
94
+ f.seek(0, os.SEEK_END)
95
+ start = max(0, f.tell() - MAX_FILE_BYTES)
96
+ f.seek(start)
97
+ data = f.read()
98
+ lines = data.decode("utf-8", errors="replace").splitlines()
99
+ if start > 0:
100
+ lines = lines[1:] # first line is probably partial
101
+ return lines[-max_lines:]
102
+
103
+
104
+ def _read_docker(container: str, max_lines: int, since_minutes: float | None) -> list[str]:
105
+ if not shutil.which("docker"):
106
+ raise RuntimeError("docker CLI not found on PATH")
107
+ cmd = ["docker", "logs", "--timestamps", "--tail", str(max_lines)]
108
+ if since_minutes:
109
+ cmd += ["--since", f"{int(since_minutes * 60)}s"]
110
+ proc = subprocess.run(cmd + [container], capture_output=True, timeout=6,
111
+ text=True, encoding="utf-8", errors="replace")
112
+ if proc.returncode != 0:
113
+ raise RuntimeError(proc.stderr.strip()[:300] or f"docker logs exited {proc.returncode}")
114
+ # Container stdout and stderr arrive separately; timestamps let us re-interleave them.
115
+ return sorted(proc.stdout.splitlines() + proc.stderr.splitlines())
116
+
117
+
118
+ def _read(kind: str, target: str, max_lines: int, since_minutes: float | None = None) -> list[dict]:
119
+ raw = _read_file(target, max_lines) if kind == "file" else _read_docker(target, max_lines, since_minutes)
120
+ out, last_ts, trailing = [], None, []
121
+ for line in raw:
122
+ ts, text = _parse_ts(line)
123
+ if ts:
124
+ last_ts, trailing = ts, []
125
+ entry = {"ts": ts or last_ts, "line": text[:MAX_LINE]} # tracebacks inherit the last timestamp
126
+ if not ts:
127
+ trailing.append(entry)
128
+ out.append(entry)
129
+ if kind == "file" and trailing:
130
+ # Lines after the last timestamp were written by the file's last modification at the latest.
131
+ # This dates lines printed without a timestamp, like Celery's "worker: Warm shutdown (MainProcess)",
132
+ # which Celery writes straight to stdout (it never goes through logging or --logfile).
133
+ mtime = to_utc(os.path.getmtime(target))
134
+ for entry in trailing:
135
+ entry["ts"] = max(entry["ts"], mtime) if entry["ts"] else mtime
136
+ return out
137
+
138
+
139
+ # ---------------------------------------------------------------------------
140
+ # Tools
141
+ # ---------------------------------------------------------------------------
142
+
143
+ def tail_logs(source: str, lines: int = 100) -> dict:
144
+ cfg = get_config()
145
+ if not cfg.log_sources:
146
+ return skipped("LOG_SOURCES not set")
147
+ resolved = _resolve(source)
148
+ if not resolved:
149
+ return {"error": f"Unknown log source '{source}'.", "configured_sources": cfg.log_sources}
150
+ label, kind, target = resolved
151
+ entries = _read(kind, target, max(1, min(int(lines), 500)))
152
+ return {"source": label, "count": len(entries), "lines": entries}
153
+
154
+
155
+ def search_logs(pattern: str, since_minutes: int = 30, source: str | None = None) -> dict:
156
+ cfg = get_config()
157
+ if not cfg.log_sources:
158
+ return skipped("LOG_SOURCES not set")
159
+ try:
160
+ regex = re.compile(pattern, re.IGNORECASE)
161
+ except re.error:
162
+ regex = re.compile(re.escape(pattern), re.IGNORECASE)
163
+ since = utcnow() - dt.timedelta(minutes=max(1, min(int(since_minutes), 24 * 60)))
164
+ targets = [_resolve(source)] if source else [(s, *_parse_source(s)) for s in cfg.log_sources]
165
+ if source and not targets[0]:
166
+ return {"error": f"Unknown log source '{source}'.", "configured_sources": cfg.log_sources}
167
+
168
+ matches, errors = [], {}
169
+ for label, kind, target in targets:
170
+ try:
171
+ entries = _read(kind, target, 20_000, since_minutes)
172
+ except Exception as e:
173
+ errors[label] = f"{type(e).__name__}: {e}"
174
+ continue
175
+ for e in entries:
176
+ if (e["ts"] is None or e["ts"] >= since) and regex.search(e["line"]):
177
+ matches.append({"source": label, **e})
178
+ matches.sort(key=lambda m: m["ts"] or since)
179
+ return {"pattern": pattern, "since": since, "match_count": len(matches),
180
+ "matches": matches[-200:], "errors": errors or None}
181
+
182
+
183
+ def classify(line: str) -> tuple[str, str] | None:
184
+ for kind, rx, severity in _RULES:
185
+ if rx.search(line):
186
+ return kind, severity
187
+ return None
188
+
189
+
190
+ def recent_problems(since_minutes: int = 30) -> dict:
191
+ """Classify recent log lines into problem kinds, aggregated per source and kind."""
192
+ cfg = get_config()
193
+ if not cfg.log_sources:
194
+ return skipped("LOG_SOURCES not set")
195
+ now = utcnow()
196
+ since = now - dt.timedelta(minutes=since_minutes)
197
+ spike_since = now - dt.timedelta(minutes=5)
198
+ groups: dict[tuple[str, str], dict] = {}
199
+ recent_errors: dict[str, list] = defaultdict(list)
200
+ errors, last_activity = {}, {}
201
+ for spec in cfg.log_sources:
202
+ kind, target = _parse_source(spec)
203
+ try:
204
+ entries = _read(kind, target, 20_000, since_minutes)
205
+ except Exception as e:
206
+ errors[spec] = f"{type(e).__name__}: {e}"
207
+ continue
208
+ last_activity[_label(spec)] = max((e["ts"] for e in entries if e["ts"]), default=None)
209
+ for e in entries:
210
+ if e["ts"] and e["ts"] < since:
211
+ continue
212
+ hit = classify(e["line"])
213
+ if not hit:
214
+ continue
215
+ k, sev = hit
216
+ g = groups.setdefault((spec, k), {"source": spec, "kind": k, "severity": sev, "count": 0,
217
+ "first": e["ts"], "last": e["ts"], "sample": e["line"][:300]})
218
+ g["count"] += 1
219
+ g["last"] = e["ts"] or g["last"]
220
+ if e["ts"] and e["ts"] >= spike_since:
221
+ recent_errors[spec].append(e["ts"])
222
+
223
+ events = []
224
+ for g in groups.values():
225
+ events.append(event(g["first"], SOURCE, g["kind"],
226
+ f"[{_label(g['source'])}] {g['count']}x {g['kind']} (last {g['last'] and g['last'].isoformat(timespec='seconds')}): {g['sample']}",
227
+ g["severity"], log_source=g["source"], count=g["count"], last=g["last"]))
228
+ for spec, stamps in recent_errors.items():
229
+ if len(stamps) >= cfg.log_error_spike:
230
+ events.append(event(min(stamps), SOURCE, "error_spike",
231
+ f"[{_label(spec)}] {len(stamps)} problem lines in the last 5 minutes "
232
+ f"(threshold {cfg.log_error_spike})", "warning", log_source=spec))
233
+ summary = sorted(groups.values(), key=lambda g: (g["first"] or now))
234
+ return {"since_minutes": since_minutes, "problems": summary, "last_activity": last_activity,
235
+ "errors": errors or None, "events": events}