stackdoctor 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- stackdoctor/__init__.py +0 -0
- stackdoctor/__main__.py +3 -0
- stackdoctor/checks/__init__.py +29 -0
- stackdoctor/checks/celery.py +319 -0
- stackdoctor/checks/logs.py +235 -0
- stackdoctor/checks/postgres.py +202 -0
- stackdoctor/checks/redis.py +154 -0
- stackdoctor/config.py +80 -0
- stackdoctor/diagnose.py +248 -0
- stackdoctor/safety.py +331 -0
- stackdoctor/server.py +164 -0
- stackdoctor-0.1.0.dist-info/METADATA +341 -0
- stackdoctor-0.1.0.dist-info/RECORD +16 -0
- stackdoctor-0.1.0.dist-info/WHEEL +4 -0
- stackdoctor-0.1.0.dist-info/entry_points.txt +2 -0
- stackdoctor-0.1.0.dist-info/licenses/LICENSE +21 -0
stackdoctor/__init__.py
ADDED
|
File without changes
|
stackdoctor/__main__.py
ADDED
|
@@ -0,0 +1,29 @@
|
|
|
1
|
+
"""Checks. Every check is a plain sync function returning a dict.
|
|
2
|
+
|
|
3
|
+
Checks may include an `events` list: notable things with a timestamp, which
|
|
4
|
+
diagnose() merges into one cross-system timeline.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
import datetime as dt
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
def utcnow() -> dt.datetime:
|
|
13
|
+
return dt.datetime.now(dt.timezone.utc)
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
def to_utc(ts: dt.datetime | float | None) -> dt.datetime | None:
|
|
17
|
+
if ts is None:
|
|
18
|
+
return None
|
|
19
|
+
if isinstance(ts, (int, float)):
|
|
20
|
+
return dt.datetime.fromtimestamp(ts, dt.timezone.utc)
|
|
21
|
+
if ts.tzinfo is None:
|
|
22
|
+
ts = ts.astimezone() # naive = local time
|
|
23
|
+
return ts.astimezone(dt.timezone.utc)
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def event(ts, source: str, kind: str, detail: str, severity: str = "info", **extra) -> dict:
|
|
27
|
+
"""A timeline event. `kind` is a stable keyword used by the cause → effect rules."""
|
|
28
|
+
return {"ts": to_utc(ts) or utcnow(), "source": source, "kind": kind,
|
|
29
|
+
"severity": severity, "detail": detail, **extra}
|
|
@@ -0,0 +1,319 @@
|
|
|
1
|
+
"""Celery checks without Flower: the inspect API plus reading the broker directly.
|
|
2
|
+
|
|
3
|
+
Only ping/active/reserved/active_queues/stats/query_task are used. Nothing here
|
|
4
|
+
sends, revokes, retries or shuts down tasks or workers.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
import contextlib
|
|
10
|
+
import datetime as dt
|
|
11
|
+
import importlib
|
|
12
|
+
import json
|
|
13
|
+
import os
|
|
14
|
+
import re
|
|
15
|
+
import sys
|
|
16
|
+
import threading
|
|
17
|
+
import time
|
|
18
|
+
|
|
19
|
+
from celery import Celery
|
|
20
|
+
|
|
21
|
+
from ..config import get_config, skipped
|
|
22
|
+
from ..safety import preview
|
|
23
|
+
from . import event, to_utc
|
|
24
|
+
from .redis import client as redis_client, glob_escape
|
|
25
|
+
|
|
26
|
+
SOURCE = "celery"
|
|
27
|
+
_TASK_ID_RE = re.compile(r"^[A-Za-z0-9_.:-]{1,128}$")
|
|
28
|
+
_META_PREFIX = "celery-task-meta-"
|
|
29
|
+
_HISTORY: dict[str, list[tuple[float, int]]] = {} # queue -> recent (time, length) samples
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
_app = None
|
|
33
|
+
_app_lock = threading.Lock() # diagnose() calls checks from several threads at once
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def get_app():
|
|
37
|
+
global _app
|
|
38
|
+
with _app_lock:
|
|
39
|
+
if _app is None:
|
|
40
|
+
_app = _load_app()
|
|
41
|
+
return _app
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def _load_app():
|
|
45
|
+
"""Load CELERY_APP (`pkg.module:app` or `pkg.module`), else a bare app on the broker."""
|
|
46
|
+
cfg = get_config()
|
|
47
|
+
if cfg.celery_app:
|
|
48
|
+
if os.getcwd() not in sys.path:
|
|
49
|
+
sys.path.insert(0, os.getcwd())
|
|
50
|
+
mod_name, _, attr = cfg.celery_app.partition(":")
|
|
51
|
+
# Importing user code must never print to stdout: that is the MCP channel.
|
|
52
|
+
with contextlib.redirect_stdout(sys.stderr):
|
|
53
|
+
module = importlib.import_module(mod_name)
|
|
54
|
+
if attr:
|
|
55
|
+
return getattr(module, attr)
|
|
56
|
+
for name in ("app", "celery", "celery_app"):
|
|
57
|
+
if hasattr(module, name):
|
|
58
|
+
return getattr(module, name)
|
|
59
|
+
raise ImportError(f"No Celery app found in {mod_name} (use module:attr)")
|
|
60
|
+
if cfg.celery_broker_url:
|
|
61
|
+
return Celery("stackdoctor", broker=cfg.celery_broker_url,
|
|
62
|
+
backend=cfg.celery_result_backend)
|
|
63
|
+
return None
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
def _not_configured():
|
|
67
|
+
cfg = get_config()
|
|
68
|
+
if not (cfg.celery_app or cfg.celery_broker_url):
|
|
69
|
+
return skipped("CELERY_APP / CELERY_BROKER_URL not set")
|
|
70
|
+
return None
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def _inspect(app, destination=None):
|
|
74
|
+
timeout = get_config().celery_inspect_timeout_s
|
|
75
|
+
return app.control.inspect(timeout=timeout, destination=destination,
|
|
76
|
+
limit=len(destination) if destination else None)
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
def _task_summary(t: dict) -> dict:
|
|
80
|
+
started = t.get("time_start")
|
|
81
|
+
return {
|
|
82
|
+
"id": t.get("id"), "name": t.get("name"),
|
|
83
|
+
"args": preview(t.get("args"), 120), "kwargs": preview(t.get("kwargs"), 120),
|
|
84
|
+
"started": to_utc(started),
|
|
85
|
+
"runtime_s": round(time.time() - started, 1) if started else None,
|
|
86
|
+
"queue": (t.get("delivery_info") or {}).get("routing_key"),
|
|
87
|
+
}
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
def workers() -> dict:
|
|
91
|
+
if (s := _not_configured()):
|
|
92
|
+
return s
|
|
93
|
+
cfg, app = get_config(), get_app()
|
|
94
|
+
ping = _inspect(app).ping() or {}
|
|
95
|
+
events = []
|
|
96
|
+
if not ping:
|
|
97
|
+
events.append(event(None, SOURCE, "no_workers",
|
|
98
|
+
"No Celery workers replied to ping", "critical"))
|
|
99
|
+
return {"alive_count": 0, "workers": [], "events": events}
|
|
100
|
+
|
|
101
|
+
names = sorted(ping)
|
|
102
|
+
insp = _inspect(app, destination=names)
|
|
103
|
+
active, reserved = insp.active() or {}, insp.reserved() or {}
|
|
104
|
+
queues, stats = insp.active_queues() or {}, insp.stats() or {}
|
|
105
|
+
|
|
106
|
+
out = []
|
|
107
|
+
for name in names:
|
|
108
|
+
st = stats.get(name, {})
|
|
109
|
+
concurrency = (st.get("pool") or {}).get("max-concurrency")
|
|
110
|
+
act = [_task_summary(t) for t in active.get(name, [])]
|
|
111
|
+
res = [_task_summary(t) for t in reserved.get(name, [])]
|
|
112
|
+
out.append({
|
|
113
|
+
"name": name, "alive": True,
|
|
114
|
+
"queues": [q.get("name") for q in queues.get(name, [])],
|
|
115
|
+
"concurrency": concurrency,
|
|
116
|
+
"active_count": len(act), "reserved_count": len(res),
|
|
117
|
+
"active": act[:20], "reserved": res[:10],
|
|
118
|
+
"processed_total": sum((st.get("total") or {}).values()),
|
|
119
|
+
})
|
|
120
|
+
for t in act:
|
|
121
|
+
if t["runtime_s"] and t["runtime_s"] >= cfg.long_task_s:
|
|
122
|
+
events.append(event(t["started"], SOURCE, "task_long_running",
|
|
123
|
+
f"{t['name']}[{t['id']}] running {t['runtime_s']}s on {name}",
|
|
124
|
+
"warning", task_id=t["id"]))
|
|
125
|
+
if concurrency and len(act) >= concurrency:
|
|
126
|
+
events.append(event(None, SOURCE, "workers_saturated",
|
|
127
|
+
f"{name}: all {concurrency} slots busy, {len(res)} reserved", "warning"))
|
|
128
|
+
|
|
129
|
+
if cfg.expected_workers and len(names) < cfg.expected_workers:
|
|
130
|
+
events.append(event(None, SOURCE, "worker_missing",
|
|
131
|
+
f"Only {len(names)} of {cfg.expected_workers} expected workers replied",
|
|
132
|
+
"critical"))
|
|
133
|
+
return {"alive_count": len(names), "workers": out, "events": events}
|
|
134
|
+
|
|
135
|
+
|
|
136
|
+
def _queue_names(app, given: list[str] | None) -> list[str]:
|
|
137
|
+
if given:
|
|
138
|
+
return given
|
|
139
|
+
names = set(get_config().celery_queues)
|
|
140
|
+
names.add(app.conf.task_default_queue or "celery")
|
|
141
|
+
for q in app.conf.task_queues or []:
|
|
142
|
+
names.add(getattr(q, "name", q))
|
|
143
|
+
for route in (app.conf.task_routes or {}).values() if isinstance(app.conf.task_routes, dict) else []:
|
|
144
|
+
if isinstance(route, dict) and route.get("queue"):
|
|
145
|
+
names.add(route["queue"])
|
|
146
|
+
return sorted(names)
|
|
147
|
+
|
|
148
|
+
|
|
149
|
+
def _redis_queue_lengths(app, url: str, names: list[str]) -> dict:
|
|
150
|
+
opts = app.conf.broker_transport_options or {}
|
|
151
|
+
steps = opts.get("priority_steps") or [0, 3, 6, 9]
|
|
152
|
+
sep = opts.get("sep", "\x06\x16")
|
|
153
|
+
prefix = opts.get("global_keyprefix", "")
|
|
154
|
+
r = redis_client(url)
|
|
155
|
+
result = {}
|
|
156
|
+
for q in names:
|
|
157
|
+
by_priority = {}
|
|
158
|
+
for step in steps:
|
|
159
|
+
key = prefix + (f"{q}{sep}{step}" if step else q)
|
|
160
|
+
n = r.cmd("LLEN", key)
|
|
161
|
+
if n:
|
|
162
|
+
by_priority[str(step)] = n
|
|
163
|
+
# Steps we don't know about (custom priority_steps on the producer side).
|
|
164
|
+
extra, _ = r.scan_iter(glob_escape(prefix + q + sep) + "*", 50)
|
|
165
|
+
for key in extra:
|
|
166
|
+
step = key.rsplit(sep, 1)[-1]
|
|
167
|
+
if step.isdigit() and int(step) not in steps:
|
|
168
|
+
by_priority[step] = r.cmd("LLEN", key)
|
|
169
|
+
result[q] = {"messages": sum(by_priority.values()),
|
|
170
|
+
"by_priority": by_priority if len(by_priority) > 1 else None}
|
|
171
|
+
unacked = r.cmd("HLEN", prefix + "unacked")
|
|
172
|
+
return {"queues": result, "unacked_total": unacked}
|
|
173
|
+
|
|
174
|
+
|
|
175
|
+
def _amqp_queue_lengths(app, names: list[str]) -> dict:
|
|
176
|
+
result = {}
|
|
177
|
+
for q in names:
|
|
178
|
+
with app.connection_for_read() as conn:
|
|
179
|
+
try:
|
|
180
|
+
# passive=True only checks the queue; it never creates anything.
|
|
181
|
+
_, messages, consumers = conn.default_channel.queue_declare(queue=q, passive=True)
|
|
182
|
+
result[q] = {"messages": messages, "consumers": consumers}
|
|
183
|
+
except conn.channel_errors:
|
|
184
|
+
result[q] = {"error": "queue does not exist"}
|
|
185
|
+
return {"queues": result}
|
|
186
|
+
|
|
187
|
+
|
|
188
|
+
def queue_lengths(queues: list[str] | None = None) -> dict:
|
|
189
|
+
if (s := _not_configured()):
|
|
190
|
+
return s
|
|
191
|
+
cfg, app = get_config(), get_app()
|
|
192
|
+
url = app.conf.broker_url or cfg.celery_broker_url or ""
|
|
193
|
+
names = _queue_names(app, queues)
|
|
194
|
+
if url.startswith(("redis://", "rediss://", "unix://")):
|
|
195
|
+
out = _redis_queue_lengths(app, url, names)
|
|
196
|
+
elif url.startswith(("amqp://", "amqps://", "pyamqp://")):
|
|
197
|
+
out = _amqp_queue_lengths(app, names)
|
|
198
|
+
else:
|
|
199
|
+
return {"error": f"Queue lengths are supported for Redis and RabbitMQ brokers, not {url.split(':')[0]}"}
|
|
200
|
+
|
|
201
|
+
now, events = time.time(), []
|
|
202
|
+
for q, info in out["queues"].items():
|
|
203
|
+
n = info.get("messages")
|
|
204
|
+
if n is None:
|
|
205
|
+
continue
|
|
206
|
+
hist = _HISTORY.setdefault(q, [])
|
|
207
|
+
hist.append((now, n))
|
|
208
|
+
del hist[:-10]
|
|
209
|
+
older = [(t, m) for t, m in hist if now - t >= 5]
|
|
210
|
+
if older:
|
|
211
|
+
t0, m0 = older[-1]
|
|
212
|
+
info["change_since_last_check"] = {"seconds_ago": round(now - t0), "delta": n - m0}
|
|
213
|
+
if n >= cfg.queue_threshold:
|
|
214
|
+
events.append(event(None, SOURCE, "queue_backlog",
|
|
215
|
+
f"Queue '{q}' has {n} waiting messages (threshold {cfg.queue_threshold})",
|
|
216
|
+
"warning", queue=q, messages=n))
|
|
217
|
+
if info.get("consumers") == 0 and n:
|
|
218
|
+
events.append(event(None, SOURCE, "queue_no_consumer",
|
|
219
|
+
f"Queue '{q}' has {n} messages and no consumers", "critical", queue=q))
|
|
220
|
+
out["events"] = events
|
|
221
|
+
return out
|
|
222
|
+
|
|
223
|
+
|
|
224
|
+
def _backend(app) -> tuple[str | None, str | None]:
|
|
225
|
+
"""Return (redis_url, None) if results are readable, else (None, reason)."""
|
|
226
|
+
cfg = get_config()
|
|
227
|
+
url = cfg.celery_result_backend or app.conf.result_backend
|
|
228
|
+
if not url:
|
|
229
|
+
return None, "no result backend is configured (CELERY_RESULT_BACKEND / app.conf.result_backend)"
|
|
230
|
+
if not str(url).startswith(("redis://", "rediss://")):
|
|
231
|
+
return None, f"the result backend is '{str(url).split(':')[0]}'; stackdoctor only reads Redis result backends"
|
|
232
|
+
return url, None
|
|
233
|
+
|
|
234
|
+
|
|
235
|
+
def failed_tasks(limit: int = 20) -> dict:
|
|
236
|
+
if (s := _not_configured()):
|
|
237
|
+
return s
|
|
238
|
+
app = get_app()
|
|
239
|
+
url, reason = _backend(app)
|
|
240
|
+
if reason:
|
|
241
|
+
return {"visible": False, "reason": f"Task failures are not visible because {reason}. "
|
|
242
|
+
"Check logs for task errors instead."}
|
|
243
|
+
r = redis_client(url)
|
|
244
|
+
keys, complete = r.scan_iter(_META_PREFIX + "*", 2000)
|
|
245
|
+
if not keys:
|
|
246
|
+
why = ("task_ignore_result is on" if app.conf.task_ignore_result
|
|
247
|
+
else f"no results are stored: results may have expired (result_expires="
|
|
248
|
+
f"{app.conf.result_expires}) or tasks use ignore_result=True")
|
|
249
|
+
return {"visible": False, "reason": f"Task failures are not visible because {why}. "
|
|
250
|
+
"Check logs for task errors instead."}
|
|
251
|
+
failures = []
|
|
252
|
+
for i in range(0, len(keys), 200):
|
|
253
|
+
for raw in r.cmd("MGET", *keys[i:i + 200]):
|
|
254
|
+
try:
|
|
255
|
+
meta = json.loads(raw) if raw else None
|
|
256
|
+
except ValueError:
|
|
257
|
+
continue # pickle or other serializer
|
|
258
|
+
if meta and meta.get("status") == "FAILURE":
|
|
259
|
+
failures.append(meta)
|
|
260
|
+
failures.sort(key=lambda m: m.get("date_done") or "", reverse=True)
|
|
261
|
+
events, out = [], []
|
|
262
|
+
for m in failures[: max(1, min(int(limit), 100))]:
|
|
263
|
+
res = m.get("result") or {}
|
|
264
|
+
tb_tail = (m.get("traceback") or "").strip().splitlines()[-3:]
|
|
265
|
+
item = {"task_id": m.get("task_id"), "name": m.get("name"),
|
|
266
|
+
"date_done": m.get("date_done"),
|
|
267
|
+
"exception": preview(f"{res.get('exc_type')}: {res.get('exc_message')}", 300)
|
|
268
|
+
if isinstance(res, dict) else preview(res, 300),
|
|
269
|
+
"traceback_tail": [preview(l, 200) for l in tb_tail]}
|
|
270
|
+
out.append(item)
|
|
271
|
+
events.append(event(_parse_iso(m.get("date_done")), SOURCE, "task_failed",
|
|
272
|
+
f"{item['name'] or 'task'}[{item['task_id']}] failed: {item['exception']}",
|
|
273
|
+
"warning", task_id=item["task_id"]))
|
|
274
|
+
return {"visible": True, "results_scanned": len(keys), "scan_complete": complete,
|
|
275
|
+
"failed_count": len(failures), "failed": out, "events": events}
|
|
276
|
+
|
|
277
|
+
|
|
278
|
+
def _parse_iso(value):
|
|
279
|
+
try:
|
|
280
|
+
return dt.datetime.fromisoformat(value) if value else None
|
|
281
|
+
except ValueError:
|
|
282
|
+
return None
|
|
283
|
+
|
|
284
|
+
|
|
285
|
+
def task_details(task_id: str) -> dict:
|
|
286
|
+
if (s := _not_configured()):
|
|
287
|
+
return s
|
|
288
|
+
if not _TASK_ID_RE.match(task_id or ""):
|
|
289
|
+
return {"error": "Invalid task id"}
|
|
290
|
+
app = get_app()
|
|
291
|
+
out: dict = {"task_id": task_id}
|
|
292
|
+
url, reason = _backend(app)
|
|
293
|
+
if url:
|
|
294
|
+
raw = redis_client(url).cmd("GET", _META_PREFIX + task_id)
|
|
295
|
+
if raw:
|
|
296
|
+
try:
|
|
297
|
+
meta = json.loads(raw)
|
|
298
|
+
out["result_backend"] = {
|
|
299
|
+
"status": meta.get("status"), "name": meta.get("name"),
|
|
300
|
+
"date_done": meta.get("date_done"), "retries": meta.get("retries"),
|
|
301
|
+
"worker": meta.get("worker"),
|
|
302
|
+
"args": preview(meta.get("args"), 200), "kwargs": preview(meta.get("kwargs"), 200),
|
|
303
|
+
"result": preview(meta.get("result"), 300),
|
|
304
|
+
"traceback_tail": [preview(l, 200) for l in
|
|
305
|
+
(meta.get("traceback") or "").strip().splitlines()[-5:]],
|
|
306
|
+
}
|
|
307
|
+
except ValueError:
|
|
308
|
+
out["result_backend"] = {"note": "result is not JSON-encoded"}
|
|
309
|
+
else:
|
|
310
|
+
out["result_backend"] = {"note": "no stored result (pending, expired, or ignore_result)"}
|
|
311
|
+
else:
|
|
312
|
+
out["result_backend"] = {"note": f"not readable: {reason}"}
|
|
313
|
+
# Ask workers whether they currently hold the task (active/reserved/scheduled).
|
|
314
|
+
found = _inspect(app).query_task(task_id) or {}
|
|
315
|
+
out["on_workers"] = {w: {tid: {"state": state, "name": info.get("name"),
|
|
316
|
+
"args": preview(info.get("args"), 120)}
|
|
317
|
+
for tid, (state, info) in tasks.items()}
|
|
318
|
+
for w, tasks in found.items() if tasks}
|
|
319
|
+
return out
|
|
@@ -0,0 +1,235 @@
|
|
|
1
|
+
"""Log checks for files and docker containers listed in LOG_SOURCES.
|
|
2
|
+
|
|
3
|
+
Only configured sources can be read, so the AI can't use this to read arbitrary files.
|
|
4
|
+
"""
|
|
5
|
+
|
|
6
|
+
from __future__ import annotations
|
|
7
|
+
|
|
8
|
+
import datetime as dt
|
|
9
|
+
import os
|
|
10
|
+
import re
|
|
11
|
+
import shutil
|
|
12
|
+
import subprocess
|
|
13
|
+
from collections import defaultdict
|
|
14
|
+
|
|
15
|
+
from ..config import get_config, skipped
|
|
16
|
+
from . import event, to_utc, utcnow
|
|
17
|
+
|
|
18
|
+
SOURCE = "logs"
|
|
19
|
+
MAX_LINE = 2000
|
|
20
|
+
MAX_FILE_BYTES = 5_000_000
|
|
21
|
+
|
|
22
|
+
_DOCKER_TS = re.compile(r"^(\d{4}-\d\d-\d\dT\d\d:\d\d:\d\d)(?:\.(\d+))?Z\s?")
|
|
23
|
+
_GENERIC_TS = re.compile(
|
|
24
|
+
r"(\d{4}-\d\d-\d\d)[T ](\d\d:\d\d:\d\d)(?:[.,](\d+))?\s*(Z|UTC|[+-]\d\d:?\d\d)?")
|
|
25
|
+
|
|
26
|
+
# (kind, pattern, severity) — first match wins, so specific rules come first.
|
|
27
|
+
RULES = [
|
|
28
|
+
("worker_lost", r"WorkerLostError|worker lost|exited with 'signal 9|exited prematurely", "critical"),
|
|
29
|
+
("worker_shutdown", r"\b(?:warm|cold) shutdown\b", "warning"), # Celery: "worker: Warm shutdown (MainProcess)"
|
|
30
|
+
("task_timeout", r"SoftTimeLimitExceeded|TimeLimitExceeded|(?:hard|soft) time limit", "warning"),
|
|
31
|
+
("task_failed", r"Task [\w.]+\[[\w-]+\] raised unexpected|Task [\w.]+\[[\w-]+\] failed", "warning"),
|
|
32
|
+
("db_lock_wait", r"still waiting for \w*Lock", "warning"),
|
|
33
|
+
("db_lock_error", r"lock timeout|deadlock detected|could not obtain lock|LockNotAvailable", "warning"),
|
|
34
|
+
("db_timeout", r"statement timeout|QueryCanceled|canceling statement", "warning"),
|
|
35
|
+
("db_connection_error", r"could not connect to server|connection to server .* failed|too many clients|"
|
|
36
|
+
r"remaining connection slots|server closed the connection", "warning"),
|
|
37
|
+
("redis_error", r"OOM command not allowed|MISCONF|Error \d+ connecting to|redis\.exceptions\.\w*Error", "warning"),
|
|
38
|
+
("timeout", r"\b504\b|Gateway Time-?out|ReadTimeout|timed out", "warning"),
|
|
39
|
+
("error", r"\b(?:ERROR|CRITICAL|FATAL|PANIC)\b|Traceback \(most recent call last\)", "warning"),
|
|
40
|
+
]
|
|
41
|
+
_RULES = [(k, re.compile(p, re.IGNORECASE), s) for k, p, s in RULES]
|
|
42
|
+
PROBLEM_KINDS = {k for k, _, _ in RULES}
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
# ---------------------------------------------------------------------------
|
|
46
|
+
# Sources
|
|
47
|
+
# ---------------------------------------------------------------------------
|
|
48
|
+
|
|
49
|
+
def _parse_source(spec: str) -> tuple[str, str]:
|
|
50
|
+
"""'docker:api' / 'file:/x.log' / '/var/log/x.log' / 'api' -> (kind, target)."""
|
|
51
|
+
if spec.startswith("docker:"):
|
|
52
|
+
return "docker", spec[7:]
|
|
53
|
+
if spec.startswith("file:"):
|
|
54
|
+
return "file", spec[5:]
|
|
55
|
+
looks_like_path = any(c in spec for c in "/\\") or spec.endswith((".log", ".txt")) or os.path.exists(spec)
|
|
56
|
+
return ("file", os.path.expanduser(spec)) if looks_like_path else ("docker", spec)
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def _label(spec: str) -> str:
|
|
60
|
+
kind, target = _parse_source(spec)
|
|
61
|
+
return os.path.basename(target) if kind == "file" else f"docker:{target}"
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def _resolve(source: str) -> tuple[str, str, str] | None:
|
|
65
|
+
"""Match a user-provided name against configured sources. Returns (label, kind, target)."""
|
|
66
|
+
for spec in get_config().log_sources:
|
|
67
|
+
kind, target = _parse_source(spec)
|
|
68
|
+
if source in (spec, target, os.path.basename(target)):
|
|
69
|
+
return spec, kind, target
|
|
70
|
+
return None
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def _parse_ts(line: str) -> tuple[dt.datetime | None, str]:
|
|
74
|
+
"""Return (timestamp, line without docker's timestamp prefix)."""
|
|
75
|
+
m = _DOCKER_TS.match(line)
|
|
76
|
+
if m:
|
|
77
|
+
frac = (m.group(2) or "0")[:6].ljust(6, "0")
|
|
78
|
+
ts = dt.datetime.fromisoformat(f"{m.group(1)}.{frac}+00:00")
|
|
79
|
+
return ts, line[m.end():]
|
|
80
|
+
m = _GENERIC_TS.search(line[:80])
|
|
81
|
+
if m:
|
|
82
|
+
date, clock, frac, tz = m.groups()
|
|
83
|
+
frac = (frac or "0")[:6].ljust(6, "0")
|
|
84
|
+
tz = "+00:00" if tz in ("Z", "UTC") else (tz if tz and ":" in tz else (f"{tz[:3]}:{tz[3:]}" if tz else ""))
|
|
85
|
+
try:
|
|
86
|
+
return to_utc(dt.datetime.fromisoformat(f"{date}T{clock}.{frac}{tz}")), line
|
|
87
|
+
except ValueError:
|
|
88
|
+
pass
|
|
89
|
+
return None, line
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
def _read_file(path: str, max_lines: int) -> list[str]:
|
|
93
|
+
with open(path, "rb") as f:
|
|
94
|
+
f.seek(0, os.SEEK_END)
|
|
95
|
+
start = max(0, f.tell() - MAX_FILE_BYTES)
|
|
96
|
+
f.seek(start)
|
|
97
|
+
data = f.read()
|
|
98
|
+
lines = data.decode("utf-8", errors="replace").splitlines()
|
|
99
|
+
if start > 0:
|
|
100
|
+
lines = lines[1:] # first line is probably partial
|
|
101
|
+
return lines[-max_lines:]
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
def _read_docker(container: str, max_lines: int, since_minutes: float | None) -> list[str]:
|
|
105
|
+
if not shutil.which("docker"):
|
|
106
|
+
raise RuntimeError("docker CLI not found on PATH")
|
|
107
|
+
cmd = ["docker", "logs", "--timestamps", "--tail", str(max_lines)]
|
|
108
|
+
if since_minutes:
|
|
109
|
+
cmd += ["--since", f"{int(since_minutes * 60)}s"]
|
|
110
|
+
proc = subprocess.run(cmd + [container], capture_output=True, timeout=6,
|
|
111
|
+
text=True, encoding="utf-8", errors="replace")
|
|
112
|
+
if proc.returncode != 0:
|
|
113
|
+
raise RuntimeError(proc.stderr.strip()[:300] or f"docker logs exited {proc.returncode}")
|
|
114
|
+
# Container stdout and stderr arrive separately; timestamps let us re-interleave them.
|
|
115
|
+
return sorted(proc.stdout.splitlines() + proc.stderr.splitlines())
|
|
116
|
+
|
|
117
|
+
|
|
118
|
+
def _read(kind: str, target: str, max_lines: int, since_minutes: float | None = None) -> list[dict]:
|
|
119
|
+
raw = _read_file(target, max_lines) if kind == "file" else _read_docker(target, max_lines, since_minutes)
|
|
120
|
+
out, last_ts, trailing = [], None, []
|
|
121
|
+
for line in raw:
|
|
122
|
+
ts, text = _parse_ts(line)
|
|
123
|
+
if ts:
|
|
124
|
+
last_ts, trailing = ts, []
|
|
125
|
+
entry = {"ts": ts or last_ts, "line": text[:MAX_LINE]} # tracebacks inherit the last timestamp
|
|
126
|
+
if not ts:
|
|
127
|
+
trailing.append(entry)
|
|
128
|
+
out.append(entry)
|
|
129
|
+
if kind == "file" and trailing:
|
|
130
|
+
# Lines after the last timestamp were written by the file's last modification at the latest.
|
|
131
|
+
# This dates lines printed without a timestamp, like Celery's "worker: Warm shutdown (MainProcess)",
|
|
132
|
+
# which Celery writes straight to stdout (it never goes through logging or --logfile).
|
|
133
|
+
mtime = to_utc(os.path.getmtime(target))
|
|
134
|
+
for entry in trailing:
|
|
135
|
+
entry["ts"] = max(entry["ts"], mtime) if entry["ts"] else mtime
|
|
136
|
+
return out
|
|
137
|
+
|
|
138
|
+
|
|
139
|
+
# ---------------------------------------------------------------------------
|
|
140
|
+
# Tools
|
|
141
|
+
# ---------------------------------------------------------------------------
|
|
142
|
+
|
|
143
|
+
def tail_logs(source: str, lines: int = 100) -> dict:
|
|
144
|
+
cfg = get_config()
|
|
145
|
+
if not cfg.log_sources:
|
|
146
|
+
return skipped("LOG_SOURCES not set")
|
|
147
|
+
resolved = _resolve(source)
|
|
148
|
+
if not resolved:
|
|
149
|
+
return {"error": f"Unknown log source '{source}'.", "configured_sources": cfg.log_sources}
|
|
150
|
+
label, kind, target = resolved
|
|
151
|
+
entries = _read(kind, target, max(1, min(int(lines), 500)))
|
|
152
|
+
return {"source": label, "count": len(entries), "lines": entries}
|
|
153
|
+
|
|
154
|
+
|
|
155
|
+
def search_logs(pattern: str, since_minutes: int = 30, source: str | None = None) -> dict:
|
|
156
|
+
cfg = get_config()
|
|
157
|
+
if not cfg.log_sources:
|
|
158
|
+
return skipped("LOG_SOURCES not set")
|
|
159
|
+
try:
|
|
160
|
+
regex = re.compile(pattern, re.IGNORECASE)
|
|
161
|
+
except re.error:
|
|
162
|
+
regex = re.compile(re.escape(pattern), re.IGNORECASE)
|
|
163
|
+
since = utcnow() - dt.timedelta(minutes=max(1, min(int(since_minutes), 24 * 60)))
|
|
164
|
+
targets = [_resolve(source)] if source else [(s, *_parse_source(s)) for s in cfg.log_sources]
|
|
165
|
+
if source and not targets[0]:
|
|
166
|
+
return {"error": f"Unknown log source '{source}'.", "configured_sources": cfg.log_sources}
|
|
167
|
+
|
|
168
|
+
matches, errors = [], {}
|
|
169
|
+
for label, kind, target in targets:
|
|
170
|
+
try:
|
|
171
|
+
entries = _read(kind, target, 20_000, since_minutes)
|
|
172
|
+
except Exception as e:
|
|
173
|
+
errors[label] = f"{type(e).__name__}: {e}"
|
|
174
|
+
continue
|
|
175
|
+
for e in entries:
|
|
176
|
+
if (e["ts"] is None or e["ts"] >= since) and regex.search(e["line"]):
|
|
177
|
+
matches.append({"source": label, **e})
|
|
178
|
+
matches.sort(key=lambda m: m["ts"] or since)
|
|
179
|
+
return {"pattern": pattern, "since": since, "match_count": len(matches),
|
|
180
|
+
"matches": matches[-200:], "errors": errors or None}
|
|
181
|
+
|
|
182
|
+
|
|
183
|
+
def classify(line: str) -> tuple[str, str] | None:
|
|
184
|
+
for kind, rx, severity in _RULES:
|
|
185
|
+
if rx.search(line):
|
|
186
|
+
return kind, severity
|
|
187
|
+
return None
|
|
188
|
+
|
|
189
|
+
|
|
190
|
+
def recent_problems(since_minutes: int = 30) -> dict:
|
|
191
|
+
"""Classify recent log lines into problem kinds, aggregated per source and kind."""
|
|
192
|
+
cfg = get_config()
|
|
193
|
+
if not cfg.log_sources:
|
|
194
|
+
return skipped("LOG_SOURCES not set")
|
|
195
|
+
now = utcnow()
|
|
196
|
+
since = now - dt.timedelta(minutes=since_minutes)
|
|
197
|
+
spike_since = now - dt.timedelta(minutes=5)
|
|
198
|
+
groups: dict[tuple[str, str], dict] = {}
|
|
199
|
+
recent_errors: dict[str, list] = defaultdict(list)
|
|
200
|
+
errors, last_activity = {}, {}
|
|
201
|
+
for spec in cfg.log_sources:
|
|
202
|
+
kind, target = _parse_source(spec)
|
|
203
|
+
try:
|
|
204
|
+
entries = _read(kind, target, 20_000, since_minutes)
|
|
205
|
+
except Exception as e:
|
|
206
|
+
errors[spec] = f"{type(e).__name__}: {e}"
|
|
207
|
+
continue
|
|
208
|
+
last_activity[_label(spec)] = max((e["ts"] for e in entries if e["ts"]), default=None)
|
|
209
|
+
for e in entries:
|
|
210
|
+
if e["ts"] and e["ts"] < since:
|
|
211
|
+
continue
|
|
212
|
+
hit = classify(e["line"])
|
|
213
|
+
if not hit:
|
|
214
|
+
continue
|
|
215
|
+
k, sev = hit
|
|
216
|
+
g = groups.setdefault((spec, k), {"source": spec, "kind": k, "severity": sev, "count": 0,
|
|
217
|
+
"first": e["ts"], "last": e["ts"], "sample": e["line"][:300]})
|
|
218
|
+
g["count"] += 1
|
|
219
|
+
g["last"] = e["ts"] or g["last"]
|
|
220
|
+
if e["ts"] and e["ts"] >= spike_since:
|
|
221
|
+
recent_errors[spec].append(e["ts"])
|
|
222
|
+
|
|
223
|
+
events = []
|
|
224
|
+
for g in groups.values():
|
|
225
|
+
events.append(event(g["first"], SOURCE, g["kind"],
|
|
226
|
+
f"[{_label(g['source'])}] {g['count']}x {g['kind']} (last {g['last'] and g['last'].isoformat(timespec='seconds')}): {g['sample']}",
|
|
227
|
+
g["severity"], log_source=g["source"], count=g["count"], last=g["last"]))
|
|
228
|
+
for spec, stamps in recent_errors.items():
|
|
229
|
+
if len(stamps) >= cfg.log_error_spike:
|
|
230
|
+
events.append(event(min(stamps), SOURCE, "error_spike",
|
|
231
|
+
f"[{_label(spec)}] {len(stamps)} problem lines in the last 5 minutes "
|
|
232
|
+
f"(threshold {cfg.log_error_spike})", "warning", log_source=spec))
|
|
233
|
+
summary = sorted(groups.values(), key=lambda g: (g["first"] or now))
|
|
234
|
+
return {"since_minutes": since_minutes, "problems": summary, "last_activity": last_activity,
|
|
235
|
+
"errors": errors or None, "events": events}
|