stackdoctor 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- stackdoctor/__init__.py +0 -0
- stackdoctor/__main__.py +3 -0
- stackdoctor/checks/__init__.py +29 -0
- stackdoctor/checks/celery.py +319 -0
- stackdoctor/checks/logs.py +235 -0
- stackdoctor/checks/postgres.py +202 -0
- stackdoctor/checks/redis.py +154 -0
- stackdoctor/config.py +80 -0
- stackdoctor/diagnose.py +248 -0
- stackdoctor/safety.py +331 -0
- stackdoctor/server.py +164 -0
- stackdoctor-0.1.0.dist-info/METADATA +341 -0
- stackdoctor-0.1.0.dist-info/RECORD +16 -0
- stackdoctor-0.1.0.dist-info/WHEEL +4 -0
- stackdoctor-0.1.0.dist-info/entry_points.txt +2 -0
- stackdoctor-0.1.0.dist-info/licenses/LICENSE +21 -0
stackdoctor/server.py
ADDED
|
@@ -0,0 +1,164 @@
|
|
|
1
|
+
"""stackdoctor MCP server (stdio). Strictly read-only."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import functools
|
|
6
|
+
import inspect
|
|
7
|
+
import logging
|
|
8
|
+
import sys
|
|
9
|
+
|
|
10
|
+
from mcp.server.mcpserver import MCPServer
|
|
11
|
+
from mcp.types import ToolAnnotations
|
|
12
|
+
|
|
13
|
+
from . import diagnose as diag
|
|
14
|
+
from .checks import celery, logs, postgres, redis
|
|
15
|
+
from .config import get_config
|
|
16
|
+
from .safety import safe_output
|
|
17
|
+
|
|
18
|
+
log = logging.getLogger("stackdoctor")
|
|
19
|
+
|
|
20
|
+
mcp = MCPServer(
|
|
21
|
+
"stackdoctor",
|
|
22
|
+
instructions=(
|
|
23
|
+
"Read-only diagnostics for a Python backend stack (Postgres, Celery, Redis, logs). "
|
|
24
|
+
"Start with diagnose(symptom): it runs all checks at once and returns findings, a merged "
|
|
25
|
+
"cross-system timeline and possible cause → effect chains. Chains are hypotheses based on "
|
|
26
|
+
"timing; present them as possible causes, not facts. Use the other tools to drill down."
|
|
27
|
+
),
|
|
28
|
+
)
|
|
29
|
+
|
|
30
|
+
READ_ONLY = ToolAnnotations(read_only_hint=True, destructive_hint=False,
|
|
31
|
+
idempotent_hint=True, open_world_hint=False)
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def tool(fn):
|
|
35
|
+
"""Register a tool whose output is always redacted and size-capped."""
|
|
36
|
+
if inspect.iscoroutinefunction(fn):
|
|
37
|
+
@functools.wraps(fn)
|
|
38
|
+
async def wrapper(*args, **kwargs):
|
|
39
|
+
try:
|
|
40
|
+
result = await fn(*args, **kwargs)
|
|
41
|
+
except Exception as e:
|
|
42
|
+
log.exception("tool %s failed", fn.__name__)
|
|
43
|
+
result = {"error": f"{type(e).__name__}: {str(e)[:500]}"}
|
|
44
|
+
return safe_output(result, get_config().max_output_chars * 2)
|
|
45
|
+
else:
|
|
46
|
+
@functools.wraps(fn)
|
|
47
|
+
def wrapper(*args, **kwargs):
|
|
48
|
+
try:
|
|
49
|
+
result = fn(*args, **kwargs)
|
|
50
|
+
except Exception as e:
|
|
51
|
+
log.exception("tool %s failed", fn.__name__)
|
|
52
|
+
result = {"error": f"{type(e).__name__}: {str(e)[:500]}"}
|
|
53
|
+
return safe_output(result, get_config().max_output_chars)
|
|
54
|
+
mcp.tool(annotations=READ_ONLY)(wrapper)
|
|
55
|
+
return wrapper
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
# --- Main -------------------------------------------------------------------
|
|
59
|
+
|
|
60
|
+
@tool
|
|
61
|
+
async def diagnose(symptom: str = "") -> dict:
|
|
62
|
+
"""Run all relevant checks concurrently and return one timestamped snapshot.
|
|
63
|
+
|
|
64
|
+
Use this first for questions like "why are my jobs stuck?" or "why is the API slow?".
|
|
65
|
+
Returns findings, a merged timeline across Postgres/Celery/Redis/logs, and
|
|
66
|
+
'possible cause' chains (timing-based hypotheses with evidence timestamps).
|
|
67
|
+
"""
|
|
68
|
+
return await diag.diagnose(symptom)
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
# --- Postgres (light; for deep Postgres health use Postgres MCP Pro) --------
|
|
72
|
+
|
|
73
|
+
@tool
|
|
74
|
+
def active_queries(min_duration_s: float = 0) -> dict:
|
|
75
|
+
"""Non-idle Postgres sessions (incl. 'idle in transaction') with duration and wait events."""
|
|
76
|
+
return postgres.active_queries(min_duration_s)
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
@tool
|
|
80
|
+
def blocking_locks() -> dict:
|
|
81
|
+
"""Postgres sessions waiting on locks, and the sessions blocking them."""
|
|
82
|
+
return postgres.blocking_locks()
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
@tool
|
|
86
|
+
def slow_queries(limit: int = 10) -> dict:
|
|
87
|
+
"""Slowest statements by mean time from pg_stat_statements, plus seq-scan-heavy tables."""
|
|
88
|
+
return postgres.slow_queries(limit)
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
@tool
|
|
92
|
+
def run_select(sql: str, limit: int = 100) -> dict:
|
|
93
|
+
"""Run one read-only SELECT / WITH / EXPLAIN (no ANALYZE). A row LIMIT is always applied (max 1000)."""
|
|
94
|
+
return postgres.run_select(sql, limit)
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
# --- Celery (inspect API + broker, no Flower) -------------------------------
|
|
98
|
+
|
|
99
|
+
@tool
|
|
100
|
+
def workers() -> dict:
|
|
101
|
+
"""Live Celery workers with their queues, concurrency, and active/reserved tasks."""
|
|
102
|
+
return celery.workers()
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
@tool
|
|
106
|
+
def queue_lengths(queues: list[str] | None = None) -> dict:
|
|
107
|
+
"""Messages waiting per Celery queue, read from the broker (Redis incl. priority queues, or RabbitMQ)."""
|
|
108
|
+
return celery.queue_lengths(queues)
|
|
109
|
+
|
|
110
|
+
|
|
111
|
+
@tool
|
|
112
|
+
def failed_tasks(limit: int = 20) -> dict:
|
|
113
|
+
"""Recent failed tasks from a Redis result backend (explains why if they aren't visible)."""
|
|
114
|
+
return celery.failed_tasks(limit)
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
@tool
|
|
118
|
+
def task_details(task_id: str) -> dict:
|
|
119
|
+
"""Stored result/traceback for one task, and whether a worker currently holds it."""
|
|
120
|
+
return celery.task_details(task_id)
|
|
121
|
+
|
|
122
|
+
|
|
123
|
+
# --- Redis ------------------------------------------------------------------
|
|
124
|
+
|
|
125
|
+
@tool
|
|
126
|
+
def memory_and_clients() -> dict:
|
|
127
|
+
"""Redis memory vs maxmemory, evictions, clients, keyspace and slowlog."""
|
|
128
|
+
return redis.memory_and_clients()
|
|
129
|
+
|
|
130
|
+
|
|
131
|
+
@tool
|
|
132
|
+
def scan_keys(pattern: str = "*", limit: int = 50) -> dict:
|
|
133
|
+
"""List Redis keys matching a glob pattern using SCAN (never KEYS). Max 500."""
|
|
134
|
+
return redis.scan_keys(pattern, limit)
|
|
135
|
+
|
|
136
|
+
|
|
137
|
+
@tool
|
|
138
|
+
def key_info(key: str) -> dict:
|
|
139
|
+
"""Type, TTL, size, encoding and a short redacted preview of one Redis key."""
|
|
140
|
+
return redis.key_info(key)
|
|
141
|
+
|
|
142
|
+
|
|
143
|
+
# --- Logs -------------------------------------------------------------------
|
|
144
|
+
|
|
145
|
+
@tool
|
|
146
|
+
def tail_logs(source: str, lines: int = 100) -> dict:
|
|
147
|
+
"""Last N lines (max 500) of a configured log source (file path or docker container)."""
|
|
148
|
+
return logs.tail_logs(source, lines)
|
|
149
|
+
|
|
150
|
+
|
|
151
|
+
@tool
|
|
152
|
+
def search_logs(pattern: str, since_minutes: int = 30, source: str | None = None) -> dict:
|
|
153
|
+
"""Search configured log sources for a regex (case-insensitive) within the last N minutes."""
|
|
154
|
+
return logs.search_logs(pattern, since_minutes, source)
|
|
155
|
+
|
|
156
|
+
|
|
157
|
+
def main() -> None:
|
|
158
|
+
logging.basicConfig(level=logging.WARNING, stream=sys.stderr) # stdout is the MCP channel
|
|
159
|
+
get_config()
|
|
160
|
+
mcp.run("stdio")
|
|
161
|
+
|
|
162
|
+
|
|
163
|
+
if __name__ == "__main__":
|
|
164
|
+
main()
|
|
@@ -0,0 +1,341 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: stackdoctor
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Read-only MCP server that diagnoses a Python backend stack (Postgres, Celery, Redis, logs) with cross-system correlation.
|
|
5
|
+
Project-URL: Homepage, https://github.com/lepri89/stackdoctor
|
|
6
|
+
Project-URL: Issues, https://github.com/lepri89/stackdoctor/issues
|
|
7
|
+
License: MIT
|
|
8
|
+
License-File: LICENSE
|
|
9
|
+
Keywords: celery,diagnostics,mcp,observability,postgres,redis
|
|
10
|
+
Requires-Python: >=3.11
|
|
11
|
+
Requires-Dist: celery>=5.3
|
|
12
|
+
Requires-Dist: mcp<3,>=2.3
|
|
13
|
+
Requires-Dist: psycopg[binary]>=3.1
|
|
14
|
+
Requires-Dist: python-dotenv>=1.0
|
|
15
|
+
Requires-Dist: redis>=5.0
|
|
16
|
+
Requires-Dist: sqlglot>=25
|
|
17
|
+
Description-Content-Type: text/markdown
|
|
18
|
+
|
|
19
|
+
# stackdoctor
|
|
20
|
+
|
|
21
|
+
[](https://github.com/lepri89/stackdoctor/actions/workflows/tests.yml)
|
|
22
|
+
[](https://github.com/lepri89/stackdoctor/actions/workflows/demo.yml)
|
|
23
|
+

|
|
24
|
+

|
|
25
|
+
|
|
26
|
+

|
|
27
|
+
|
|
28
|
+
*Claude finds that the worker shut down and 40 jobs are waiting, in one diagnose() call.*
|
|
29
|
+
|
|
30
|
+
> **Strictly read-only.** stackdoctor never writes, deletes, restarts, retries, revokes or sends anything.
|
|
31
|
+
> Postgres sessions are forced read-only by the server, Redis commands go through an allowlist,
|
|
32
|
+
> and Celery is only *inspected*. Secrets are redacted from every response.
|
|
33
|
+
|
|
34
|
+
stackdoctor is an MCP server that lets Claude, Cursor or any MCP client diagnose a Python backend stack
|
|
35
|
+
(**Postgres, Celery, Redis and logs**) from a single tool call.
|
|
36
|
+
|
|
37
|
+
Ask *"why are my jobs stuck?"* and the assistant calls `diagnose(symptom)`. It runs every relevant check
|
|
38
|
+
in parallel and returns one timestamped snapshot with:
|
|
39
|
+
|
|
40
|
+
- **findings**: blocked queries, missing workers, queue backlogs, log error spikes, Redis memory pressure, …
|
|
41
|
+
- **a merged timeline** of events from Postgres, Celery, Redis and your logs
|
|
42
|
+
- **possible cause → effect chains** when events line up in time, each with its evidence timestamps
|
|
43
|
+
|
|
44
|
+
```text
|
|
45
|
+
possible cause (confidence: medium): Database lock → blocked queries → stuck or failing tasks
|
|
46
|
+
cause: 21:58:01 postgres.lock_held pid 7959 holds a lock blocking 2 sessions: LOCK TABLE orders …
|
|
47
|
+
effect: 21:58:02 logs.db_lock_wait [postgres] process 7962 still waiting for RowExclusiveLock …
|
|
48
|
+
effect: 21:58:21 celery.task_failed tasks.process_order[…] failed: LockNotAvailable: lock timeout
|
|
49
|
+
effect: 21:58:21 celery.task_long_running tasks.process_order[…] running 13.2s on worker1
|
|
50
|
+
effect: 21:58:34 celery.workers_saturated worker1: all 2 slots busy, 2 reserved
|
|
51
|
+
caveat: Inferred from timing only. Verify before acting; this is not a confirmed root cause.
|
|
52
|
+
```
|
|
53
|
+
|
|
54
|
+
## Install
|
|
55
|
+
|
|
56
|
+
You need [uv](https://docs.astral.sh/uv/getting-started/installation/):
|
|
57
|
+
|
|
58
|
+
```sh
|
|
59
|
+
curl -LsSf https://astral.sh/uv/install.sh | sh # macOS / Linux
|
|
60
|
+
powershell -c "irm https://astral.sh/uv/install.ps1 | iex" # Windows
|
|
61
|
+
```
|
|
62
|
+
|
|
63
|
+
Then one command runs the server, with no other install step:
|
|
64
|
+
|
|
65
|
+
```sh
|
|
66
|
+
uvx stackdoctor
|
|
67
|
+
```
|
|
68
|
+
|
|
69
|
+
Until the package is on PyPI, run it from a checkout with `uvx --from /path/to/stackdoctor stackdoctor`,
|
|
70
|
+
or from git with `uvx --from git+https://github.com/lepri89/stackdoctor stackdoctor`.
|
|
71
|
+
|
|
72
|
+
> **macOS 12 (Monterey):** `uvx stackdoctor` fails with `realpath: command not found`, because uv's
|
|
73
|
+
> launcher script needs `realpath`, which only ships with macOS 13+. Use
|
|
74
|
+
> `uvx --from stackdoctor python -m stackdoctor` instead. In client configs that means
|
|
75
|
+
> `"args": ["--from", "stackdoctor", "python", "-m", "stackdoctor"]`. Alternatively, run
|
|
76
|
+
> `uv tool install stackdoctor` once and use `stackdoctor` as the command.
|
|
77
|
+
|
|
78
|
+
## Configure
|
|
79
|
+
|
|
80
|
+
Set environment variables in your MCP client config, or in a `.env` file. stackdoctor looks for `.env` in
|
|
81
|
+
the working directory, or uses the file named by `STACKDOCTOR_ENV_FILE`.
|
|
82
|
+
**Any source you don't configure is skipped**, so it's fine to start with only `DATABASE_URL`.
|
|
83
|
+
|
|
84
|
+
| Variable | Example | Used for |
|
|
85
|
+
|---|---|---|
|
|
86
|
+
| `DATABASE_URL` | `postgresql://stackdoctor_ro:…@localhost:5432/app` | Postgres checks |
|
|
87
|
+
| `REDIS_URL` | `redis://localhost:6379/0` | Redis checks |
|
|
88
|
+
| `CELERY_BROKER_URL` | `redis://localhost:6379/0` (RabbitMQ `amqp://…` is experimental) | Celery inspect + queue lengths |
|
|
89
|
+
| `CELERY_RESULT_BACKEND` | `redis://localhost:6379/1` | failed tasks / task details |
|
|
90
|
+
| `CELERY_APP` | `myproject.celery:app` | optional: use your app's queues/routes/config (run from your project dir) |
|
|
91
|
+
| `CELERY_QUEUES` | `default,emails` | extra queue names to measure |
|
|
92
|
+
| `LOG_SOURCES` | `./logs/worker.log,docker:api,docker:worker` | log files and/or docker containers |
|
|
93
|
+
|
|
94
|
+
`LOG_SOURCES` entries are file paths, `docker:<container>`, or a bare container name.
|
|
95
|
+
Only configured sources can be read.
|
|
96
|
+
|
|
97
|
+
> **Celery workers: point `LOG_SOURCES` at the worker's stdout.** Celery prints
|
|
98
|
+
> `worker: Warm shutdown (MainProcess)` straight to stdout, not through logging, so it never reaches a
|
|
99
|
+
> `--logfile`. Docker containers already capture stdout. For a file, redirect stdout to it
|
|
100
|
+
> (`celery … worker >> worker.log 2>&1`, or supervisor/systemd stdout capture) so stackdoctor can see
|
|
101
|
+
> when a worker stopped.
|
|
102
|
+
|
|
103
|
+
<details>
|
|
104
|
+
<summary>Thresholds and limits</summary>
|
|
105
|
+
|
|
106
|
+
| Variable | Default | Meaning |
|
|
107
|
+
|---|---|---|
|
|
108
|
+
| `QUEUE_THRESHOLD` | 100 | messages waiting before a queue counts as a backlog |
|
|
109
|
+
| `EXPECTED_WORKERS` | 0 (off) | report missing workers if fewer reply |
|
|
110
|
+
| `LONG_TASK_S` / `LONG_QUERY_S` | 60 / 30 | when a task / query counts as long-running |
|
|
111
|
+
| `REDIS_MEM_WARN_PCT` | 85 | % of `maxmemory` that counts as "near the limit" |
|
|
112
|
+
| `LOG_ERROR_SPIKE` | 10 | problem lines in 5 minutes that count as a spike |
|
|
113
|
+
| `CHAIN_WINDOW_MIN` | 30 | how far apart cause and effect may be |
|
|
114
|
+
| `CHECK_TIMEOUT_S` | 8 | per-check timeout inside `diagnose` |
|
|
115
|
+
| `CELERY_INSPECT_TIMEOUT_S` | 1.0 | how long to wait for worker replies |
|
|
116
|
+
| `MAX_OUTPUT_CHARS` | 20000 | per-tool response cap (`diagnose` gets 2×) |
|
|
117
|
+
|
|
118
|
+
</details>
|
|
119
|
+
|
|
120
|
+
### Claude Desktop
|
|
121
|
+
|
|
122
|
+
`~/Library/Application Support/Claude/claude_desktop_config.json` (macOS) or
|
|
123
|
+
`%APPDATA%\Claude\claude_desktop_config.json` (Windows):
|
|
124
|
+
|
|
125
|
+
```json
|
|
126
|
+
{
|
|
127
|
+
"mcpServers": {
|
|
128
|
+
"stackdoctor": {
|
|
129
|
+
"command": "uvx",
|
|
130
|
+
"args": ["stackdoctor"],
|
|
131
|
+
"env": {
|
|
132
|
+
"DATABASE_URL": "postgresql://stackdoctor_ro:password@localhost:5432/app",
|
|
133
|
+
"REDIS_URL": "redis://localhost:6379/0",
|
|
134
|
+
"CELERY_BROKER_URL": "redis://localhost:6379/0",
|
|
135
|
+
"CELERY_RESULT_BACKEND": "redis://localhost:6379/1",
|
|
136
|
+
"LOG_SOURCES": "docker:api,docker:worker"
|
|
137
|
+
}
|
|
138
|
+
}
|
|
139
|
+
}
|
|
140
|
+
}
|
|
141
|
+
```
|
|
142
|
+
|
|
143
|
+
GUI apps often don't see your shell `PATH`. If Claude Desktop can't find `uvx`, use the full path:
|
|
144
|
+
`/Users/<username>/.local/bin/uvx` on macOS, or `C:\\Users\\<username>\\.local\\bin\\uvx.exe` on Windows.
|
|
145
|
+
|
|
146
|
+
### Claude Code
|
|
147
|
+
|
|
148
|
+
```sh
|
|
149
|
+
claude mcp add stackdoctor \
|
|
150
|
+
-e DATABASE_URL=postgresql://stackdoctor_ro:password@localhost:5432/app \
|
|
151
|
+
-e REDIS_URL=redis://localhost:6379/0 \
|
|
152
|
+
-e CELERY_BROKER_URL=redis://localhost:6379/0 \
|
|
153
|
+
-e LOG_SOURCES=docker:api,docker:worker \
|
|
154
|
+
-- uvx stackdoctor
|
|
155
|
+
```
|
|
156
|
+
|
|
157
|
+
Or, from a project directory with a `.env` file, simply `claude mcp add stackdoctor -- uvx stackdoctor`.
|
|
158
|
+
|
|
159
|
+
### Cursor
|
|
160
|
+
|
|
161
|
+
`.cursor/mcp.json` in your project, or `~/.cursor/mcp.json` globally. This uses the same
|
|
162
|
+
`mcpServers` format as Claude Desktop:
|
|
163
|
+
|
|
164
|
+
```json
|
|
165
|
+
{
|
|
166
|
+
"mcpServers": {
|
|
167
|
+
"stackdoctor": {
|
|
168
|
+
"command": "uvx",
|
|
169
|
+
"args": ["stackdoctor"],
|
|
170
|
+
"env": { "STACKDOCTOR_ENV_FILE": "${workspaceFolder}/.env" }
|
|
171
|
+
}
|
|
172
|
+
}
|
|
173
|
+
}
|
|
174
|
+
```
|
|
175
|
+
|
|
176
|
+
### VS Code (Copilot agent mode)
|
|
177
|
+
|
|
178
|
+
`.vscode/mcp.json`:
|
|
179
|
+
|
|
180
|
+
```json
|
|
181
|
+
{
|
|
182
|
+
"servers": {
|
|
183
|
+
"stackdoctor": {
|
|
184
|
+
"type": "stdio",
|
|
185
|
+
"command": "uvx",
|
|
186
|
+
"args": ["stackdoctor"],
|
|
187
|
+
"env": { "STACKDOCTOR_ENV_FILE": "${workspaceFolder}/.env" }
|
|
188
|
+
}
|
|
189
|
+
}
|
|
190
|
+
}
|
|
191
|
+
```
|
|
192
|
+
|
|
193
|
+
## Tools
|
|
194
|
+
|
|
195
|
+
| Tool | What it returns |
|
|
196
|
+
|---|---|
|
|
197
|
+
| `diagnose(symptom)` | **Start here.** Runs the relevant checks concurrently, each with its own timeout, and returns findings, a merged timeline and possible causes |
|
|
198
|
+
| `active_queries(min_duration_s)` | non-idle sessions, incl. *idle in transaction* |
|
|
199
|
+
| `blocking_locks()` | waiting sessions and the sessions blocking them |
|
|
200
|
+
| `slow_queries(limit)` | top statements from `pg_stat_statements` (explains how to enable it if missing) and seq-scan-heavy tables |
|
|
201
|
+
| `run_select(sql, limit)` | one validated `SELECT` / `WITH` / `EXPLAIN` with a row limit |
|
|
202
|
+
| `workers()` | live workers, their queues, concurrency, active/reserved tasks |
|
|
203
|
+
| `queue_lengths(queues)` | messages waiting per queue, read from the broker (Redis incl. priority queues; RabbitMQ experimental) |
|
|
204
|
+
| `failed_tasks(limit)` | recent failures from a Redis result backend, or *why* they aren't visible |
|
|
205
|
+
| `task_details(task_id)` | stored result/traceback, and whether a worker holds the task right now |
|
|
206
|
+
| `memory_and_clients()` | Redis memory vs `maxmemory`, evictions, clients, slowlog |
|
|
207
|
+
| `scan_keys(pattern, limit)` | key names via `SCAN` |
|
|
208
|
+
| `key_info(key)` | type, TTL, size, encoding, short redacted preview |
|
|
209
|
+
| `tail_logs(source, lines)` | last lines of a configured log source |
|
|
210
|
+
| `search_logs(pattern, since_minutes, source)` | regex search over recent log lines |
|
|
211
|
+
|
|
212
|
+
Postgres checks are intentionally light. For deep Postgres health (index advice, vacuum, bloat), run
|
|
213
|
+
[Postgres MCP Pro](https://github.com/crystaldba/postgres-mcp) next to stackdoctor.
|
|
214
|
+
|
|
215
|
+
## Safety
|
|
216
|
+
|
|
217
|
+
stackdoctor is built so that a confused or prompt-injected assistant still can't change your systems.
|
|
218
|
+
|
|
219
|
+
**Postgres**
|
|
220
|
+
- Every session is opened with `default_transaction_read_only=on`, `statement_timeout=5s` and
|
|
221
|
+
`lock_timeout=2s`, passed as connection options so they override your URL. stackdoctor also checks
|
|
222
|
+
the setting and refuses to run if the session isn't read-only.
|
|
223
|
+
- `run_select` parses SQL with [sqlglot](https://github.com/tobymao/sqlglot), not regexes. It accepts
|
|
224
|
+
exactly one `SELECT` / `WITH … SELECT` / `EXPLAIN` statement and rejects:
|
|
225
|
+
- writable CTEs, `SELECT … INTO` and `FOR UPDATE/SHARE`
|
|
226
|
+
- `EXPLAIN ANALYZE`, because it executes the query
|
|
227
|
+
- functions with side effects, even inside a read-only transaction: `pg_terminate_backend`,
|
|
228
|
+
`pg_cancel_backend`, `pg_reload_conf`, `pg_read_file`, `pg_read_binary_file`, `pg_ls_dir`, `lo_*`,
|
|
229
|
+
`dblink*`, `set_config`, `pg_advisory_*`, `query_to_xml` (which runs SQL from a string), `pg_sleep`, …
|
|
230
|
+
- Results are always limited: queries are wrapped as `SELECT * FROM (<q>) AS sd_sub LIMIT n`
|
|
231
|
+
(default 100, max 1000).
|
|
232
|
+
|
|
233
|
+
**Recommended: a dedicated read-only role.** Defense in depth, so the database enforces the rules too:
|
|
234
|
+
|
|
235
|
+
```sql
|
|
236
|
+
CREATE ROLE stackdoctor_ro LOGIN PASSWORD 'change-me';
|
|
237
|
+
GRANT CONNECT ON DATABASE app TO stackdoctor_ro;
|
|
238
|
+
GRANT USAGE ON SCHEMA public TO stackdoctor_ro;
|
|
239
|
+
GRANT SELECT ON ALL TABLES IN SCHEMA public TO stackdoctor_ro;
|
|
240
|
+
ALTER DEFAULT PRIVILEGES IN SCHEMA public GRANT SELECT ON TABLES TO stackdoctor_ro;
|
|
241
|
+
GRANT pg_monitor TO stackdoctor_ro; -- see other sessions' queries and pg_stat_statements
|
|
242
|
+
ALTER ROLE stackdoctor_ro SET default_transaction_read_only = on;
|
|
243
|
+
```
|
|
244
|
+
|
|
245
|
+
Anything the role can `SELECT`, the assistant can read. If some tables hold data you don't want to
|
|
246
|
+
share with an AI, leave them out of the `GRANT SELECT`.
|
|
247
|
+
|
|
248
|
+
**Redis.** Every command is checked against an allowlist that is aware of subcommands (`INFO`, `SCAN`,
|
|
249
|
+
`TYPE`, `PTTL`, `LLEN`, `GET`, …). `CLIENT LIST` is allowed; `CLIENT KILL` / `PAUSE` are not. The same goes
|
|
250
|
+
for `OBJECT`, `MEMORY` and `CONFIG` (only `CONFIG GET`). `KEYS` is never used; listing always goes through `SCAN`.
|
|
251
|
+
|
|
252
|
+
**Celery.** Only `inspect` (`ping`, `active`, `reserved`, `active_queues`, `stats`, `query_task`) and
|
|
253
|
+
broker reads are used: `LLEN` on Redis, and passive `queue_declare` on RabbitMQ.
|
|
254
|
+
|
|
255
|
+
> **RabbitMQ support is experimental.** Queue lengths via passive `queue_declare` and the `inspect`
|
|
256
|
+
> calls are implemented but not yet covered by CI. The Redis broker is the tested path. Reports welcome. stackdoctor never revokes,
|
|
257
|
+
retries, shuts down or sends tasks.
|
|
258
|
+
*To be precise:* `inspect` works by publishing a broadcast message on Celery's **control channel**
|
|
259
|
+
(`celery.pidbox`) and reading replies from a temporary reply queue. That is how Celery's own
|
|
260
|
+
`celery inspect` command works. It doesn't touch your task queues or results, but it isn't zero
|
|
261
|
+
traffic. Everything else stackdoctor does is pure reads.
|
|
262
|
+
|
|
263
|
+
**Redaction.** Every response is redacted before it leaves the server. This covers:
|
|
264
|
+
- passwords in URLs (`postgres://user:***@host`)
|
|
265
|
+
- `Authorization` headers and bearer tokens
|
|
266
|
+
- `PASSWORD=`, `SECRET_KEY=`, `api_key:` and other secret-looking keys
|
|
267
|
+
- AWS, GitHub, Slack, Stripe, OpenAI/Anthropic-style and Google keys, JWTs, and private keys
|
|
268
|
+
|
|
269
|
+
Task args/kwargs and Redis value previews are truncated and redacted too. Redaction is pattern-based,
|
|
270
|
+
so treat it as a safety net, not a guarantee.
|
|
271
|
+
|
|
272
|
+
**Output caps.** Every tool caps its response (`MAX_OUTPUT_CHARS`) by trimming lists and long strings,
|
|
273
|
+
so a huge table or log can't flood the assistant's context.
|
|
274
|
+
|
|
275
|
+
**Logs.** Only sources listed in `LOG_SOURCES` can be read, so the assistant can't ask for `/etc/passwd`.
|
|
276
|
+
Docker logs are read with `docker logs`, called without a shell.
|
|
277
|
+
|
|
278
|
+
## Try the demo
|
|
279
|
+
|
|
280
|
+
The demo is a small FastAPI + Celery + Redis + Postgres app with scripts that break it in realistic ways.
|
|
281
|
+
CI runs every scenario on each push ([`demo.yml`](.github/workflows/demo.yml)): it breaks the stack,
|
|
282
|
+
calls `diagnose()` through a real MCP stdio client and checks the findings and chains.
|
|
283
|
+
To try it without Docker (local Postgres, Redis and worker), see [DEMO.md](DEMO.md).
|
|
284
|
+
|
|
285
|
+
```sh
|
|
286
|
+
cd demo
|
|
287
|
+
docker compose up -d --build # first run builds the app and loads 1M rows (~1 min)
|
|
288
|
+
cp .env.example .env # stackdoctor config for the demo (uses the read-only role)
|
|
289
|
+
```
|
|
290
|
+
|
|
291
|
+
Point your MCP client at the demo by setting `STACKDOCTOR_ENV_FILE` to the absolute path of `demo/.env`.
|
|
292
|
+
Then break something and ask:
|
|
293
|
+
|
|
294
|
+
| Script (macOS/Linux · Windows) | What it does | Ask |
|
|
295
|
+
|---|---|---|
|
|
296
|
+
| `./break_worker.sh` · `.\break_worker.ps1` | stops the Celery worker and enqueues 50 tasks | "why are my jobs stuck?" |
|
|
297
|
+
| `./lock_table.sh` · `.\lock_table.ps1` | holds an `ACCESS EXCLUSIVE` lock on `orders` for 3 min, then enqueues tasks that block on it | "why are my jobs stuck?" |
|
|
298
|
+
| `./slow_query.sh` · `.\slow_query.ps1` | runs three ~1-minute sequential scans | "why is the API slow?" |
|
|
299
|
+
| `./reset.sh` · `.\reset.ps1` | restarts the worker and ends the demo's locks and slow queries | |
|
|
300
|
+
|
|
301
|
+
**Colima (macOS).** `docker compose` works unchanged once Colima is running:
|
|
302
|
+
`colima start --cpu 2 --memory 4`. On macOS 13+ you can add `--vm-type vz`. On macOS 12 Colima uses
|
|
303
|
+
QEMU, which must be installed first (`brew install qemu`). **Windows:** use Docker Desktop or Docker
|
|
304
|
+
in WSL2 and run the `.ps1` scripts from PowerShell.
|
|
305
|
+
|
|
306
|
+
## Why stackdoctor instead of separate Postgres, Celery and log MCPs?
|
|
307
|
+
|
|
308
|
+
| | Separate MCP servers | stackdoctor |
|
|
309
|
+
|---|---|---|
|
|
310
|
+
| Install | one server per system, each with its own config | one `uvx stackdoctor` |
|
|
311
|
+
| Celery | usually needs Flower running | inspect API + broker, no Flower |
|
|
312
|
+
| Answering "why is it stuck?" | the AI calls 5–10 tools one after another, each a different moment in time | one `diagnose()` call: all checks run in parallel and share one timestamp |
|
|
313
|
+
| Correlation | left to the AI, across separate outputs | merged timeline + timing-based cause → effect hypotheses, with evidence |
|
|
314
|
+
| Safety | varies per server | one read-only policy for every system, redaction and output caps everywhere |
|
|
315
|
+
| Postgres depth | Postgres MCP Pro is deeper | intentionally light; use both |
|
|
316
|
+
|
|
317
|
+
## Development
|
|
318
|
+
|
|
319
|
+
```sh
|
|
320
|
+
uv sync
|
|
321
|
+
uv run pytest # unit tests (safety layer, diagnose, logs)
|
|
322
|
+
STACKDOCTOR_TEST_DATABASE_URL=postgresql://shop:shop@localhost:55432/shop \
|
|
323
|
+
STACKDOCTOR_TEST_REDIS_URL=redis://localhost:56379/0 \
|
|
324
|
+
uv run pytest # plus live tests against the demo
|
|
325
|
+
```
|
|
326
|
+
|
|
327
|
+
## Releasing
|
|
328
|
+
|
|
329
|
+
Releases go to PyPI from GitHub Actions with
|
|
330
|
+
[trusted publishing](https://docs.pypi.org/trusted-publishers/), so no API token is stored anywhere.
|
|
331
|
+
|
|
332
|
+
1. One-time: on PyPI, add a *pending publisher* (Account → Publishing) with project `stackdoctor`, owner
|
|
333
|
+
`lepri89`, repository `stackdoctor`, workflow `release.yml` and environment `pypi`. In GitHub, create
|
|
334
|
+
an environment named `pypi` (Settings → Environments); adding yourself as a required reviewer is a good idea.
|
|
335
|
+
2. Bump `version` in `pyproject.toml`, commit, then tag and push:
|
|
336
|
+
`git tag v0.1.0 && git push origin v0.1.0`.
|
|
337
|
+
The workflow checks that the tag matches the version, runs the tests, builds, and publishes.
|
|
338
|
+
|
|
339
|
+
## License
|
|
340
|
+
|
|
341
|
+
MIT
|
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
stackdoctor/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
2
|
+
stackdoctor/__main__.py,sha256=3dYKHfmWsrdExFlTFlcR5a_icR9fAkn06Yh14TQkEd8,33
|
|
3
|
+
stackdoctor/config.py,sha256=YxUxmeGITSIC5ZdoWDEyfxHR__7EtHnpx6-WdddJpr4,2714
|
|
4
|
+
stackdoctor/diagnose.py,sha256=1MwHu2i97N4agYt0XC_HbDOGyiATpMUxlyPIa4mEAPQ,11910
|
|
5
|
+
stackdoctor/safety.py,sha256=uZbeBysUrZRrZsT3oBZT-47i5QLpfb-33qpgH6-TZXU,13566
|
|
6
|
+
stackdoctor/server.py,sha256=nFFan4ozY0fK1HidHapKQfflZSTTbBOwcg7GrAkEZsc,5463
|
|
7
|
+
stackdoctor/checks/__init__.py,sha256=ls7w0uq46ROy2bF3EvuKbu-4_9fWcx1ZYdqqayiJXyc,977
|
|
8
|
+
stackdoctor/checks/celery.py,sha256=GSU9YHIKHRpGC0E8JjqOtACdS_3mSomUIkXCvSAJaPk,13363
|
|
9
|
+
stackdoctor/checks/logs.py,sha256=Ow0HTKnHl0KUFXKWI7JsXbGfhcJNYuSW-pA502Ekk6o,10627
|
|
10
|
+
stackdoctor/checks/postgres.py,sha256=GcOJyqAwL_zCMhbVav3fBsZgYBFTTFAAeF4Yr1qWek4,8676
|
|
11
|
+
stackdoctor/checks/redis.py,sha256=-7y1raaxHCnQDq0UsltkUSUIjypo34JumXMhouQcD4I,6469
|
|
12
|
+
stackdoctor-0.1.0.dist-info/METADATA,sha256=CGLazwVRJGWYxitlI4hjEhEoNDgUJF2GHyUiFKcfjPA,16860
|
|
13
|
+
stackdoctor-0.1.0.dist-info/WHEEL,sha256=W3fkpkm7-wf9vBI5Z-7s0eWkeM-spu78I8Neb98DeEg,87
|
|
14
|
+
stackdoctor-0.1.0.dist-info/entry_points.txt,sha256=xZHXkjytc-9LZS7jgDM87r1MfhlUId13BYJF97nlBAY,56
|
|
15
|
+
stackdoctor-0.1.0.dist-info/licenses/LICENSE,sha256=NfdI-pslhirrlvs1y4jfAdxeH1I7LNDCb9IVHyk1Zww,1081
|
|
16
|
+
stackdoctor-0.1.0.dist-info/RECORD,,
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 stackdoctor contributors
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|