stackdoctor 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- stackdoctor-0.1.0/.gitattributes +2 -0
- stackdoctor-0.1.0/.github/workflows/demo.yml +53 -0
- stackdoctor-0.1.0/.github/workflows/release.yml +43 -0
- stackdoctor-0.1.0/.github/workflows/tests.yml +23 -0
- stackdoctor-0.1.0/.gitignore +8 -0
- stackdoctor-0.1.0/DEMO.md +161 -0
- stackdoctor-0.1.0/LICENSE +21 -0
- stackdoctor-0.1.0/PKG-INFO +341 -0
- stackdoctor-0.1.0/README.md +323 -0
- stackdoctor-0.1.0/demo/.env.example +10 -0
- stackdoctor-0.1.0/demo/app/Dockerfile +5 -0
- stackdoctor-0.1.0/demo/app/api.py +26 -0
- stackdoctor-0.1.0/demo/app/enqueue.py +10 -0
- stackdoctor-0.1.0/demo/app/requirements.txt +4 -0
- stackdoctor-0.1.0/demo/app/tasks.py +29 -0
- stackdoctor-0.1.0/demo/break_worker.ps1 +8 -0
- stackdoctor-0.1.0/demo/break_worker.sh +9 -0
- stackdoctor-0.1.0/demo/docker-compose.yml +51 -0
- stackdoctor-0.1.0/demo/e2e.py +134 -0
- stackdoctor-0.1.0/demo/lock_table.ps1 +9 -0
- stackdoctor-0.1.0/demo/lock_table.sh +12 -0
- stackdoctor-0.1.0/demo/postgres/init.sql +20 -0
- stackdoctor-0.1.0/demo/reset.ps1 +6 -0
- stackdoctor-0.1.0/demo/reset.sh +8 -0
- stackdoctor-0.1.0/demo/slow_query.ps1 +8 -0
- stackdoctor-0.1.0/demo/slow_query.sh +10 -0
- stackdoctor-0.1.0/docs/images/diagnosis.png +0 -0
- stackdoctor-0.1.0/pyproject.toml +37 -0
- stackdoctor-0.1.0/stackdoctor/__init__.py +0 -0
- stackdoctor-0.1.0/stackdoctor/__main__.py +3 -0
- stackdoctor-0.1.0/stackdoctor/checks/__init__.py +29 -0
- stackdoctor-0.1.0/stackdoctor/checks/celery.py +319 -0
- stackdoctor-0.1.0/stackdoctor/checks/logs.py +235 -0
- stackdoctor-0.1.0/stackdoctor/checks/postgres.py +202 -0
- stackdoctor-0.1.0/stackdoctor/checks/redis.py +154 -0
- stackdoctor-0.1.0/stackdoctor/config.py +80 -0
- stackdoctor-0.1.0/stackdoctor/diagnose.py +248 -0
- stackdoctor-0.1.0/stackdoctor/safety.py +331 -0
- stackdoctor-0.1.0/stackdoctor/server.py +164 -0
- stackdoctor-0.1.0/tests/test_diagnose.py +113 -0
- stackdoctor-0.1.0/tests/test_integration.py +51 -0
- stackdoctor-0.1.0/tests/test_logs.py +92 -0
- stackdoctor-0.1.0/tests/test_safety.py +247 -0
- stackdoctor-0.1.0/uv.lock +1159 -0
|
@@ -0,0 +1,53 @@
|
|
|
1
|
+
name: demo
|
|
2
|
+
|
|
3
|
+
on:
|
|
4
|
+
push:
|
|
5
|
+
pull_request:
|
|
6
|
+
workflow_dispatch:
|
|
7
|
+
|
|
8
|
+
jobs:
|
|
9
|
+
e2e:
|
|
10
|
+
name: docker compose demo, end to end
|
|
11
|
+
runs-on: ubuntu-latest
|
|
12
|
+
timeout-minutes: 30
|
|
13
|
+
steps:
|
|
14
|
+
- uses: actions/checkout@v7
|
|
15
|
+
- uses: astral-sh/setup-uv@v10.2.0
|
|
16
|
+
with:
|
|
17
|
+
python-version: "3.12"
|
|
18
|
+
- run: uv sync
|
|
19
|
+
|
|
20
|
+
- name: Start the demo stack
|
|
21
|
+
working-directory: demo
|
|
22
|
+
run: |
|
|
23
|
+
cp .env.example .env
|
|
24
|
+
docker compose up -d --build --wait --wait-timeout 300
|
|
25
|
+
|
|
26
|
+
- name: Live integration tests (read-only enforcement against the demo)
|
|
27
|
+
env:
|
|
28
|
+
STACKDOCTOR_TEST_DATABASE_URL: postgresql://shop:shop@localhost:55432/shop
|
|
29
|
+
STACKDOCTOR_TEST_REDIS_URL: redis://localhost:56379/0
|
|
30
|
+
run: uv run pytest -q tests/test_integration.py
|
|
31
|
+
|
|
32
|
+
- name: Break, diagnose over MCP stdio, assert, reset
|
|
33
|
+
run: uv run python demo/e2e.py
|
|
34
|
+
|
|
35
|
+
- name: Dump docker logs
|
|
36
|
+
if: failure()
|
|
37
|
+
working-directory: demo
|
|
38
|
+
run: |
|
|
39
|
+
docker compose ps -a
|
|
40
|
+
docker compose logs --no-color --timestamps
|
|
41
|
+
|
|
42
|
+
- name: Upload diagnose() output
|
|
43
|
+
if: always()
|
|
44
|
+
uses: actions/upload-artifact@v7
|
|
45
|
+
with:
|
|
46
|
+
name: diagnose-output
|
|
47
|
+
path: demo/e2e-output/
|
|
48
|
+
if-no-files-found: ignore
|
|
49
|
+
|
|
50
|
+
- name: Stop the demo stack
|
|
51
|
+
if: always()
|
|
52
|
+
working-directory: demo
|
|
53
|
+
run: docker compose down -v
|
|
@@ -0,0 +1,43 @@
|
|
|
1
|
+
name: release
|
|
2
|
+
|
|
3
|
+
# Publishes to PyPI with trusted publishing (OIDC), no API token.
|
|
4
|
+
# Triggered by pushing a version tag that matches pyproject.toml, e.g. `git tag v0.1.0 && git push origin v0.1.0`.
|
|
5
|
+
|
|
6
|
+
on:
|
|
7
|
+
push:
|
|
8
|
+
tags: ["v*"]
|
|
9
|
+
|
|
10
|
+
jobs:
|
|
11
|
+
build:
|
|
12
|
+
runs-on: ubuntu-latest
|
|
13
|
+
steps:
|
|
14
|
+
- uses: actions/checkout@v7
|
|
15
|
+
- uses: astral-sh/setup-uv@v10.2.0
|
|
16
|
+
- name: Check the tag matches the package version
|
|
17
|
+
run: |
|
|
18
|
+
version="$(uv version --short)"
|
|
19
|
+
if [ "v$version" != "$GITHUB_REF_NAME" ]; then
|
|
20
|
+
echo "Tag $GITHUB_REF_NAME does not match pyproject version $version" >&2
|
|
21
|
+
exit 1
|
|
22
|
+
fi
|
|
23
|
+
- run: uv run pytest -q
|
|
24
|
+
- run: uv build
|
|
25
|
+
- uses: actions/upload-artifact@v7
|
|
26
|
+
with:
|
|
27
|
+
name: dist
|
|
28
|
+
path: dist/
|
|
29
|
+
|
|
30
|
+
publish:
|
|
31
|
+
needs: build
|
|
32
|
+
runs-on: ubuntu-latest
|
|
33
|
+
environment:
|
|
34
|
+
name: pypi
|
|
35
|
+
url: https://pypi.org/p/stackdoctor
|
|
36
|
+
permissions:
|
|
37
|
+
id-token: write # required for trusted publishing
|
|
38
|
+
steps:
|
|
39
|
+
- uses: actions/download-artifact@v8
|
|
40
|
+
with:
|
|
41
|
+
name: dist
|
|
42
|
+
path: dist/
|
|
43
|
+
- uses: pypa/gh-action-pypi-publish@release/v1
|
|
@@ -0,0 +1,23 @@
|
|
|
1
|
+
name: tests
|
|
2
|
+
|
|
3
|
+
on:
|
|
4
|
+
push:
|
|
5
|
+
pull_request:
|
|
6
|
+
workflow_dispatch:
|
|
7
|
+
|
|
8
|
+
jobs:
|
|
9
|
+
unit:
|
|
10
|
+
name: ${{ matrix.os }} / py${{ matrix.python }}
|
|
11
|
+
runs-on: ${{ matrix.os }}
|
|
12
|
+
strategy:
|
|
13
|
+
fail-fast: false
|
|
14
|
+
matrix:
|
|
15
|
+
os: [ubuntu-latest, macos-latest, windows-latest]
|
|
16
|
+
python: ["3.11", "3.12", "3.13"]
|
|
17
|
+
steps:
|
|
18
|
+
- uses: actions/checkout@v7
|
|
19
|
+
- uses: astral-sh/setup-uv@v10.2.0
|
|
20
|
+
with:
|
|
21
|
+
python-version: ${{ matrix.python }}
|
|
22
|
+
- name: Run unit tests
|
|
23
|
+
run: uv run --python ${{ matrix.python }} pytest -q
|
|
@@ -0,0 +1,161 @@
|
|
|
1
|
+
# Demo without Docker: "why are my jobs stuck?"
|
|
2
|
+
|
|
3
|
+
This reproduces the *worker stopped* scenario on a Mac with Homebrew Postgres and Redis, with no Docker
|
|
4
|
+
needed. It was tested on macOS 12 with `postgresql@15` and `redis`. It takes about 5 minutes, and the
|
|
5
|
+
last step cleans everything up.
|
|
6
|
+
|
|
7
|
+
You'll use **three terminals**, all in the repo folder:
|
|
8
|
+
|
|
9
|
+
| Terminal | Runs |
|
|
10
|
+
|---|---|
|
|
11
|
+
| **A** | setup, breaking things, checking the result |
|
|
12
|
+
| **B** | the Celery worker (keep it visible while recording) |
|
|
13
|
+
| **C** | Claude Code (or use Claude Desktop) |
|
|
14
|
+
|
|
15
|
+
## 1. Setup (terminal A)
|
|
16
|
+
|
|
17
|
+
```sh
|
|
18
|
+
cd ~/Documents/"stack doctor"
|
|
19
|
+
source "$HOME/.local/bin/env" # puts uv on PATH
|
|
20
|
+
uv sync
|
|
21
|
+
|
|
22
|
+
export PATH="$(brew --prefix postgresql@15)/bin:$PATH"
|
|
23
|
+
export LANG=en_US.UTF-8 LC_ALL=en_US.UTF-8 # initdb needs a UTF-8 locale
|
|
24
|
+
export SD=/tmp/sd-demo
|
|
25
|
+
rm -rf "$SD" && mkdir -p "$SD"
|
|
26
|
+
```
|
|
27
|
+
|
|
28
|
+
Start Postgres on port 55432 and load the demo schema, 1M orders. Loading takes about 15 seconds.
|
|
29
|
+
|
|
30
|
+
```sh
|
|
31
|
+
initdb -D "$SD/pg" -U shop --auth=trust >/dev/null
|
|
32
|
+
pg_ctl -D "$SD/pg" -l "$SD/postgres.log" -o "-p 55432 -c unix_socket_directories='' \
|
|
33
|
+
-c shared_preload_libraries=pg_stat_statements -c log_lock_waits=on \
|
|
34
|
+
-c deadlock_timeout=1s -c log_line_prefix='%m [%p] '" start
|
|
35
|
+
createdb -h localhost -p 55432 -U shop shop
|
|
36
|
+
psql -q -h localhost -p 55432 -U shop -d shop -f demo/postgres/init.sql
|
|
37
|
+
```
|
|
38
|
+
|
|
39
|
+
Start Redis on port 56379:
|
|
40
|
+
|
|
41
|
+
```sh
|
|
42
|
+
redis-server --port 56379 --maxmemory 64mb --maxmemory-policy noeviction \
|
|
43
|
+
--daemonize yes --dir "$SD" --logfile "$SD/redis.log"
|
|
44
|
+
```
|
|
45
|
+
|
|
46
|
+
Write the stackdoctor config. It connects with the read-only role that `init.sql` created:
|
|
47
|
+
|
|
48
|
+
```sh
|
|
49
|
+
cat > "$SD/stackdoctor.env" <<EOF
|
|
50
|
+
DATABASE_URL=postgresql://stackdoctor_ro:readonly@localhost:55432/shop
|
|
51
|
+
REDIS_URL=redis://localhost:56379/0
|
|
52
|
+
CELERY_BROKER_URL=redis://localhost:56379/0
|
|
53
|
+
CELERY_RESULT_BACKEND=redis://localhost:56379/1
|
|
54
|
+
LOG_SOURCES=$SD/worker.log,$SD/postgres.log
|
|
55
|
+
EXPECTED_WORKERS=1
|
|
56
|
+
QUEUE_THRESHOLD=20
|
|
57
|
+
EOF
|
|
58
|
+
```
|
|
59
|
+
|
|
60
|
+
## 2. Start the worker (terminal B)
|
|
61
|
+
|
|
62
|
+
```sh
|
|
63
|
+
cd ~/Documents/"stack doctor"/demo/app
|
|
64
|
+
export SD=/tmp/sd-demo
|
|
65
|
+
../../.venv/bin/celery -A tasks worker --pool threads --concurrency 2 -n worker1@%h \
|
|
66
|
+
--loglevel INFO --pidfile "$SD/worker.pid" 2>&1 | tee -a "$SD/worker.log"
|
|
67
|
+
```
|
|
68
|
+
|
|
69
|
+
Two details matter here:
|
|
70
|
+
|
|
71
|
+
- **`tee` instead of `--logfile`.** Celery prints `worker: Warm shutdown` to stdout only, so the log
|
|
72
|
+
file has to capture stdout for stackdoctor to see when the worker stopped.
|
|
73
|
+
- **`--pool threads`.** Celery's default prefork pool is unreliable on macOS with Python 3.13. The
|
|
74
|
+
Docker demo uses prefork on Linux.
|
|
75
|
+
- **Never stop it with Ctrl+C. Use `kill -TERM` from terminal A** (step 4). Ctrl+C sends SIGINT to the
|
|
76
|
+
whole foreground pipeline, so `tee` dies together with Celery. A moment later Celery prints
|
|
77
|
+
`worker: Warm shutdown` into a pipe nobody is reading, and the line never reaches `worker.log`.
|
|
78
|
+
`kill -TERM` signals only the Celery process, so `tee` stays alive and writes the line to the file.
|
|
79
|
+
|
|
80
|
+
Back in **terminal A**, check that everything is healthy:
|
|
81
|
+
|
|
82
|
+
```sh
|
|
83
|
+
(cd demo/app && ../../.venv/bin/python enqueue.py 6) # B shows 6 tasks succeed
|
|
84
|
+
```
|
|
85
|
+
|
|
86
|
+
## 3. Connect Claude (terminal C)
|
|
87
|
+
|
|
88
|
+
Use the repo's own venv, so you don't need PyPI and avoid the macOS 12 `uvx`/`realpath` issue:
|
|
89
|
+
|
|
90
|
+
```sh
|
|
91
|
+
cd ~/Documents/"stack doctor"
|
|
92
|
+
claude mcp add stackdoctor -e STACKDOCTOR_ENV_FILE=/tmp/sd-demo/stackdoctor.env \
|
|
93
|
+
-- "$PWD/.venv/bin/stackdoctor"
|
|
94
|
+
claude
|
|
95
|
+
```
|
|
96
|
+
|
|
97
|
+
If you're recording in **Claude Desktop** instead, add this to
|
|
98
|
+
`~/Library/Application Support/Claude/claude_desktop_config.json` and restart the app:
|
|
99
|
+
|
|
100
|
+
```json
|
|
101
|
+
{
|
|
102
|
+
"mcpServers": {
|
|
103
|
+
"stackdoctor": {
|
|
104
|
+
"command": "/Users/leprismacbookpro/Documents/stack doctor/.venv/bin/stackdoctor",
|
|
105
|
+
"env": { "STACKDOCTOR_ENV_FILE": "/tmp/sd-demo/stackdoctor.env" }
|
|
106
|
+
}
|
|
107
|
+
}
|
|
108
|
+
}
|
|
109
|
+
```
|
|
110
|
+
|
|
111
|
+
## 4. Break it (terminal A)
|
|
112
|
+
|
|
113
|
+
```sh
|
|
114
|
+
kill -TERM $(cat "$SD/worker.pid") # stop the worker (not Ctrl+C in B, see step 2)
|
|
115
|
+
while [ -f "$SD/worker.pid" ]; do sleep 1; done # wait until the worker has fully exited
|
|
116
|
+
grep -i "warm shutdown" "$SD/worker.log" # must print: worker: Warm shutdown (MainProcess)
|
|
117
|
+
(cd demo/app && ../../.venv/bin/python enqueue.py 40) # these 40 tasks have nobody to run them
|
|
118
|
+
```
|
|
119
|
+
|
|
120
|
+
If the `grep` prints nothing, the worker was stopped some other way (for example Ctrl+C) and
|
|
121
|
+
stackdoctor can't see the shutdown. Restart the worker (step 2) and stop it again with `kill -TERM`.
|
|
122
|
+
|
|
123
|
+
## 5. Ask (terminal C)
|
|
124
|
+
|
|
125
|
+
> why are my jobs stuck?
|
|
126
|
+
|
|
127
|
+
Claude calls `diagnose("why are my jobs stuck?")`. Expect roughly:
|
|
128
|
+
|
|
129
|
+
- **findings:**
|
|
130
|
+
- `no_workers` (critical): No Celery workers replied to ping
|
|
131
|
+
- `queue_no_consumer` (critical): Queue 'celery' has 40 messages and no live worker consumes it
|
|
132
|
+
- `queue_backlog`: 40 waiting messages (threshold 20)
|
|
133
|
+
- `worker_shutdown` from `worker.log`: `worker: Warm shutdown (MainProcess)`
|
|
134
|
+
- **possible cause** (confidence: medium): "Celery worker went down → queue is not being consumed". The cause is the
|
|
135
|
+
`Warm shutdown` line at the moment you ran `kill`, and the effects are the backlog and the
|
|
136
|
+
unconsumed queue, each with a timestamp.
|
|
137
|
+
|
|
138
|
+
To check the same output without an AI (for example, before you hit record):
|
|
139
|
+
|
|
140
|
+
```sh
|
|
141
|
+
STACKDOCTOR_ENV_FILE="$SD/stackdoctor.env" uv run python -c "
|
|
142
|
+
import asyncio, json; from stackdoctor.server import diagnose
|
|
143
|
+
out = asyncio.run(diagnose('why are my jobs stuck?'))
|
|
144
|
+
print(json.dumps({k: out[k] for k in ('findings', 'possible_causes')}, indent=2))"
|
|
145
|
+
```
|
|
146
|
+
|
|
147
|
+
## 6. Recover, then clean up
|
|
148
|
+
|
|
149
|
+
Restart the worker in **terminal B** with the same command as step 2. It drains the 40 tasks, and
|
|
150
|
+
asking again shows a healthy stack.
|
|
151
|
+
|
|
152
|
+
When you're done, run this in terminal A. The first two lines stop the worker the same way as step 4:
|
|
153
|
+
|
|
154
|
+
```sh
|
|
155
|
+
kill -TERM $(cat "$SD/worker.pid")
|
|
156
|
+
while [ -f "$SD/worker.pid" ]; do sleep 1; done
|
|
157
|
+
claude mcp remove stackdoctor
|
|
158
|
+
redis-cli -p 56379 shutdown nosave
|
|
159
|
+
pg_ctl -D "$SD/pg" stop -m fast
|
|
160
|
+
rm -rf "$SD"
|
|
161
|
+
```
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 stackdoctor contributors
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,341 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: stackdoctor
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Read-only MCP server that diagnoses a Python backend stack (Postgres, Celery, Redis, logs) with cross-system correlation.
|
|
5
|
+
Project-URL: Homepage, https://github.com/lepri89/stackdoctor
|
|
6
|
+
Project-URL: Issues, https://github.com/lepri89/stackdoctor/issues
|
|
7
|
+
License: MIT
|
|
8
|
+
License-File: LICENSE
|
|
9
|
+
Keywords: celery,diagnostics,mcp,observability,postgres,redis
|
|
10
|
+
Requires-Python: >=3.11
|
|
11
|
+
Requires-Dist: celery>=5.3
|
|
12
|
+
Requires-Dist: mcp<3,>=2.3
|
|
13
|
+
Requires-Dist: psycopg[binary]>=3.1
|
|
14
|
+
Requires-Dist: python-dotenv>=1.0
|
|
15
|
+
Requires-Dist: redis>=5.0
|
|
16
|
+
Requires-Dist: sqlglot>=25
|
|
17
|
+
Description-Content-Type: text/markdown
|
|
18
|
+
|
|
19
|
+
# stackdoctor
|
|
20
|
+
|
|
21
|
+
[](https://github.com/lepri89/stackdoctor/actions/workflows/tests.yml)
|
|
22
|
+
[](https://github.com/lepri89/stackdoctor/actions/workflows/demo.yml)
|
|
23
|
+

|
|
24
|
+

|
|
25
|
+
|
|
26
|
+

|
|
27
|
+
|
|
28
|
+
*Claude finds that the worker shut down and 40 jobs are waiting, in one diagnose() call.*
|
|
29
|
+
|
|
30
|
+
> **Strictly read-only.** stackdoctor never writes, deletes, restarts, retries, revokes or sends anything.
|
|
31
|
+
> Postgres sessions are forced read-only by the server, Redis commands go through an allowlist,
|
|
32
|
+
> and Celery is only *inspected*. Secrets are redacted from every response.
|
|
33
|
+
|
|
34
|
+
stackdoctor is an MCP server that lets Claude, Cursor or any MCP client diagnose a Python backend stack
|
|
35
|
+
(**Postgres, Celery, Redis and logs**) from a single tool call.
|
|
36
|
+
|
|
37
|
+
Ask *"why are my jobs stuck?"* and the assistant calls `diagnose(symptom)`. It runs every relevant check
|
|
38
|
+
in parallel and returns one timestamped snapshot with:
|
|
39
|
+
|
|
40
|
+
- **findings**: blocked queries, missing workers, queue backlogs, log error spikes, Redis memory pressure, …
|
|
41
|
+
- **a merged timeline** of events from Postgres, Celery, Redis and your logs
|
|
42
|
+
- **possible cause → effect chains** when events line up in time, each with its evidence timestamps
|
|
43
|
+
|
|
44
|
+
```text
|
|
45
|
+
possible cause (confidence: medium): Database lock → blocked queries → stuck or failing tasks
|
|
46
|
+
cause: 21:58:01 postgres.lock_held pid 7959 holds a lock blocking 2 sessions: LOCK TABLE orders …
|
|
47
|
+
effect: 21:58:02 logs.db_lock_wait [postgres] process 7962 still waiting for RowExclusiveLock …
|
|
48
|
+
effect: 21:58:21 celery.task_failed tasks.process_order[…] failed: LockNotAvailable: lock timeout
|
|
49
|
+
effect: 21:58:21 celery.task_long_running tasks.process_order[…] running 13.2s on worker1
|
|
50
|
+
effect: 21:58:34 celery.workers_saturated worker1: all 2 slots busy, 2 reserved
|
|
51
|
+
caveat: Inferred from timing only. Verify before acting; this is not a confirmed root cause.
|
|
52
|
+
```
|
|
53
|
+
|
|
54
|
+
## Install
|
|
55
|
+
|
|
56
|
+
You need [uv](https://docs.astral.sh/uv/getting-started/installation/):
|
|
57
|
+
|
|
58
|
+
```sh
|
|
59
|
+
curl -LsSf https://astral.sh/uv/install.sh | sh # macOS / Linux
|
|
60
|
+
powershell -c "irm https://astral.sh/uv/install.ps1 | iex" # Windows
|
|
61
|
+
```
|
|
62
|
+
|
|
63
|
+
Then one command runs the server, with no other install step:
|
|
64
|
+
|
|
65
|
+
```sh
|
|
66
|
+
uvx stackdoctor
|
|
67
|
+
```
|
|
68
|
+
|
|
69
|
+
Until the package is on PyPI, run it from a checkout with `uvx --from /path/to/stackdoctor stackdoctor`,
|
|
70
|
+
or from git with `uvx --from git+https://github.com/lepri89/stackdoctor stackdoctor`.
|
|
71
|
+
|
|
72
|
+
> **macOS 12 (Monterey):** `uvx stackdoctor` fails with `realpath: command not found`, because uv's
|
|
73
|
+
> launcher script needs `realpath`, which only ships with macOS 13+. Use
|
|
74
|
+
> `uvx --from stackdoctor python -m stackdoctor` instead. In client configs that means
|
|
75
|
+
> `"args": ["--from", "stackdoctor", "python", "-m", "stackdoctor"]`. Alternatively, run
|
|
76
|
+
> `uv tool install stackdoctor` once and use `stackdoctor` as the command.
|
|
77
|
+
|
|
78
|
+
## Configure
|
|
79
|
+
|
|
80
|
+
Set environment variables in your MCP client config, or in a `.env` file. stackdoctor looks for `.env` in
|
|
81
|
+
the working directory, or uses the file named by `STACKDOCTOR_ENV_FILE`.
|
|
82
|
+
**Any source you don't configure is skipped**, so it's fine to start with only `DATABASE_URL`.
|
|
83
|
+
|
|
84
|
+
| Variable | Example | Used for |
|
|
85
|
+
|---|---|---|
|
|
86
|
+
| `DATABASE_URL` | `postgresql://stackdoctor_ro:…@localhost:5432/app` | Postgres checks |
|
|
87
|
+
| `REDIS_URL` | `redis://localhost:6379/0` | Redis checks |
|
|
88
|
+
| `CELERY_BROKER_URL` | `redis://localhost:6379/0` (RabbitMQ `amqp://…` is experimental) | Celery inspect + queue lengths |
|
|
89
|
+
| `CELERY_RESULT_BACKEND` | `redis://localhost:6379/1` | failed tasks / task details |
|
|
90
|
+
| `CELERY_APP` | `myproject.celery:app` | optional: use your app's queues/routes/config (run from your project dir) |
|
|
91
|
+
| `CELERY_QUEUES` | `default,emails` | extra queue names to measure |
|
|
92
|
+
| `LOG_SOURCES` | `./logs/worker.log,docker:api,docker:worker` | log files and/or docker containers |
|
|
93
|
+
|
|
94
|
+
`LOG_SOURCES` entries are file paths, `docker:<container>`, or a bare container name.
|
|
95
|
+
Only configured sources can be read.
|
|
96
|
+
|
|
97
|
+
> **Celery workers: point `LOG_SOURCES` at the worker's stdout.** Celery prints
|
|
98
|
+
> `worker: Warm shutdown (MainProcess)` straight to stdout, not through logging, so it never reaches a
|
|
99
|
+
> `--logfile`. Docker containers already capture stdout. For a file, redirect stdout to it
|
|
100
|
+
> (`celery … worker >> worker.log 2>&1`, or supervisor/systemd stdout capture) so stackdoctor can see
|
|
101
|
+
> when a worker stopped.
|
|
102
|
+
|
|
103
|
+
<details>
|
|
104
|
+
<summary>Thresholds and limits</summary>
|
|
105
|
+
|
|
106
|
+
| Variable | Default | Meaning |
|
|
107
|
+
|---|---|---|
|
|
108
|
+
| `QUEUE_THRESHOLD` | 100 | messages waiting before a queue counts as a backlog |
|
|
109
|
+
| `EXPECTED_WORKERS` | 0 (off) | report missing workers if fewer reply |
|
|
110
|
+
| `LONG_TASK_S` / `LONG_QUERY_S` | 60 / 30 | when a task / query counts as long-running |
|
|
111
|
+
| `REDIS_MEM_WARN_PCT` | 85 | % of `maxmemory` that counts as "near the limit" |
|
|
112
|
+
| `LOG_ERROR_SPIKE` | 10 | problem lines in 5 minutes that count as a spike |
|
|
113
|
+
| `CHAIN_WINDOW_MIN` | 30 | how far apart cause and effect may be |
|
|
114
|
+
| `CHECK_TIMEOUT_S` | 8 | per-check timeout inside `diagnose` |
|
|
115
|
+
| `CELERY_INSPECT_TIMEOUT_S` | 1.0 | how long to wait for worker replies |
|
|
116
|
+
| `MAX_OUTPUT_CHARS` | 20000 | per-tool response cap (`diagnose` gets 2×) |
|
|
117
|
+
|
|
118
|
+
</details>
|
|
119
|
+
|
|
120
|
+
### Claude Desktop
|
|
121
|
+
|
|
122
|
+
`~/Library/Application Support/Claude/claude_desktop_config.json` (macOS) or
|
|
123
|
+
`%APPDATA%\Claude\claude_desktop_config.json` (Windows):
|
|
124
|
+
|
|
125
|
+
```json
|
|
126
|
+
{
|
|
127
|
+
"mcpServers": {
|
|
128
|
+
"stackdoctor": {
|
|
129
|
+
"command": "uvx",
|
|
130
|
+
"args": ["stackdoctor"],
|
|
131
|
+
"env": {
|
|
132
|
+
"DATABASE_URL": "postgresql://stackdoctor_ro:password@localhost:5432/app",
|
|
133
|
+
"REDIS_URL": "redis://localhost:6379/0",
|
|
134
|
+
"CELERY_BROKER_URL": "redis://localhost:6379/0",
|
|
135
|
+
"CELERY_RESULT_BACKEND": "redis://localhost:6379/1",
|
|
136
|
+
"LOG_SOURCES": "docker:api,docker:worker"
|
|
137
|
+
}
|
|
138
|
+
}
|
|
139
|
+
}
|
|
140
|
+
}
|
|
141
|
+
```
|
|
142
|
+
|
|
143
|
+
GUI apps often don't see your shell `PATH`. If Claude Desktop can't find `uvx`, use the full path:
|
|
144
|
+
`/Users/<username>/.local/bin/uvx` on macOS, or `C:\\Users\\<username>\\.local\\bin\\uvx.exe` on Windows.
|
|
145
|
+
|
|
146
|
+
### Claude Code
|
|
147
|
+
|
|
148
|
+
```sh
|
|
149
|
+
claude mcp add stackdoctor \
|
|
150
|
+
-e DATABASE_URL=postgresql://stackdoctor_ro:password@localhost:5432/app \
|
|
151
|
+
-e REDIS_URL=redis://localhost:6379/0 \
|
|
152
|
+
-e CELERY_BROKER_URL=redis://localhost:6379/0 \
|
|
153
|
+
-e LOG_SOURCES=docker:api,docker:worker \
|
|
154
|
+
-- uvx stackdoctor
|
|
155
|
+
```
|
|
156
|
+
|
|
157
|
+
Or, from a project directory with a `.env` file, simply `claude mcp add stackdoctor -- uvx stackdoctor`.
|
|
158
|
+
|
|
159
|
+
### Cursor
|
|
160
|
+
|
|
161
|
+
`.cursor/mcp.json` in your project, or `~/.cursor/mcp.json` globally. This uses the same
|
|
162
|
+
`mcpServers` format as Claude Desktop:
|
|
163
|
+
|
|
164
|
+
```json
|
|
165
|
+
{
|
|
166
|
+
"mcpServers": {
|
|
167
|
+
"stackdoctor": {
|
|
168
|
+
"command": "uvx",
|
|
169
|
+
"args": ["stackdoctor"],
|
|
170
|
+
"env": { "STACKDOCTOR_ENV_FILE": "${workspaceFolder}/.env" }
|
|
171
|
+
}
|
|
172
|
+
}
|
|
173
|
+
}
|
|
174
|
+
```
|
|
175
|
+
|
|
176
|
+
### VS Code (Copilot agent mode)
|
|
177
|
+
|
|
178
|
+
`.vscode/mcp.json`:
|
|
179
|
+
|
|
180
|
+
```json
|
|
181
|
+
{
|
|
182
|
+
"servers": {
|
|
183
|
+
"stackdoctor": {
|
|
184
|
+
"type": "stdio",
|
|
185
|
+
"command": "uvx",
|
|
186
|
+
"args": ["stackdoctor"],
|
|
187
|
+
"env": { "STACKDOCTOR_ENV_FILE": "${workspaceFolder}/.env" }
|
|
188
|
+
}
|
|
189
|
+
}
|
|
190
|
+
}
|
|
191
|
+
```
|
|
192
|
+
|
|
193
|
+
## Tools
|
|
194
|
+
|
|
195
|
+
| Tool | What it returns |
|
|
196
|
+
|---|---|
|
|
197
|
+
| `diagnose(symptom)` | **Start here.** Runs the relevant checks concurrently, each with its own timeout, and returns findings, a merged timeline and possible causes |
|
|
198
|
+
| `active_queries(min_duration_s)` | non-idle sessions, incl. *idle in transaction* |
|
|
199
|
+
| `blocking_locks()` | waiting sessions and the sessions blocking them |
|
|
200
|
+
| `slow_queries(limit)` | top statements from `pg_stat_statements` (explains how to enable it if missing) and seq-scan-heavy tables |
|
|
201
|
+
| `run_select(sql, limit)` | one validated `SELECT` / `WITH` / `EXPLAIN` with a row limit |
|
|
202
|
+
| `workers()` | live workers, their queues, concurrency, active/reserved tasks |
|
|
203
|
+
| `queue_lengths(queues)` | messages waiting per queue, read from the broker (Redis incl. priority queues; RabbitMQ experimental) |
|
|
204
|
+
| `failed_tasks(limit)` | recent failures from a Redis result backend, or *why* they aren't visible |
|
|
205
|
+
| `task_details(task_id)` | stored result/traceback, and whether a worker holds the task right now |
|
|
206
|
+
| `memory_and_clients()` | Redis memory vs `maxmemory`, evictions, clients, slowlog |
|
|
207
|
+
| `scan_keys(pattern, limit)` | key names via `SCAN` |
|
|
208
|
+
| `key_info(key)` | type, TTL, size, encoding, short redacted preview |
|
|
209
|
+
| `tail_logs(source, lines)` | last lines of a configured log source |
|
|
210
|
+
| `search_logs(pattern, since_minutes, source)` | regex search over recent log lines |
|
|
211
|
+
|
|
212
|
+
Postgres checks are intentionally light. For deep Postgres health (index advice, vacuum, bloat), run
|
|
213
|
+
[Postgres MCP Pro](https://github.com/crystaldba/postgres-mcp) next to stackdoctor.
|
|
214
|
+
|
|
215
|
+
## Safety
|
|
216
|
+
|
|
217
|
+
stackdoctor is built so that a confused or prompt-injected assistant still can't change your systems.
|
|
218
|
+
|
|
219
|
+
**Postgres**
|
|
220
|
+
- Every session is opened with `default_transaction_read_only=on`, `statement_timeout=5s` and
|
|
221
|
+
`lock_timeout=2s`, passed as connection options so they override your URL. stackdoctor also checks
|
|
222
|
+
the setting and refuses to run if the session isn't read-only.
|
|
223
|
+
- `run_select` parses SQL with [sqlglot](https://github.com/tobymao/sqlglot), not regexes. It accepts
|
|
224
|
+
exactly one `SELECT` / `WITH … SELECT` / `EXPLAIN` statement and rejects:
|
|
225
|
+
- writable CTEs, `SELECT … INTO` and `FOR UPDATE/SHARE`
|
|
226
|
+
- `EXPLAIN ANALYZE`, because it executes the query
|
|
227
|
+
- functions with side effects, even inside a read-only transaction: `pg_terminate_backend`,
|
|
228
|
+
`pg_cancel_backend`, `pg_reload_conf`, `pg_read_file`, `pg_read_binary_file`, `pg_ls_dir`, `lo_*`,
|
|
229
|
+
`dblink*`, `set_config`, `pg_advisory_*`, `query_to_xml` (which runs SQL from a string), `pg_sleep`, …
|
|
230
|
+
- Results are always limited: queries are wrapped as `SELECT * FROM (<q>) AS sd_sub LIMIT n`
|
|
231
|
+
(default 100, max 1000).
|
|
232
|
+
|
|
233
|
+
**Recommended: a dedicated read-only role.** Defense in depth, so the database enforces the rules too:
|
|
234
|
+
|
|
235
|
+
```sql
|
|
236
|
+
CREATE ROLE stackdoctor_ro LOGIN PASSWORD 'change-me';
|
|
237
|
+
GRANT CONNECT ON DATABASE app TO stackdoctor_ro;
|
|
238
|
+
GRANT USAGE ON SCHEMA public TO stackdoctor_ro;
|
|
239
|
+
GRANT SELECT ON ALL TABLES IN SCHEMA public TO stackdoctor_ro;
|
|
240
|
+
ALTER DEFAULT PRIVILEGES IN SCHEMA public GRANT SELECT ON TABLES TO stackdoctor_ro;
|
|
241
|
+
GRANT pg_monitor TO stackdoctor_ro; -- see other sessions' queries and pg_stat_statements
|
|
242
|
+
ALTER ROLE stackdoctor_ro SET default_transaction_read_only = on;
|
|
243
|
+
```
|
|
244
|
+
|
|
245
|
+
Anything the role can `SELECT`, the assistant can read. If some tables hold data you don't want to
|
|
246
|
+
share with an AI, leave them out of the `GRANT SELECT`.
|
|
247
|
+
|
|
248
|
+
**Redis.** Every command is checked against an allowlist that is aware of subcommands (`INFO`, `SCAN`,
|
|
249
|
+
`TYPE`, `PTTL`, `LLEN`, `GET`, …). `CLIENT LIST` is allowed; `CLIENT KILL` / `PAUSE` are not. The same goes
|
|
250
|
+
for `OBJECT`, `MEMORY` and `CONFIG` (only `CONFIG GET`). `KEYS` is never used; listing always goes through `SCAN`.
|
|
251
|
+
|
|
252
|
+
**Celery.** Only `inspect` (`ping`, `active`, `reserved`, `active_queues`, `stats`, `query_task`) and
|
|
253
|
+
broker reads are used: `LLEN` on Redis, and passive `queue_declare` on RabbitMQ.
|
|
254
|
+
|
|
255
|
+
> **RabbitMQ support is experimental.** Queue lengths via passive `queue_declare` and the `inspect`
|
|
256
|
+
> calls are implemented but not yet covered by CI. The Redis broker is the tested path. Reports welcome. stackdoctor never revokes,
|
|
257
|
+
retries, shuts down or sends tasks.
|
|
258
|
+
*To be precise:* `inspect` works by publishing a broadcast message on Celery's **control channel**
|
|
259
|
+
(`celery.pidbox`) and reading replies from a temporary reply queue. That is how Celery's own
|
|
260
|
+
`celery inspect` command works. It doesn't touch your task queues or results, but it isn't zero
|
|
261
|
+
traffic. Everything else stackdoctor does is pure reads.
|
|
262
|
+
|
|
263
|
+
**Redaction.** Every response is redacted before it leaves the server. This covers:
|
|
264
|
+
- passwords in URLs (`postgres://user:***@host`)
|
|
265
|
+
- `Authorization` headers and bearer tokens
|
|
266
|
+
- `PASSWORD=`, `SECRET_KEY=`, `api_key:` and other secret-looking keys
|
|
267
|
+
- AWS, GitHub, Slack, Stripe, OpenAI/Anthropic-style and Google keys, JWTs, and private keys
|
|
268
|
+
|
|
269
|
+
Task args/kwargs and Redis value previews are truncated and redacted too. Redaction is pattern-based,
|
|
270
|
+
so treat it as a safety net, not a guarantee.
|
|
271
|
+
|
|
272
|
+
**Output caps.** Every tool caps its response (`MAX_OUTPUT_CHARS`) by trimming lists and long strings,
|
|
273
|
+
so a huge table or log can't flood the assistant's context.
|
|
274
|
+
|
|
275
|
+
**Logs.** Only sources listed in `LOG_SOURCES` can be read, so the assistant can't ask for `/etc/passwd`.
|
|
276
|
+
Docker logs are read with `docker logs`, called without a shell.
|
|
277
|
+
|
|
278
|
+
## Try the demo
|
|
279
|
+
|
|
280
|
+
The demo is a small FastAPI + Celery + Redis + Postgres app with scripts that break it in realistic ways.
|
|
281
|
+
CI runs every scenario on each push ([`demo.yml`](.github/workflows/demo.yml)): it breaks the stack,
|
|
282
|
+
calls `diagnose()` through a real MCP stdio client and checks the findings and chains.
|
|
283
|
+
To try it without Docker (local Postgres, Redis and worker), see [DEMO.md](DEMO.md).
|
|
284
|
+
|
|
285
|
+
```sh
|
|
286
|
+
cd demo
|
|
287
|
+
docker compose up -d --build # first run builds the app and loads 1M rows (~1 min)
|
|
288
|
+
cp .env.example .env # stackdoctor config for the demo (uses the read-only role)
|
|
289
|
+
```
|
|
290
|
+
|
|
291
|
+
Point your MCP client at the demo by setting `STACKDOCTOR_ENV_FILE` to the absolute path of `demo/.env`.
|
|
292
|
+
Then break something and ask:
|
|
293
|
+
|
|
294
|
+
| Script (macOS/Linux · Windows) | What it does | Ask |
|
|
295
|
+
|---|---|---|
|
|
296
|
+
| `./break_worker.sh` · `.\break_worker.ps1` | stops the Celery worker and enqueues 50 tasks | "why are my jobs stuck?" |
|
|
297
|
+
| `./lock_table.sh` · `.\lock_table.ps1` | holds an `ACCESS EXCLUSIVE` lock on `orders` for 3 min, then enqueues tasks that block on it | "why are my jobs stuck?" |
|
|
298
|
+
| `./slow_query.sh` · `.\slow_query.ps1` | runs three ~1-minute sequential scans | "why is the API slow?" |
|
|
299
|
+
| `./reset.sh` · `.\reset.ps1` | restarts the worker and ends the demo's locks and slow queries | |
|
|
300
|
+
|
|
301
|
+
**Colima (macOS).** `docker compose` works unchanged once Colima is running:
|
|
302
|
+
`colima start --cpu 2 --memory 4`. On macOS 13+ you can add `--vm-type vz`. On macOS 12 Colima uses
|
|
303
|
+
QEMU, which must be installed first (`brew install qemu`). **Windows:** use Docker Desktop or Docker
|
|
304
|
+
in WSL2 and run the `.ps1` scripts from PowerShell.
|
|
305
|
+
|
|
306
|
+
## Why stackdoctor instead of separate Postgres, Celery and log MCPs?
|
|
307
|
+
|
|
308
|
+
| | Separate MCP servers | stackdoctor |
|
|
309
|
+
|---|---|---|
|
|
310
|
+
| Install | one server per system, each with its own config | one `uvx stackdoctor` |
|
|
311
|
+
| Celery | usually needs Flower running | inspect API + broker, no Flower |
|
|
312
|
+
| Answering "why is it stuck?" | the AI calls 5–10 tools one after another, each a different moment in time | one `diagnose()` call: all checks run in parallel and share one timestamp |
|
|
313
|
+
| Correlation | left to the AI, across separate outputs | merged timeline + timing-based cause → effect hypotheses, with evidence |
|
|
314
|
+
| Safety | varies per server | one read-only policy for every system, redaction and output caps everywhere |
|
|
315
|
+
| Postgres depth | Postgres MCP Pro is deeper | intentionally light; use both |
|
|
316
|
+
|
|
317
|
+
## Development
|
|
318
|
+
|
|
319
|
+
```sh
|
|
320
|
+
uv sync
|
|
321
|
+
uv run pytest # unit tests (safety layer, diagnose, logs)
|
|
322
|
+
STACKDOCTOR_TEST_DATABASE_URL=postgresql://shop:shop@localhost:55432/shop \
|
|
323
|
+
STACKDOCTOR_TEST_REDIS_URL=redis://localhost:56379/0 \
|
|
324
|
+
uv run pytest # plus live tests against the demo
|
|
325
|
+
```
|
|
326
|
+
|
|
327
|
+
## Releasing
|
|
328
|
+
|
|
329
|
+
Releases go to PyPI from GitHub Actions with
|
|
330
|
+
[trusted publishing](https://docs.pypi.org/trusted-publishers/), so no API token is stored anywhere.
|
|
331
|
+
|
|
332
|
+
1. One-time: on PyPI, add a *pending publisher* (Account → Publishing) with project `stackdoctor`, owner
|
|
333
|
+
`lepri89`, repository `stackdoctor`, workflow `release.yml` and environment `pypi`. In GitHub, create
|
|
334
|
+
an environment named `pypi` (Settings → Environments); adding yourself as a required reviewer is a good idea.
|
|
335
|
+
2. Bump `version` in `pyproject.toml`, commit, then tag and push:
|
|
336
|
+
`git tag v0.1.0 && git push origin v0.1.0`.
|
|
337
|
+
The workflow checks that the tag matches the version, runs the tests, builds, and publishes.
|
|
338
|
+
|
|
339
|
+
## License
|
|
340
|
+
|
|
341
|
+
MIT
|