stackdoctor 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (44) hide show
  1. stackdoctor-0.1.0/.gitattributes +2 -0
  2. stackdoctor-0.1.0/.github/workflows/demo.yml +53 -0
  3. stackdoctor-0.1.0/.github/workflows/release.yml +43 -0
  4. stackdoctor-0.1.0/.github/workflows/tests.yml +23 -0
  5. stackdoctor-0.1.0/.gitignore +8 -0
  6. stackdoctor-0.1.0/DEMO.md +161 -0
  7. stackdoctor-0.1.0/LICENSE +21 -0
  8. stackdoctor-0.1.0/PKG-INFO +341 -0
  9. stackdoctor-0.1.0/README.md +323 -0
  10. stackdoctor-0.1.0/demo/.env.example +10 -0
  11. stackdoctor-0.1.0/demo/app/Dockerfile +5 -0
  12. stackdoctor-0.1.0/demo/app/api.py +26 -0
  13. stackdoctor-0.1.0/demo/app/enqueue.py +10 -0
  14. stackdoctor-0.1.0/demo/app/requirements.txt +4 -0
  15. stackdoctor-0.1.0/demo/app/tasks.py +29 -0
  16. stackdoctor-0.1.0/demo/break_worker.ps1 +8 -0
  17. stackdoctor-0.1.0/demo/break_worker.sh +9 -0
  18. stackdoctor-0.1.0/demo/docker-compose.yml +51 -0
  19. stackdoctor-0.1.0/demo/e2e.py +134 -0
  20. stackdoctor-0.1.0/demo/lock_table.ps1 +9 -0
  21. stackdoctor-0.1.0/demo/lock_table.sh +12 -0
  22. stackdoctor-0.1.0/demo/postgres/init.sql +20 -0
  23. stackdoctor-0.1.0/demo/reset.ps1 +6 -0
  24. stackdoctor-0.1.0/demo/reset.sh +8 -0
  25. stackdoctor-0.1.0/demo/slow_query.ps1 +8 -0
  26. stackdoctor-0.1.0/demo/slow_query.sh +10 -0
  27. stackdoctor-0.1.0/docs/images/diagnosis.png +0 -0
  28. stackdoctor-0.1.0/pyproject.toml +37 -0
  29. stackdoctor-0.1.0/stackdoctor/__init__.py +0 -0
  30. stackdoctor-0.1.0/stackdoctor/__main__.py +3 -0
  31. stackdoctor-0.1.0/stackdoctor/checks/__init__.py +29 -0
  32. stackdoctor-0.1.0/stackdoctor/checks/celery.py +319 -0
  33. stackdoctor-0.1.0/stackdoctor/checks/logs.py +235 -0
  34. stackdoctor-0.1.0/stackdoctor/checks/postgres.py +202 -0
  35. stackdoctor-0.1.0/stackdoctor/checks/redis.py +154 -0
  36. stackdoctor-0.1.0/stackdoctor/config.py +80 -0
  37. stackdoctor-0.1.0/stackdoctor/diagnose.py +248 -0
  38. stackdoctor-0.1.0/stackdoctor/safety.py +331 -0
  39. stackdoctor-0.1.0/stackdoctor/server.py +164 -0
  40. stackdoctor-0.1.0/tests/test_diagnose.py +113 -0
  41. stackdoctor-0.1.0/tests/test_integration.py +51 -0
  42. stackdoctor-0.1.0/tests/test_logs.py +92 -0
  43. stackdoctor-0.1.0/tests/test_safety.py +247 -0
  44. stackdoctor-0.1.0/uv.lock +1159 -0
@@ -0,0 +1,2 @@
1
+ *.sh text eol=lf
2
+ *.ps1 text eol=crlf
@@ -0,0 +1,53 @@
1
+ name: demo
2
+
3
+ on:
4
+ push:
5
+ pull_request:
6
+ workflow_dispatch:
7
+
8
+ jobs:
9
+ e2e:
10
+ name: docker compose demo, end to end
11
+ runs-on: ubuntu-latest
12
+ timeout-minutes: 30
13
+ steps:
14
+ - uses: actions/checkout@v7
15
+ - uses: astral-sh/setup-uv@v10.2.0
16
+ with:
17
+ python-version: "3.12"
18
+ - run: uv sync
19
+
20
+ - name: Start the demo stack
21
+ working-directory: demo
22
+ run: |
23
+ cp .env.example .env
24
+ docker compose up -d --build --wait --wait-timeout 300
25
+
26
+ - name: Live integration tests (read-only enforcement against the demo)
27
+ env:
28
+ STACKDOCTOR_TEST_DATABASE_URL: postgresql://shop:shop@localhost:55432/shop
29
+ STACKDOCTOR_TEST_REDIS_URL: redis://localhost:56379/0
30
+ run: uv run pytest -q tests/test_integration.py
31
+
32
+ - name: Break, diagnose over MCP stdio, assert, reset
33
+ run: uv run python demo/e2e.py
34
+
35
+ - name: Dump docker logs
36
+ if: failure()
37
+ working-directory: demo
38
+ run: |
39
+ docker compose ps -a
40
+ docker compose logs --no-color --timestamps
41
+
42
+ - name: Upload diagnose() output
43
+ if: always()
44
+ uses: actions/upload-artifact@v7
45
+ with:
46
+ name: diagnose-output
47
+ path: demo/e2e-output/
48
+ if-no-files-found: ignore
49
+
50
+ - name: Stop the demo stack
51
+ if: always()
52
+ working-directory: demo
53
+ run: docker compose down -v
@@ -0,0 +1,43 @@
1
+ name: release
2
+
3
+ # Publishes to PyPI with trusted publishing (OIDC), no API token.
4
+ # Triggered by pushing a version tag that matches pyproject.toml, e.g. `git tag v0.1.0 && git push origin v0.1.0`.
5
+
6
+ on:
7
+ push:
8
+ tags: ["v*"]
9
+
10
+ jobs:
11
+ build:
12
+ runs-on: ubuntu-latest
13
+ steps:
14
+ - uses: actions/checkout@v7
15
+ - uses: astral-sh/setup-uv@v10.2.0
16
+ - name: Check the tag matches the package version
17
+ run: |
18
+ version="$(uv version --short)"
19
+ if [ "v$version" != "$GITHUB_REF_NAME" ]; then
20
+ echo "Tag $GITHUB_REF_NAME does not match pyproject version $version" >&2
21
+ exit 1
22
+ fi
23
+ - run: uv run pytest -q
24
+ - run: uv build
25
+ - uses: actions/upload-artifact@v7
26
+ with:
27
+ name: dist
28
+ path: dist/
29
+
30
+ publish:
31
+ needs: build
32
+ runs-on: ubuntu-latest
33
+ environment:
34
+ name: pypi
35
+ url: https://pypi.org/p/stackdoctor
36
+ permissions:
37
+ id-token: write # required for trusted publishing
38
+ steps:
39
+ - uses: actions/download-artifact@v8
40
+ with:
41
+ name: dist
42
+ path: dist/
43
+ - uses: pypa/gh-action-pypi-publish@release/v1
@@ -0,0 +1,23 @@
1
+ name: tests
2
+
3
+ on:
4
+ push:
5
+ pull_request:
6
+ workflow_dispatch:
7
+
8
+ jobs:
9
+ unit:
10
+ name: ${{ matrix.os }} / py${{ matrix.python }}
11
+ runs-on: ${{ matrix.os }}
12
+ strategy:
13
+ fail-fast: false
14
+ matrix:
15
+ os: [ubuntu-latest, macos-latest, windows-latest]
16
+ python: ["3.11", "3.12", "3.13"]
17
+ steps:
18
+ - uses: actions/checkout@v7
19
+ - uses: astral-sh/setup-uv@v10.2.0
20
+ with:
21
+ python-version: ${{ matrix.python }}
22
+ - name: Run unit tests
23
+ run: uv run --python ${{ matrix.python }} pytest -q
@@ -0,0 +1,8 @@
1
+ .venv/
2
+ __pycache__/
3
+ *.pyc
4
+ .env
5
+ dist/
6
+ .pytest_cache/
7
+ demo/.env
8
+ demo/e2e-output/
@@ -0,0 +1,161 @@
1
+ # Demo without Docker: "why are my jobs stuck?"
2
+
3
+ This reproduces the *worker stopped* scenario on a Mac with Homebrew Postgres and Redis, with no Docker
4
+ needed. It was tested on macOS 12 with `postgresql@15` and `redis`. It takes about 5 minutes, and the
5
+ last step cleans everything up.
6
+
7
+ You'll use **three terminals**, all in the repo folder:
8
+
9
+ | Terminal | Runs |
10
+ |---|---|
11
+ | **A** | setup, breaking things, checking the result |
12
+ | **B** | the Celery worker (keep it visible while recording) |
13
+ | **C** | Claude Code (or use Claude Desktop) |
14
+
15
+ ## 1. Setup (terminal A)
16
+
17
+ ```sh
18
+ cd ~/Documents/"stack doctor"
19
+ source "$HOME/.local/bin/env" # puts uv on PATH
20
+ uv sync
21
+
22
+ export PATH="$(brew --prefix postgresql@15)/bin:$PATH"
23
+ export LANG=en_US.UTF-8 LC_ALL=en_US.UTF-8 # initdb needs a UTF-8 locale
24
+ export SD=/tmp/sd-demo
25
+ rm -rf "$SD" && mkdir -p "$SD"
26
+ ```
27
+
28
+ Start Postgres on port 55432 and load the demo schema, 1M orders. Loading takes about 15 seconds.
29
+
30
+ ```sh
31
+ initdb -D "$SD/pg" -U shop --auth=trust >/dev/null
32
+ pg_ctl -D "$SD/pg" -l "$SD/postgres.log" -o "-p 55432 -c unix_socket_directories='' \
33
+ -c shared_preload_libraries=pg_stat_statements -c log_lock_waits=on \
34
+ -c deadlock_timeout=1s -c log_line_prefix='%m [%p] '" start
35
+ createdb -h localhost -p 55432 -U shop shop
36
+ psql -q -h localhost -p 55432 -U shop -d shop -f demo/postgres/init.sql
37
+ ```
38
+
39
+ Start Redis on port 56379:
40
+
41
+ ```sh
42
+ redis-server --port 56379 --maxmemory 64mb --maxmemory-policy noeviction \
43
+ --daemonize yes --dir "$SD" --logfile "$SD/redis.log"
44
+ ```
45
+
46
+ Write the stackdoctor config. It connects with the read-only role that `init.sql` created:
47
+
48
+ ```sh
49
+ cat > "$SD/stackdoctor.env" <<EOF
50
+ DATABASE_URL=postgresql://stackdoctor_ro:readonly@localhost:55432/shop
51
+ REDIS_URL=redis://localhost:56379/0
52
+ CELERY_BROKER_URL=redis://localhost:56379/0
53
+ CELERY_RESULT_BACKEND=redis://localhost:56379/1
54
+ LOG_SOURCES=$SD/worker.log,$SD/postgres.log
55
+ EXPECTED_WORKERS=1
56
+ QUEUE_THRESHOLD=20
57
+ EOF
58
+ ```
59
+
60
+ ## 2. Start the worker (terminal B)
61
+
62
+ ```sh
63
+ cd ~/Documents/"stack doctor"/demo/app
64
+ export SD=/tmp/sd-demo
65
+ ../../.venv/bin/celery -A tasks worker --pool threads --concurrency 2 -n worker1@%h \
66
+ --loglevel INFO --pidfile "$SD/worker.pid" 2>&1 | tee -a "$SD/worker.log"
67
+ ```
68
+
69
+ Two details matter here:
70
+
71
+ - **`tee` instead of `--logfile`.** Celery prints `worker: Warm shutdown` to stdout only, so the log
72
+ file has to capture stdout for stackdoctor to see when the worker stopped.
73
+ - **`--pool threads`.** Celery's default prefork pool is unreliable on macOS with Python 3.13. The
74
+ Docker demo uses prefork on Linux.
75
+ - **Never stop it with Ctrl+C. Use `kill -TERM` from terminal A** (step 4). Ctrl+C sends SIGINT to the
76
+ whole foreground pipeline, so `tee` dies together with Celery. A moment later Celery prints
77
+ `worker: Warm shutdown` into a pipe nobody is reading, and the line never reaches `worker.log`.
78
+ `kill -TERM` signals only the Celery process, so `tee` stays alive and writes the line to the file.
79
+
80
+ Back in **terminal A**, check that everything is healthy:
81
+
82
+ ```sh
83
+ (cd demo/app && ../../.venv/bin/python enqueue.py 6) # B shows 6 tasks succeed
84
+ ```
85
+
86
+ ## 3. Connect Claude (terminal C)
87
+
88
+ Use the repo's own venv, so you don't need PyPI and avoid the macOS 12 `uvx`/`realpath` issue:
89
+
90
+ ```sh
91
+ cd ~/Documents/"stack doctor"
92
+ claude mcp add stackdoctor -e STACKDOCTOR_ENV_FILE=/tmp/sd-demo/stackdoctor.env \
93
+ -- "$PWD/.venv/bin/stackdoctor"
94
+ claude
95
+ ```
96
+
97
+ If you're recording in **Claude Desktop** instead, add this to
98
+ `~/Library/Application Support/Claude/claude_desktop_config.json` and restart the app:
99
+
100
+ ```json
101
+ {
102
+ "mcpServers": {
103
+ "stackdoctor": {
104
+ "command": "/Users/leprismacbookpro/Documents/stack doctor/.venv/bin/stackdoctor",
105
+ "env": { "STACKDOCTOR_ENV_FILE": "/tmp/sd-demo/stackdoctor.env" }
106
+ }
107
+ }
108
+ }
109
+ ```
110
+
111
+ ## 4. Break it (terminal A)
112
+
113
+ ```sh
114
+ kill -TERM $(cat "$SD/worker.pid") # stop the worker (not Ctrl+C in B, see step 2)
115
+ while [ -f "$SD/worker.pid" ]; do sleep 1; done # wait until the worker has fully exited
116
+ grep -i "warm shutdown" "$SD/worker.log" # must print: worker: Warm shutdown (MainProcess)
117
+ (cd demo/app && ../../.venv/bin/python enqueue.py 40) # these 40 tasks have nobody to run them
118
+ ```
119
+
120
+ If the `grep` prints nothing, the worker was stopped some other way (for example Ctrl+C) and
121
+ stackdoctor can't see the shutdown. Restart the worker (step 2) and stop it again with `kill -TERM`.
122
+
123
+ ## 5. Ask (terminal C)
124
+
125
+ > why are my jobs stuck?
126
+
127
+ Claude calls `diagnose("why are my jobs stuck?")`. Expect roughly:
128
+
129
+ - **findings:**
130
+ - `no_workers` (critical): No Celery workers replied to ping
131
+ - `queue_no_consumer` (critical): Queue 'celery' has 40 messages and no live worker consumes it
132
+ - `queue_backlog`: 40 waiting messages (threshold 20)
133
+ - `worker_shutdown` from `worker.log`: `worker: Warm shutdown (MainProcess)`
134
+ - **possible cause** (confidence: medium): "Celery worker went down → queue is not being consumed". The cause is the
135
+ `Warm shutdown` line at the moment you ran `kill`, and the effects are the backlog and the
136
+ unconsumed queue, each with a timestamp.
137
+
138
+ To check the same output without an AI (for example, before you hit record):
139
+
140
+ ```sh
141
+ STACKDOCTOR_ENV_FILE="$SD/stackdoctor.env" uv run python -c "
142
+ import asyncio, json; from stackdoctor.server import diagnose
143
+ out = asyncio.run(diagnose('why are my jobs stuck?'))
144
+ print(json.dumps({k: out[k] for k in ('findings', 'possible_causes')}, indent=2))"
145
+ ```
146
+
147
+ ## 6. Recover, then clean up
148
+
149
+ Restart the worker in **terminal B** with the same command as step 2. It drains the 40 tasks, and
150
+ asking again shows a healthy stack.
151
+
152
+ When you're done, run this in terminal A. The first two lines stop the worker the same way as step 4:
153
+
154
+ ```sh
155
+ kill -TERM $(cat "$SD/worker.pid")
156
+ while [ -f "$SD/worker.pid" ]; do sleep 1; done
157
+ claude mcp remove stackdoctor
158
+ redis-cli -p 56379 shutdown nosave
159
+ pg_ctl -D "$SD/pg" stop -m fast
160
+ rm -rf "$SD"
161
+ ```
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 stackdoctor contributors
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,341 @@
1
+ Metadata-Version: 2.5
2
+ Name: stackdoctor
3
+ Version: 0.1.0
4
+ Summary: Read-only MCP server that diagnoses a Python backend stack (Postgres, Celery, Redis, logs) with cross-system correlation.
5
+ Project-URL: Homepage, https://github.com/lepri89/stackdoctor
6
+ Project-URL: Issues, https://github.com/lepri89/stackdoctor/issues
7
+ License: MIT
8
+ License-File: LICENSE
9
+ Keywords: celery,diagnostics,mcp,observability,postgres,redis
10
+ Requires-Python: >=3.11
11
+ Requires-Dist: celery>=5.3
12
+ Requires-Dist: mcp<3,>=2.3
13
+ Requires-Dist: psycopg[binary]>=3.1
14
+ Requires-Dist: python-dotenv>=1.0
15
+ Requires-Dist: redis>=5.0
16
+ Requires-Dist: sqlglot>=25
17
+ Description-Content-Type: text/markdown
18
+
19
+ # stackdoctor
20
+
21
+ [![tests](https://github.com/lepri89/stackdoctor/actions/workflows/tests.yml/badge.svg)](https://github.com/lepri89/stackdoctor/actions/workflows/tests.yml)
22
+ [![demo](https://github.com/lepri89/stackdoctor/actions/workflows/demo.yml/badge.svg)](https://github.com/lepri89/stackdoctor/actions/workflows/demo.yml)
23
+ ![python](https://img.shields.io/badge/python-3.11%20%7C%203.12%20%7C%203.13-blue)
24
+ ![license](https://img.shields.io/badge/license-MIT-green)
25
+
26
+ ![Claude diagnosing stuck Celery jobs with stackdoctor](docs/images/diagnosis.png)
27
+
28
+ *Claude finds that the worker shut down and 40 jobs are waiting, in one diagnose() call.*
29
+
30
+ > **Strictly read-only.** stackdoctor never writes, deletes, restarts, retries, revokes or sends anything.
31
+ > Postgres sessions are forced read-only by the server, Redis commands go through an allowlist,
32
+ > and Celery is only *inspected*. Secrets are redacted from every response.
33
+
34
+ stackdoctor is an MCP server that lets Claude, Cursor or any MCP client diagnose a Python backend stack
35
+ (**Postgres, Celery, Redis and logs**) from a single tool call.
36
+
37
+ Ask *"why are my jobs stuck?"* and the assistant calls `diagnose(symptom)`. It runs every relevant check
38
+ in parallel and returns one timestamped snapshot with:
39
+
40
+ - **findings**: blocked queries, missing workers, queue backlogs, log error spikes, Redis memory pressure, …
41
+ - **a merged timeline** of events from Postgres, Celery, Redis and your logs
42
+ - **possible cause → effect chains** when events line up in time, each with its evidence timestamps
43
+
44
+ ```text
45
+ possible cause (confidence: medium): Database lock → blocked queries → stuck or failing tasks
46
+ cause: 21:58:01 postgres.lock_held pid 7959 holds a lock blocking 2 sessions: LOCK TABLE orders …
47
+ effect: 21:58:02 logs.db_lock_wait [postgres] process 7962 still waiting for RowExclusiveLock …
48
+ effect: 21:58:21 celery.task_failed tasks.process_order[…] failed: LockNotAvailable: lock timeout
49
+ effect: 21:58:21 celery.task_long_running tasks.process_order[…] running 13.2s on worker1
50
+ effect: 21:58:34 celery.workers_saturated worker1: all 2 slots busy, 2 reserved
51
+ caveat: Inferred from timing only. Verify before acting; this is not a confirmed root cause.
52
+ ```
53
+
54
+ ## Install
55
+
56
+ You need [uv](https://docs.astral.sh/uv/getting-started/installation/):
57
+
58
+ ```sh
59
+ curl -LsSf https://astral.sh/uv/install.sh | sh # macOS / Linux
60
+ powershell -c "irm https://astral.sh/uv/install.ps1 | iex" # Windows
61
+ ```
62
+
63
+ Then one command runs the server, with no other install step:
64
+
65
+ ```sh
66
+ uvx stackdoctor
67
+ ```
68
+
69
+ Until the package is on PyPI, run it from a checkout with `uvx --from /path/to/stackdoctor stackdoctor`,
70
+ or from git with `uvx --from git+https://github.com/lepri89/stackdoctor stackdoctor`.
71
+
72
+ > **macOS 12 (Monterey):** `uvx stackdoctor` fails with `realpath: command not found`, because uv's
73
+ > launcher script needs `realpath`, which only ships with macOS 13+. Use
74
+ > `uvx --from stackdoctor python -m stackdoctor` instead. In client configs that means
75
+ > `"args": ["--from", "stackdoctor", "python", "-m", "stackdoctor"]`. Alternatively, run
76
+ > `uv tool install stackdoctor` once and use `stackdoctor` as the command.
77
+
78
+ ## Configure
79
+
80
+ Set environment variables in your MCP client config, or in a `.env` file. stackdoctor looks for `.env` in
81
+ the working directory, or uses the file named by `STACKDOCTOR_ENV_FILE`.
82
+ **Any source you don't configure is skipped**, so it's fine to start with only `DATABASE_URL`.
83
+
84
+ | Variable | Example | Used for |
85
+ |---|---|---|
86
+ | `DATABASE_URL` | `postgresql://stackdoctor_ro:…@localhost:5432/app` | Postgres checks |
87
+ | `REDIS_URL` | `redis://localhost:6379/0` | Redis checks |
88
+ | `CELERY_BROKER_URL` | `redis://localhost:6379/0` (RabbitMQ `amqp://…` is experimental) | Celery inspect + queue lengths |
89
+ | `CELERY_RESULT_BACKEND` | `redis://localhost:6379/1` | failed tasks / task details |
90
+ | `CELERY_APP` | `myproject.celery:app` | optional: use your app's queues/routes/config (run from your project dir) |
91
+ | `CELERY_QUEUES` | `default,emails` | extra queue names to measure |
92
+ | `LOG_SOURCES` | `./logs/worker.log,docker:api,docker:worker` | log files and/or docker containers |
93
+
94
+ `LOG_SOURCES` entries are file paths, `docker:<container>`, or a bare container name.
95
+ Only configured sources can be read.
96
+
97
+ > **Celery workers: point `LOG_SOURCES` at the worker's stdout.** Celery prints
98
+ > `worker: Warm shutdown (MainProcess)` straight to stdout, not through logging, so it never reaches a
99
+ > `--logfile`. Docker containers already capture stdout. For a file, redirect stdout to it
100
+ > (`celery … worker >> worker.log 2>&1`, or supervisor/systemd stdout capture) so stackdoctor can see
101
+ > when a worker stopped.
102
+
103
+ <details>
104
+ <summary>Thresholds and limits</summary>
105
+
106
+ | Variable | Default | Meaning |
107
+ |---|---|---|
108
+ | `QUEUE_THRESHOLD` | 100 | messages waiting before a queue counts as a backlog |
109
+ | `EXPECTED_WORKERS` | 0 (off) | report missing workers if fewer reply |
110
+ | `LONG_TASK_S` / `LONG_QUERY_S` | 60 / 30 | when a task / query counts as long-running |
111
+ | `REDIS_MEM_WARN_PCT` | 85 | % of `maxmemory` that counts as "near the limit" |
112
+ | `LOG_ERROR_SPIKE` | 10 | problem lines in 5 minutes that count as a spike |
113
+ | `CHAIN_WINDOW_MIN` | 30 | how far apart cause and effect may be |
114
+ | `CHECK_TIMEOUT_S` | 8 | per-check timeout inside `diagnose` |
115
+ | `CELERY_INSPECT_TIMEOUT_S` | 1.0 | how long to wait for worker replies |
116
+ | `MAX_OUTPUT_CHARS` | 20000 | per-tool response cap (`diagnose` gets 2×) |
117
+
118
+ </details>
119
+
120
+ ### Claude Desktop
121
+
122
+ `~/Library/Application Support/Claude/claude_desktop_config.json` (macOS) or
123
+ `%APPDATA%\Claude\claude_desktop_config.json` (Windows):
124
+
125
+ ```json
126
+ {
127
+ "mcpServers": {
128
+ "stackdoctor": {
129
+ "command": "uvx",
130
+ "args": ["stackdoctor"],
131
+ "env": {
132
+ "DATABASE_URL": "postgresql://stackdoctor_ro:password@localhost:5432/app",
133
+ "REDIS_URL": "redis://localhost:6379/0",
134
+ "CELERY_BROKER_URL": "redis://localhost:6379/0",
135
+ "CELERY_RESULT_BACKEND": "redis://localhost:6379/1",
136
+ "LOG_SOURCES": "docker:api,docker:worker"
137
+ }
138
+ }
139
+ }
140
+ }
141
+ ```
142
+
143
+ GUI apps often don't see your shell `PATH`. If Claude Desktop can't find `uvx`, use the full path:
144
+ `/Users/<username>/.local/bin/uvx` on macOS, or `C:\\Users\\<username>\\.local\\bin\\uvx.exe` on Windows.
145
+
146
+ ### Claude Code
147
+
148
+ ```sh
149
+ claude mcp add stackdoctor \
150
+ -e DATABASE_URL=postgresql://stackdoctor_ro:password@localhost:5432/app \
151
+ -e REDIS_URL=redis://localhost:6379/0 \
152
+ -e CELERY_BROKER_URL=redis://localhost:6379/0 \
153
+ -e LOG_SOURCES=docker:api,docker:worker \
154
+ -- uvx stackdoctor
155
+ ```
156
+
157
+ Or, from a project directory with a `.env` file, simply `claude mcp add stackdoctor -- uvx stackdoctor`.
158
+
159
+ ### Cursor
160
+
161
+ `.cursor/mcp.json` in your project, or `~/.cursor/mcp.json` globally. This uses the same
162
+ `mcpServers` format as Claude Desktop:
163
+
164
+ ```json
165
+ {
166
+ "mcpServers": {
167
+ "stackdoctor": {
168
+ "command": "uvx",
169
+ "args": ["stackdoctor"],
170
+ "env": { "STACKDOCTOR_ENV_FILE": "${workspaceFolder}/.env" }
171
+ }
172
+ }
173
+ }
174
+ ```
175
+
176
+ ### VS Code (Copilot agent mode)
177
+
178
+ `.vscode/mcp.json`:
179
+
180
+ ```json
181
+ {
182
+ "servers": {
183
+ "stackdoctor": {
184
+ "type": "stdio",
185
+ "command": "uvx",
186
+ "args": ["stackdoctor"],
187
+ "env": { "STACKDOCTOR_ENV_FILE": "${workspaceFolder}/.env" }
188
+ }
189
+ }
190
+ }
191
+ ```
192
+
193
+ ## Tools
194
+
195
+ | Tool | What it returns |
196
+ |---|---|
197
+ | `diagnose(symptom)` | **Start here.** Runs the relevant checks concurrently, each with its own timeout, and returns findings, a merged timeline and possible causes |
198
+ | `active_queries(min_duration_s)` | non-idle sessions, incl. *idle in transaction* |
199
+ | `blocking_locks()` | waiting sessions and the sessions blocking them |
200
+ | `slow_queries(limit)` | top statements from `pg_stat_statements` (explains how to enable it if missing) and seq-scan-heavy tables |
201
+ | `run_select(sql, limit)` | one validated `SELECT` / `WITH` / `EXPLAIN` with a row limit |
202
+ | `workers()` | live workers, their queues, concurrency, active/reserved tasks |
203
+ | `queue_lengths(queues)` | messages waiting per queue, read from the broker (Redis incl. priority queues; RabbitMQ experimental) |
204
+ | `failed_tasks(limit)` | recent failures from a Redis result backend, or *why* they aren't visible |
205
+ | `task_details(task_id)` | stored result/traceback, and whether a worker holds the task right now |
206
+ | `memory_and_clients()` | Redis memory vs `maxmemory`, evictions, clients, slowlog |
207
+ | `scan_keys(pattern, limit)` | key names via `SCAN` |
208
+ | `key_info(key)` | type, TTL, size, encoding, short redacted preview |
209
+ | `tail_logs(source, lines)` | last lines of a configured log source |
210
+ | `search_logs(pattern, since_minutes, source)` | regex search over recent log lines |
211
+
212
+ Postgres checks are intentionally light. For deep Postgres health (index advice, vacuum, bloat), run
213
+ [Postgres MCP Pro](https://github.com/crystaldba/postgres-mcp) next to stackdoctor.
214
+
215
+ ## Safety
216
+
217
+ stackdoctor is built so that a confused or prompt-injected assistant still can't change your systems.
218
+
219
+ **Postgres**
220
+ - Every session is opened with `default_transaction_read_only=on`, `statement_timeout=5s` and
221
+ `lock_timeout=2s`, passed as connection options so they override your URL. stackdoctor also checks
222
+ the setting and refuses to run if the session isn't read-only.
223
+ - `run_select` parses SQL with [sqlglot](https://github.com/tobymao/sqlglot), not regexes. It accepts
224
+ exactly one `SELECT` / `WITH … SELECT` / `EXPLAIN` statement and rejects:
225
+ - writable CTEs, `SELECT … INTO` and `FOR UPDATE/SHARE`
226
+ - `EXPLAIN ANALYZE`, because it executes the query
227
+ - functions with side effects, even inside a read-only transaction: `pg_terminate_backend`,
228
+ `pg_cancel_backend`, `pg_reload_conf`, `pg_read_file`, `pg_read_binary_file`, `pg_ls_dir`, `lo_*`,
229
+ `dblink*`, `set_config`, `pg_advisory_*`, `query_to_xml` (which runs SQL from a string), `pg_sleep`, …
230
+ - Results are always limited: queries are wrapped as `SELECT * FROM (<q>) AS sd_sub LIMIT n`
231
+ (default 100, max 1000).
232
+
233
+ **Recommended: a dedicated read-only role.** Defense in depth, so the database enforces the rules too:
234
+
235
+ ```sql
236
+ CREATE ROLE stackdoctor_ro LOGIN PASSWORD 'change-me';
237
+ GRANT CONNECT ON DATABASE app TO stackdoctor_ro;
238
+ GRANT USAGE ON SCHEMA public TO stackdoctor_ro;
239
+ GRANT SELECT ON ALL TABLES IN SCHEMA public TO stackdoctor_ro;
240
+ ALTER DEFAULT PRIVILEGES IN SCHEMA public GRANT SELECT ON TABLES TO stackdoctor_ro;
241
+ GRANT pg_monitor TO stackdoctor_ro; -- see other sessions' queries and pg_stat_statements
242
+ ALTER ROLE stackdoctor_ro SET default_transaction_read_only = on;
243
+ ```
244
+
245
+ Anything the role can `SELECT`, the assistant can read. If some tables hold data you don't want to
246
+ share with an AI, leave them out of the `GRANT SELECT`.
247
+
248
+ **Redis.** Every command is checked against an allowlist that is aware of subcommands (`INFO`, `SCAN`,
249
+ `TYPE`, `PTTL`, `LLEN`, `GET`, …). `CLIENT LIST` is allowed; `CLIENT KILL` / `PAUSE` are not. The same goes
250
+ for `OBJECT`, `MEMORY` and `CONFIG` (only `CONFIG GET`). `KEYS` is never used; listing always goes through `SCAN`.
251
+
252
+ **Celery.** Only `inspect` (`ping`, `active`, `reserved`, `active_queues`, `stats`, `query_task`) and
253
+ broker reads are used: `LLEN` on Redis, and passive `queue_declare` on RabbitMQ.
254
+
255
+ > **RabbitMQ support is experimental.** Queue lengths via passive `queue_declare` and the `inspect`
256
+ > calls are implemented but not yet covered by CI. The Redis broker is the tested path. Reports welcome. stackdoctor never revokes,
257
+ retries, shuts down or sends tasks.
258
+ *To be precise:* `inspect` works by publishing a broadcast message on Celery's **control channel**
259
+ (`celery.pidbox`) and reading replies from a temporary reply queue. That is how Celery's own
260
+ `celery inspect` command works. It doesn't touch your task queues or results, but it isn't zero
261
+ traffic. Everything else stackdoctor does is pure reads.
262
+
263
+ **Redaction.** Every response is redacted before it leaves the server. This covers:
264
+ - passwords in URLs (`postgres://user:***@host`)
265
+ - `Authorization` headers and bearer tokens
266
+ - `PASSWORD=`, `SECRET_KEY=`, `api_key:` and other secret-looking keys
267
+ - AWS, GitHub, Slack, Stripe, OpenAI/Anthropic-style and Google keys, JWTs, and private keys
268
+
269
+ Task args/kwargs and Redis value previews are truncated and redacted too. Redaction is pattern-based,
270
+ so treat it as a safety net, not a guarantee.
271
+
272
+ **Output caps.** Every tool caps its response (`MAX_OUTPUT_CHARS`) by trimming lists and long strings,
273
+ so a huge table or log can't flood the assistant's context.
274
+
275
+ **Logs.** Only sources listed in `LOG_SOURCES` can be read, so the assistant can't ask for `/etc/passwd`.
276
+ Docker logs are read with `docker logs`, called without a shell.
277
+
278
+ ## Try the demo
279
+
280
+ The demo is a small FastAPI + Celery + Redis + Postgres app with scripts that break it in realistic ways.
281
+ CI runs every scenario on each push ([`demo.yml`](.github/workflows/demo.yml)): it breaks the stack,
282
+ calls `diagnose()` through a real MCP stdio client and checks the findings and chains.
283
+ To try it without Docker (local Postgres, Redis and worker), see [DEMO.md](DEMO.md).
284
+
285
+ ```sh
286
+ cd demo
287
+ docker compose up -d --build # first run builds the app and loads 1M rows (~1 min)
288
+ cp .env.example .env # stackdoctor config for the demo (uses the read-only role)
289
+ ```
290
+
291
+ Point your MCP client at the demo by setting `STACKDOCTOR_ENV_FILE` to the absolute path of `demo/.env`.
292
+ Then break something and ask:
293
+
294
+ | Script (macOS/Linux · Windows) | What it does | Ask |
295
+ |---|---|---|
296
+ | `./break_worker.sh` · `.\break_worker.ps1` | stops the Celery worker and enqueues 50 tasks | "why are my jobs stuck?" |
297
+ | `./lock_table.sh` · `.\lock_table.ps1` | holds an `ACCESS EXCLUSIVE` lock on `orders` for 3 min, then enqueues tasks that block on it | "why are my jobs stuck?" |
298
+ | `./slow_query.sh` · `.\slow_query.ps1` | runs three ~1-minute sequential scans | "why is the API slow?" |
299
+ | `./reset.sh` · `.\reset.ps1` | restarts the worker and ends the demo's locks and slow queries | |
300
+
301
+ **Colima (macOS).** `docker compose` works unchanged once Colima is running:
302
+ `colima start --cpu 2 --memory 4`. On macOS 13+ you can add `--vm-type vz`. On macOS 12 Colima uses
303
+ QEMU, which must be installed first (`brew install qemu`). **Windows:** use Docker Desktop or Docker
304
+ in WSL2 and run the `.ps1` scripts from PowerShell.
305
+
306
+ ## Why stackdoctor instead of separate Postgres, Celery and log MCPs?
307
+
308
+ | | Separate MCP servers | stackdoctor |
309
+ |---|---|---|
310
+ | Install | one server per system, each with its own config | one `uvx stackdoctor` |
311
+ | Celery | usually needs Flower running | inspect API + broker, no Flower |
312
+ | Answering "why is it stuck?" | the AI calls 5–10 tools one after another, each a different moment in time | one `diagnose()` call: all checks run in parallel and share one timestamp |
313
+ | Correlation | left to the AI, across separate outputs | merged timeline + timing-based cause → effect hypotheses, with evidence |
314
+ | Safety | varies per server | one read-only policy for every system, redaction and output caps everywhere |
315
+ | Postgres depth | Postgres MCP Pro is deeper | intentionally light; use both |
316
+
317
+ ## Development
318
+
319
+ ```sh
320
+ uv sync
321
+ uv run pytest # unit tests (safety layer, diagnose, logs)
322
+ STACKDOCTOR_TEST_DATABASE_URL=postgresql://shop:shop@localhost:55432/shop \
323
+ STACKDOCTOR_TEST_REDIS_URL=redis://localhost:56379/0 \
324
+ uv run pytest # plus live tests against the demo
325
+ ```
326
+
327
+ ## Releasing
328
+
329
+ Releases go to PyPI from GitHub Actions with
330
+ [trusted publishing](https://docs.pypi.org/trusted-publishers/), so no API token is stored anywhere.
331
+
332
+ 1. One-time: on PyPI, add a *pending publisher* (Account → Publishing) with project `stackdoctor`, owner
333
+ `lepri89`, repository `stackdoctor`, workflow `release.yml` and environment `pypi`. In GitHub, create
334
+ an environment named `pypi` (Settings → Environments); adding yourself as a required reviewer is a good idea.
335
+ 2. Bump `version` in `pyproject.toml`, commit, then tag and push:
336
+ `git tag v0.1.0 && git push origin v0.1.0`.
337
+ The workflow checks that the tag matches the version, runs the tests, builds, and publishes.
338
+
339
+ ## License
340
+
341
+ MIT