mcpwatchman 0.0.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (31) hide show
  1. mcpwatchman-0.0.2/.github/dependabot.yml +8 -0
  2. mcpwatchman-0.0.2/.github/workflows/ci.yml +35 -0
  3. mcpwatchman-0.0.2/.github/workflows/release.yml +68 -0
  4. mcpwatchman-0.0.2/.gitignore +51 -0
  5. mcpwatchman-0.0.2/LICENSE +21 -0
  6. mcpwatchman-0.0.2/PKG-INFO +183 -0
  7. mcpwatchman-0.0.2/README.md +149 -0
  8. mcpwatchman-0.0.2/docker/Dockerfile +15 -0
  9. mcpwatchman-0.0.2/docs/.gitkeep +0 -0
  10. mcpwatchman-0.0.2/evals/gold-set/.gitkeep +0 -0
  11. mcpwatchman-0.0.2/pyproject.toml +73 -0
  12. mcpwatchman-0.0.2/rules/README.md +23 -0
  13. mcpwatchman-0.0.2/rules/_meta/.gitkeep +0 -0
  14. mcpwatchman-0.0.2/rules/go/.gitkeep +0 -0
  15. mcpwatchman-0.0.2/rules/javascript/.gitkeep +0 -0
  16. mcpwatchman-0.0.2/rules/python/.gitkeep +0 -0
  17. mcpwatchman-0.0.2/site/README.md +8 -0
  18. mcpwatchman-0.0.2/src/mcpwatchman/__init__.py +3 -0
  19. mcpwatchman-0.0.2/src/mcpwatchman/api/__init__.py +1 -0
  20. mcpwatchman-0.0.2/src/mcpwatchman/api/main.py +18 -0
  21. mcpwatchman-0.0.2/src/mcpwatchman/cli/__init__.py +1 -0
  22. mcpwatchman-0.0.2/src/mcpwatchman/cli/main.py +28 -0
  23. mcpwatchman-0.0.2/src/mcpwatchman/client/__init__.py +1 -0
  24. mcpwatchman-0.0.2/src/mcpwatchman/db/__init__.py +1 -0
  25. mcpwatchman-0.0.2/src/mcpwatchman/db/models.py +14 -0
  26. mcpwatchman-0.0.2/src/mcpwatchman/workers/__init__.py +1 -0
  27. mcpwatchman-0.0.2/src/mcpwatchman/workers/crawler/__init__.py +1 -0
  28. mcpwatchman-0.0.2/src/mcpwatchman/workers/scanner/__init__.py +1 -0
  29. mcpwatchman-0.0.2/src/mcpwatchman/workers/scoring/__init__.py +1 -0
  30. mcpwatchman-0.0.2/src/mcpwatchman/workers/scoring/weights.py +54 -0
  31. mcpwatchman-0.0.2/tests/test_weights.py +53 -0
@@ -0,0 +1,8 @@
1
+ version: 2
2
+ updates:
3
+ # Release actions are SHA-pinned, which is only safe with something moving
4
+ # the pins — an unmaintained pin is a stale dependency that looks deliberate.
5
+ - package-ecosystem: github-actions
6
+ directory: "/"
7
+ schedule:
8
+ interval: weekly
@@ -0,0 +1,35 @@
1
+ name: ci
2
+
3
+ on:
4
+ push:
5
+ branches: [dev, main]
6
+ pull_request:
7
+
8
+ # A lint job reads the checkout and writes nothing; the default grant is wider.
9
+ permissions:
10
+ contents: read
11
+
12
+ jobs:
13
+ lint:
14
+ runs-on: ubuntu-latest
15
+ steps:
16
+ - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
17
+ - uses: actions/setup-python@5fda3b95a4ea91299a34e894583c3862153e4b97 # v7.0.0
18
+ with:
19
+ python-version: "3.12"
20
+ # Via the dev extra so CI and a local `ruff check` share one pinned
21
+ # version — a floating ruff reds CI the day a release adds a rule.
22
+ - run: pip install -e ".[dev]"
23
+ - run: ruff check .
24
+
25
+ # Publishing lives in release.yml, fired by a vX.Y.Z tag.
26
+
27
+ test:
28
+ runs-on: ubuntu-latest
29
+ steps:
30
+ - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
31
+ - uses: actions/setup-python@5fda3b95a4ea91299a34e894583c3862153e4b97 # v7.0.0
32
+ with:
33
+ python-version: "3.12"
34
+ - run: pip install -e ".[dev]"
35
+ - run: pytest
@@ -0,0 +1,68 @@
1
+ name: release
2
+
3
+ on:
4
+ push:
5
+ tags:
6
+ - 'v*.*.*'
7
+
8
+ concurrency:
9
+ group: release-${{ github.ref }}
10
+ cancel-in-progress: false
11
+
12
+ # Split deliberately: `pip install` executes arbitrary code from transitive
13
+ # dependencies, so it must not run in a job that can mint a publish credential.
14
+ # The build job holds no id-token; the publish job installs nothing.
15
+ jobs:
16
+ build:
17
+ runs-on: ubuntu-latest
18
+ permissions:
19
+ contents: read
20
+ steps:
21
+ - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
22
+ - uses: actions/setup-python@5fda3b95a4ea91299a34e894583c3862153e4b97 # v7.0.0
23
+ with:
24
+ python-version: "3.12"
25
+ - run: pip install build twine
26
+ - run: python -m build
27
+ # Malformed metadata is rejected at upload — and a rejected upload still
28
+ # burns the version number, which PyPI never lets you reuse.
29
+ - run: twine check dist/*
30
+ # The tag is the only value a human types; everything else is derived from
31
+ # the tree. Three files must already agree (bump-audit checks one of them),
32
+ # and __version__ is what `mcpwatchman --version` reports, so a half-bump
33
+ # would ship a CLI that lies about itself — permanently, at this registry.
34
+ - name: Tag, artifact and source versions agree
35
+ run: |
36
+ set -euo pipefail
37
+ tag="${GITHUB_REF_NAME#v}"
38
+ built=$(python -c "import glob,os; print(os.path.basename(glob.glob('dist/*.whl')[0]).split('-')[1])")
39
+ proj=$(python -c "import tomllib; print(tomllib.load(open('pyproject.toml','rb'))['project']['version'])")
40
+ init=$(python -c "import re; print(re.search(r'__version__ = \"([^\"]+)\"', open('src/mcpwatchman/__init__.py').read()).group(1))")
41
+ echo "tag=$tag wheel=$built pyproject=$proj __version__=$init"
42
+ for v in "$built" "$proj" "$init"; do
43
+ [ "$v" = "$tag" ] || { echo "::error::version mismatch — tag $tag vs $v"; exit 1; }
44
+ done
45
+ - uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1
46
+ with:
47
+ name: dist
48
+ path: dist/
49
+
50
+ publish:
51
+ needs: build
52
+ runs-on: ubuntu-latest
53
+ permissions:
54
+ id-token: write
55
+ # Actions are pinned to commit SHAs, not tags: a tag is mutable, and any
56
+ # code running in this job can request the PyPI credential. Dependabot
57
+ # keeps the pins current (.github/dependabot.yml) so they cannot go stale.
58
+ steps:
59
+ - uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1
60
+ with:
61
+ name: dist
62
+ path: dist/
63
+ - uses: pypa/gh-action-pypi-publish@dc37677b2e1c63e2034f94d8a5b11f265b73ba33 # v1.14.2
64
+ with:
65
+ # Explicit: the action's canonical `packages-dir` declares no default,
66
+ # and `dist` reaches it only via the deprecated `packages_dir` alias
67
+ # that upstream has marked for removal in v3+.
68
+ packages-dir: dist/
@@ -0,0 +1,51 @@
1
+ # ── Internal design docs — NEVER committed (workspace policy) ──
2
+ # Strategy, methodology internals, and project memory stay local-only.
3
+ CLAUDE.md
4
+ STATE.md
5
+ .claude/
6
+ /[0-9][0-9]-*.md
7
+ /adrs/
8
+ SPEC*.md
9
+ /scratch/
10
+
11
+ # ── Python ──
12
+ __pycache__/
13
+ *.py[cod]
14
+ *.egg-info/
15
+ .eggs/
16
+ build/
17
+ dist/
18
+ .venv/
19
+ venv/
20
+ env/
21
+ .pytest_cache/
22
+ .mypy_cache/
23
+ .ruff_cache/
24
+ .coverage
25
+ htmlcov/
26
+ .python-version
27
+
28
+ # ── Node / Astro (site/) ──
29
+ node_modules/
30
+ site/dist/
31
+ site/.astro/
32
+ .astro/
33
+
34
+ # ── Env / secrets ──
35
+ .env
36
+ .env.*
37
+ !.env.example
38
+ *.local
39
+
40
+ # ── OS / editor ──
41
+ .DS_Store
42
+ Thumbs.db
43
+ *.swp
44
+ .idea/
45
+ .vscode/
46
+
47
+ # Wrapper-managed state file: indicates origin push is failing
48
+ ops/autonomous/.push-broken
49
+
50
+ # Run-log scratch file — local-only, like STATE.md
51
+ RUNLOG.md
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 KeMeK Network
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,183 @@
1
+ Metadata-Version: 2.5
2
+ Name: mcpwatchman
3
+ Version: 0.0.2
4
+ Summary: Independent security and quality audit for Model Context Protocol servers.
5
+ Project-URL: Homepage, https://mcpwatchman.com
6
+ Project-URL: Repository, https://github.com/kVadrum/mcpwatchman
7
+ Author: kVadrum
8
+ License-Expression: MIT
9
+ License-File: LICENSE
10
+ Keywords: audit,mcp,model-context-protocol,security,static-analysis,supply-chain
11
+ Classifier: Development Status :: 2 - Pre-Alpha
12
+ Classifier: Intended Audience :: Developers
13
+ Classifier: Programming Language :: Python :: 3.12
14
+ Classifier: Topic :: Security
15
+ Requires-Python: >=3.12
16
+ Requires-Dist: click>=8.1
17
+ Requires-Dist: httpx>=0.27
18
+ Requires-Dist: pydantic-settings>=2.3
19
+ Requires-Dist: pydantic>=2.7
20
+ Requires-Dist: rich>=13.7
21
+ Provides-Extra: api
22
+ Requires-Dist: fastapi>=0.111; extra == 'api'
23
+ Requires-Dist: uvicorn[standard]>=0.30; extra == 'api'
24
+ Provides-Extra: dev
25
+ Requires-Dist: mypy>=1.10; extra == 'dev'
26
+ Requires-Dist: pytest>=8.2; extra == 'dev'
27
+ Requires-Dist: ruff~=0.16.7; extra == 'dev'
28
+ Provides-Extra: workers
29
+ Requires-Dist: alembic>=1.13; extra == 'workers'
30
+ Requires-Dist: detect-secrets>=1.5; extra == 'workers'
31
+ Requires-Dist: semgrep>=1.75; extra == 'workers'
32
+ Requires-Dist: sqlalchemy>=2.0; extra == 'workers'
33
+ Description-Content-Type: text/markdown
34
+
35
+ # mcpwatchman
36
+
37
+ **Independent security and quality audit for [Model Context Protocol](https://modelcontextprotocol.io) servers.**
38
+
39
+ `mcpwatchman` continuously scans every server in the official MCP registry and publishes a transparent, evidence-linked assessment of each one — so you can answer "is this MCP server safe to install?" before you wire it into your agent.
40
+
41
+ > **Status: early — building in public.**
42
+ > The methodology and architecture are settled and the repository is scaffolded; the scanner, scoring engine, site, and CLI are under active construction.
43
+ > **Nothing here is production-ready, and no scores have been published yet.**
44
+ > Watch the repo to follow along.
45
+
46
+ ---
47
+
48
+ ## Why
49
+
50
+ MCP adoption is accelerating across Claude Code, Cursor, Continue, Zed, Goose and more — and people are installing servers with the same blind trust they once gave `curl | bash`.
51
+ Research on thousands of public MCP servers has found widespread server-side request forgery, unsafe command execution, and servers exposed over HTTP with no authentication at all.
52
+
53
+ The official registry is **metadata-only by design** — it lists what exists, not what's safe.
54
+ `mcpwatchman` is the independent safety layer on top of it.
55
+ The closest analogues are [Mozilla Observatory](https://observatory.mozilla.org/) and [OpenSSF Scorecard](https://scorecard.dev/): independent, transparent, free, and trusted precisely because they aren't selling anything to the projects they score.
56
+
57
+ ## How it works
58
+
59
+ A daily pass over the official registry, then per server:
60
+
61
+ 1. **Resolve** the declared source repository and package artifacts from registry metadata.
62
+ 2. **Fetch** published source and lockfiles — a shallow, sparse clone at a pinned commit, into a per-job sandboxed workspace with resource limits and no inbound network.
63
+ 3. **Analyze statically** — an MCP-specific [semgrep](https://semgrep.dev) ruleset for the patterns that matter in this ecosystem (tool-handler shell exec, unvalidated URL fetches, unsafe deserialization, path traversal), plus dependency CVE scanning, transport and auth inspection, and repository maintenance signals.
64
+ 4. **Score** each axis from the findings, with every point of the score traceable to the artifact that produced it.
65
+ 5. **Publish** to the site, JSON API, RSS feeds, and CLI.
66
+
67
+ **The analysis is static, always.**
68
+ We read published source, manifests and lockfiles.
69
+ **We never execute the code we scan**, and we never connect to a running server instance.
70
+
71
+ **So here is what a good score cannot tell you.**
72
+ Static analysis reads what a server *publishes*, so the blind spot is whatever that source does not determine: behavior gated on remote configuration, code fetched or generated at runtime, or a deployed server that differs from the repository it declares.
73
+ An unsafe pattern that merely *executes* at tool-invocation time is still detectable — catching those is the ruleset's core job — but a score here is evidence about published source, never a runtime guarantee.
74
+ Running every server we scan is a substantially larger sandboxing problem than reading it, and we would rather be narrow and honest about it than broad and quietly wrong.
75
+
76
+ ## How it scores
77
+
78
+ Every server is rated 0–100 on five axes, and **every score links to its evidence** — a file and line, a CVE identifier, a commit date.
79
+ No mystery numbers.
80
+
81
+ | Axis | Default weight | Captures |
82
+ |---|---|---|
83
+ | **Code Safety** | 30% | Static-analysis findings: shell exec, SSRF, deserialization, path traversal |
84
+ | **Auth Posture** | 20% | Authentication model, transport security, secret handling, scope granularity |
85
+ | **Dependency Health** | 20% | Known CVEs in direct and transitive dependencies |
86
+ | **Maintenance** | 15% | Activity, release cadence, issue responsiveness, bus factor |
87
+ | **Transparency** | 15% | License validity, documented behavior, declared scopes |
88
+
89
+ **Per-axis scores ship first; the composite waits for calibration.**
90
+ The weights above are *provisional* starting values informed by the threat literature — they live in [`src/mcpwatchman/workers/scoring/weights.py`](src/mcpwatchman/workers/scoring/weights.py) and are versioned like any other code.
91
+ They become final only after calibration against a hand-audited gold set of roughly 30 servers spanning categories and a deliberate range of expected results.
92
+ Until that regression suite is green, public surfaces show the per-axis scores and suppress the composite.
93
+ A single blended number is the easiest thing to publish and the easiest thing to get quietly wrong, so it is the last thing we will ship.
94
+
95
+ The full methodology is published openly — anyone can audit our auditing.
96
+
97
+ ## Principles
98
+
99
+ - **Methodology over marketing.**
100
+ Every check, weight and threshold is documented and versioned.
101
+ - **Evidence or it doesn't ship.**
102
+ A finding with no linked artifact doesn't appear.
103
+ - **Facts, not characterizations.**
104
+ We describe observed code patterns, never intent.
105
+ Any finding can be appealed.
106
+ - **Collaborative, not competitive.**
107
+ We overlay the official registry; we don't replace it.
108
+ A server exists here only if the official registry lists it.
109
+ - **Free, and staying that way.**
110
+ The scanner, ruleset, site, API, RSS feeds and CLI are MIT-licensed and free.
111
+
112
+ ## Disclosure
113
+
114
+ Findings fall into two tracks, and they are handled differently on purpose.
115
+
116
+ **Track A — publicly observable patterns.**
117
+ Static-analysis hits, license and transport facts: anyone can run the same tools against the same published source and see the same thing.
118
+ There is no informational asymmetry to protect, so these are **not embargoed** — they appear on the server's page from the first scan that detects them.
119
+ Maintainer notification before publication is the right courtesy; delaying a fact every reader could derive themselves is not.
120
+
121
+ **Track B — findings that warrant coordination.**
122
+ Default embargo is **14 days** from maintainer contact, shortened to **7** where there is evidence of active exploitation, and extended up to **45 days total** when a maintainer comes back with a concrete fix timeline.
123
+ If we cannot reach a maintainer within 48 hours of the first attempt the clock still starts, and the target becomes **21 days from that first attempt** rather than 14 — being slow to check email is not the same as being unresponsive, and we don't punish it.
124
+
125
+ **Appeals.**
126
+ Anyone can contest a finding — maintainers and third parties alike, per finding, by ID.
127
+ Scores are recomputed, not negotiated: if the evidence is wrong the finding goes, if the code changed a rescan reflects it, and where a finding is technically valid but mitigated by context our rules don't model, it stays on the page with the maintainer's explanation attached.
128
+
129
+ ## Surfaces
130
+
131
+ | Surface | What it is |
132
+ |---|---|
133
+ | **Site** | Per-server pages with the evidence behind every axis, search and filtering, and the full methodology |
134
+ | **JSON API** | The same data, machine-readable — built to be consumed by agents and CI as a first-class audience, not as an afterthought |
135
+ | **RSS** | High-severity findings and score drops, for people who want to watch the ecosystem rather than one server |
136
+ | **Badges** | An auto-updating SVG a maintainer can put in their own README |
137
+ | **CLI** | `pip install mcpwatchman`, then `mcpwatchman check <server>` before you install it — with a `--threshold` exit code for CI |
138
+
139
+ ## Install
140
+
141
+ ```sh
142
+ pip install mcpwatchman # not yet published
143
+ ```
144
+
145
+ Requires Python 3.12+.
146
+
147
+ ## Repository layout
148
+
149
+ ```
150
+ rules/ MCP-specific semgrep ruleset, by language and category
151
+ src/mcpwatchman/
152
+ cli/ the `mcpwatchman` command
153
+ api/ JSON API (FastAPI)
154
+ client/ API client shared by the CLI
155
+ db/ schema and models
156
+ workers/crawler/ registry polling and source resolution
157
+ workers/scanner/ static analysis and dependency scanning
158
+ workers/scoring/ axis scoring and the versioned weights
159
+ evals/gold-set/ hand-audited servers the scoring calibrates against
160
+ docker/ worker image
161
+ site/ public site (Astro)
162
+ ```
163
+
164
+ ## Contributing
165
+
166
+ It is early, and the most useful contributions right now are adversarial ones: tell us where the methodology is wrong.
167
+ Open an issue if a proposed check produces false positives you can demonstrate, if an axis misses a real class of MCP risk, or if a weight looks indefensible.
168
+ Rule contributions become genuinely useful once the scanner lands — `rules/` is deliberately open so that the detection logic can be argued with rather than taken on faith.
169
+
170
+ **Reporting a vulnerability in `mcpwatchman` itself** — not in a server we scan — goes through [GitHub's private vulnerability reporting](https://github.com/kVadrum/mcpwatchman/security/advisories/new) on this repository.
171
+ We hold ourselves to the same disclosure terms we apply to everyone else, with no exception for ourselves.
172
+
173
+ ## License
174
+
175
+ [MIT](./LICENSE).
176
+ KeMeK Network © 2026.
177
+
178
+ ### Trademarks
179
+
180
+ The "mcpwatchman" name is a trademark of KeMeK Network.
181
+ It is not covered by the code or content licenses.
182
+ No rights to use the name are granted by this repository.
183
+ Independent forks must replace the brand name with their own.
@@ -0,0 +1,149 @@
1
+ # mcpwatchman
2
+
3
+ **Independent security and quality audit for [Model Context Protocol](https://modelcontextprotocol.io) servers.**
4
+
5
+ `mcpwatchman` continuously scans every server in the official MCP registry and publishes a transparent, evidence-linked assessment of each one — so you can answer "is this MCP server safe to install?" before you wire it into your agent.
6
+
7
+ > **Status: early — building in public.**
8
+ > The methodology and architecture are settled and the repository is scaffolded; the scanner, scoring engine, site, and CLI are under active construction.
9
+ > **Nothing here is production-ready, and no scores have been published yet.**
10
+ > Watch the repo to follow along.
11
+
12
+ ---
13
+
14
+ ## Why
15
+
16
+ MCP adoption is accelerating across Claude Code, Cursor, Continue, Zed, Goose and more — and people are installing servers with the same blind trust they once gave `curl | bash`.
17
+ Research on thousands of public MCP servers has found widespread server-side request forgery, unsafe command execution, and servers exposed over HTTP with no authentication at all.
18
+
19
+ The official registry is **metadata-only by design** — it lists what exists, not what's safe.
20
+ `mcpwatchman` is the independent safety layer on top of it.
21
+ The closest analogues are [Mozilla Observatory](https://observatory.mozilla.org/) and [OpenSSF Scorecard](https://scorecard.dev/): independent, transparent, free, and trusted precisely because they aren't selling anything to the projects they score.
22
+
23
+ ## How it works
24
+
25
+ A daily pass over the official registry, then per server:
26
+
27
+ 1. **Resolve** the declared source repository and package artifacts from registry metadata.
28
+ 2. **Fetch** published source and lockfiles — a shallow, sparse clone at a pinned commit, into a per-job sandboxed workspace with resource limits and no inbound network.
29
+ 3. **Analyze statically** — an MCP-specific [semgrep](https://semgrep.dev) ruleset for the patterns that matter in this ecosystem (tool-handler shell exec, unvalidated URL fetches, unsafe deserialization, path traversal), plus dependency CVE scanning, transport and auth inspection, and repository maintenance signals.
30
+ 4. **Score** each axis from the findings, with every point of the score traceable to the artifact that produced it.
31
+ 5. **Publish** to the site, JSON API, RSS feeds, and CLI.
32
+
33
+ **The analysis is static, always.**
34
+ We read published source, manifests and lockfiles.
35
+ **We never execute the code we scan**, and we never connect to a running server instance.
36
+
37
+ **So here is what a good score cannot tell you.**
38
+ Static analysis reads what a server *publishes*, so the blind spot is whatever that source does not determine: behavior gated on remote configuration, code fetched or generated at runtime, or a deployed server that differs from the repository it declares.
39
+ An unsafe pattern that merely *executes* at tool-invocation time is still detectable — catching those is the ruleset's core job — but a score here is evidence about published source, never a runtime guarantee.
40
+ Running every server we scan is a substantially larger sandboxing problem than reading it, and we would rather be narrow and honest about it than broad and quietly wrong.
41
+
42
+ ## How it scores
43
+
44
+ Every server is rated 0–100 on five axes, and **every score links to its evidence** — a file and line, a CVE identifier, a commit date.
45
+ No mystery numbers.
46
+
47
+ | Axis | Default weight | Captures |
48
+ |---|---|---|
49
+ | **Code Safety** | 30% | Static-analysis findings: shell exec, SSRF, deserialization, path traversal |
50
+ | **Auth Posture** | 20% | Authentication model, transport security, secret handling, scope granularity |
51
+ | **Dependency Health** | 20% | Known CVEs in direct and transitive dependencies |
52
+ | **Maintenance** | 15% | Activity, release cadence, issue responsiveness, bus factor |
53
+ | **Transparency** | 15% | License validity, documented behavior, declared scopes |
54
+
55
+ **Per-axis scores ship first; the composite waits for calibration.**
56
+ The weights above are *provisional* starting values informed by the threat literature — they live in [`src/mcpwatchman/workers/scoring/weights.py`](src/mcpwatchman/workers/scoring/weights.py) and are versioned like any other code.
57
+ They become final only after calibration against a hand-audited gold set of roughly 30 servers spanning categories and a deliberate range of expected results.
58
+ Until that regression suite is green, public surfaces show the per-axis scores and suppress the composite.
59
+ A single blended number is the easiest thing to publish and the easiest thing to get quietly wrong, so it is the last thing we will ship.
60
+
61
+ The full methodology is published openly — anyone can audit our auditing.
62
+
63
+ ## Principles
64
+
65
+ - **Methodology over marketing.**
66
+ Every check, weight and threshold is documented and versioned.
67
+ - **Evidence or it doesn't ship.**
68
+ A finding with no linked artifact doesn't appear.
69
+ - **Facts, not characterizations.**
70
+ We describe observed code patterns, never intent.
71
+ Any finding can be appealed.
72
+ - **Collaborative, not competitive.**
73
+ We overlay the official registry; we don't replace it.
74
+ A server exists here only if the official registry lists it.
75
+ - **Free, and staying that way.**
76
+ The scanner, ruleset, site, API, RSS feeds and CLI are MIT-licensed and free.
77
+
78
+ ## Disclosure
79
+
80
+ Findings fall into two tracks, and they are handled differently on purpose.
81
+
82
+ **Track A — publicly observable patterns.**
83
+ Static-analysis hits, license and transport facts: anyone can run the same tools against the same published source and see the same thing.
84
+ There is no informational asymmetry to protect, so these are **not embargoed** — they appear on the server's page from the first scan that detects them.
85
+ Maintainer notification before publication is the right courtesy; delaying a fact every reader could derive themselves is not.
86
+
87
+ **Track B — findings that warrant coordination.**
88
+ Default embargo is **14 days** from maintainer contact, shortened to **7** where there is evidence of active exploitation, and extended up to **45 days total** when a maintainer comes back with a concrete fix timeline.
89
+ If we cannot reach a maintainer within 48 hours of the first attempt the clock still starts, and the target becomes **21 days from that first attempt** rather than 14 — being slow to check email is not the same as being unresponsive, and we don't punish it.
90
+
91
+ **Appeals.**
92
+ Anyone can contest a finding — maintainers and third parties alike, per finding, by ID.
93
+ Scores are recomputed, not negotiated: if the evidence is wrong the finding goes, if the code changed a rescan reflects it, and where a finding is technically valid but mitigated by context our rules don't model, it stays on the page with the maintainer's explanation attached.
94
+
95
+ ## Surfaces
96
+
97
+ | Surface | What it is |
98
+ |---|---|
99
+ | **Site** | Per-server pages with the evidence behind every axis, search and filtering, and the full methodology |
100
+ | **JSON API** | The same data, machine-readable — built to be consumed by agents and CI as a first-class audience, not as an afterthought |
101
+ | **RSS** | High-severity findings and score drops, for people who want to watch the ecosystem rather than one server |
102
+ | **Badges** | An auto-updating SVG a maintainer can put in their own README |
103
+ | **CLI** | `pip install mcpwatchman`, then `mcpwatchman check <server>` before you install it — with a `--threshold` exit code for CI |
104
+
105
+ ## Install
106
+
107
+ ```sh
108
+ pip install mcpwatchman # not yet published
109
+ ```
110
+
111
+ Requires Python 3.12+.
112
+
113
+ ## Repository layout
114
+
115
+ ```
116
+ rules/ MCP-specific semgrep ruleset, by language and category
117
+ src/mcpwatchman/
118
+ cli/ the `mcpwatchman` command
119
+ api/ JSON API (FastAPI)
120
+ client/ API client shared by the CLI
121
+ db/ schema and models
122
+ workers/crawler/ registry polling and source resolution
123
+ workers/scanner/ static analysis and dependency scanning
124
+ workers/scoring/ axis scoring and the versioned weights
125
+ evals/gold-set/ hand-audited servers the scoring calibrates against
126
+ docker/ worker image
127
+ site/ public site (Astro)
128
+ ```
129
+
130
+ ## Contributing
131
+
132
+ It is early, and the most useful contributions right now are adversarial ones: tell us where the methodology is wrong.
133
+ Open an issue if a proposed check produces false positives you can demonstrate, if an axis misses a real class of MCP risk, or if a weight looks indefensible.
134
+ Rule contributions become genuinely useful once the scanner lands — `rules/` is deliberately open so that the detection logic can be argued with rather than taken on faith.
135
+
136
+ **Reporting a vulnerability in `mcpwatchman` itself** — not in a server we scan — goes through [GitHub's private vulnerability reporting](https://github.com/kVadrum/mcpwatchman/security/advisories/new) on this repository.
137
+ We hold ourselves to the same disclosure terms we apply to everyone else, with no exception for ourselves.
138
+
139
+ ## License
140
+
141
+ [MIT](./LICENSE).
142
+ KeMeK Network © 2026.
143
+
144
+ ### Trademarks
145
+
146
+ The "mcpwatchman" name is a trademark of KeMeK Network.
147
+ It is not covered by the code or content licenses.
148
+ No rights to use the name are granted by this repository.
149
+ Independent forks must replace the brand name with their own.
@@ -0,0 +1,15 @@
1
+ # Worker image for mcpwatchman scanners (scaffold). See ADR-002 for the tool stack.
2
+ #
3
+ # The production image also installs osv-scanner (a Go binary) and pins the base
4
+ # by SHA digest; this scaffold installs the Python worker deps only.
5
+ FROM python:3.12-slim AS base
6
+
7
+ # Scans run as a non-root user (uid 1000) per the sandboxing posture.
8
+ RUN useradd --create-home --uid 1000 scanner
9
+
10
+ WORKDIR /app
11
+ COPY pyproject.toml README.md ./
12
+ COPY src ./src
13
+ RUN pip install --no-cache-dir ".[workers]"
14
+
15
+ USER scanner
File without changes
File without changes
@@ -0,0 +1,73 @@
1
+ [build-system]
2
+ requires = ["hatchling"]
3
+ build-backend = "hatchling.build"
4
+
5
+ [project]
6
+ name = "mcpwatchman"
7
+ version = "0.0.2"
8
+ description = "Independent security and quality audit for Model Context Protocol servers."
9
+ readme = "README.md"
10
+ requires-python = ">=3.12"
11
+ license = "MIT"
12
+ license-files = ["LICENSE"]
13
+ authors = [{ name = "kVadrum" }]
14
+ keywords = ["mcp", "model-context-protocol", "security", "audit", "static-analysis", "supply-chain"]
15
+ classifiers = [
16
+ "Development Status :: 2 - Pre-Alpha",
17
+ "Intended Audience :: Developers",
18
+ "Topic :: Security",
19
+ "Programming Language :: Python :: 3.12",
20
+ ]
21
+ # Core deps = the CLI + API client only, so `pip install mcpwatchman` stays light.
22
+ # Scanner/worker and API deps live in optional groups below.
23
+ dependencies = [
24
+ "click>=8.1",
25
+ "rich>=13.7",
26
+ "httpx>=0.27",
27
+ "pydantic>=2.7",
28
+ "pydantic-settings>=2.3",
29
+ ]
30
+
31
+ [project.optional-dependencies]
32
+ api = [
33
+ "fastapi>=0.111",
34
+ "uvicorn[standard]>=0.30",
35
+ ]
36
+ workers = [
37
+ # DB: ADR-004 (Postgres + SQLAlchemy 2.x + Alembic)
38
+ "sqlalchemy>=2.0",
39
+ "alembic>=1.13",
40
+ # Static analysis: ADR-002 (semgrep core; osv-scanner is an external Go binary)
41
+ "semgrep>=1.75",
42
+ "detect-secrets>=1.5",
43
+ # Queue: leaning Procrastinate (Postgres-native) over RQ — see STATE.md.
44
+ # Uncomment when the worker queue lands and the choice is ratified.
45
+ # "procrastinate>=2.0",
46
+ ]
47
+ dev = [
48
+ # Pinned to a minor: ruff gates CI, and new lint rules ship in 0.x releases.
49
+ "ruff~=0.16.7",
50
+ "mypy>=1.10",
51
+ "pytest>=8.2",
52
+ ]
53
+
54
+ [project.urls]
55
+ Homepage = "https://mcpwatchman.com"
56
+ Repository = "https://github.com/kVadrum/mcpwatchman"
57
+
58
+ [project.scripts]
59
+ mcpwatchman = "mcpwatchman.cli.main:cli"
60
+
61
+ [tool.hatch.build.targets.wheel]
62
+ packages = ["src/mcpwatchman"]
63
+
64
+ [tool.ruff]
65
+ target-version = "py312"
66
+ line-length = 100
67
+ src = ["src"]
68
+
69
+ [tool.ruff.lint]
70
+ select = ["E", "F", "I", "UP", "B", "SIM"]
71
+
72
+ [tool.pytest.ini_options]
73
+ testpaths = ["src", "tests"]
@@ -0,0 +1,23 @@
1
+ # mcpwatchman semgrep rules
2
+
3
+ The custom MCP-specific semgrep ruleset — the differentiating asset of mcpwatchman.
4
+ Published openly under [MIT](../LICENSE).
5
+
6
+ Organized by language and category:
7
+
8
+ ```
9
+ rules/
10
+ ├── python/ # shell-exec, ssrf, deserialization, path-traversal, mcp-specific
11
+ ├── javascript/ # + prototype-pollution
12
+ ├── go/
13
+ └── _meta/ # severity-mapping (rule_id -> severity, confidence), changelog
14
+ ```
15
+
16
+ Each rule is tagged with a severity (`critical`/`high`/`medium`/`low`/`informational`)
17
+ and a confidence (`high`/`medium`/`low`). `_meta/severity-mapping.yaml` is the
18
+ canonical map consumed by the scoring engine.
19
+
20
+ Every rule ships with fixture tests (true positive, true negative, and a
21
+ false-positive-resistance case) and a documented false-positive rate measured
22
+ against the calibration gold set. Rules above a 20% false-positive rate are
23
+ demoted from high to medium confidence (smaller scoring deduction).
File without changes
File without changes
File without changes
File without changes
@@ -0,0 +1,8 @@
1
+ # mcpwatchman site (Astro)
2
+
3
+ The public site — per-server pages, search/filter, methodology pages, badges,
4
+ and RSS feeds — is built here as a separate Astro project (static-first,
5
+ Cloudflare Pages).
6
+
7
+ Not scaffolded yet. It will be created with the workspace `frontend-design`
8
+ skill rather than freehanded, to keep the design distinctive and accessible.
@@ -0,0 +1,3 @@
1
+ """mcpwatchman — independent security and quality audit for MCP servers."""
2
+
3
+ __version__ = "0.0.2"
@@ -0,0 +1 @@
1
+ """JSON API for mcpwatchman (FastAPI)."""
@@ -0,0 +1,18 @@
1
+ """mcpwatchman JSON API (scaffold — only the health probe is live).
2
+
3
+ Run with: uvicorn mcpwatchman.api.main:app (requires the `api` extra).
4
+ """
5
+
6
+ from __future__ import annotations
7
+
8
+ from fastapi import FastAPI
9
+
10
+ from mcpwatchman import __version__
11
+
12
+ app = FastAPI(title="mcpwatchman API", version=__version__)
13
+
14
+
15
+ @app.get("/v1/healthz")
16
+ def healthz() -> dict[str, str]:
17
+ """Liveness probe."""
18
+ return {"status": "ok"}
@@ -0,0 +1 @@
1
+ """Command-line interface for mcpwatchman."""
@@ -0,0 +1,28 @@
1
+ """mcpwatchman command-line interface (scaffold)."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import click
6
+
7
+ from mcpwatchman import __version__
8
+
9
+
10
+ @click.group()
11
+ @click.version_option(__version__, prog_name="mcpwatchman")
12
+ def cli() -> None:
13
+ """Independent security and quality audit for MCP servers."""
14
+
15
+
16
+ @cli.command()
17
+ @click.argument("server")
18
+ def check(server: str) -> None:
19
+ """Show the trust score for SERVER (not implemented yet)."""
20
+ # Output formats (--format pretty/json/compact) return once scanning lands.
21
+ raise click.ClickException(
22
+ f"scanning {server!r} is not implemented yet — mcpwatchman is pre-v0.1. "
23
+ "Follow https://github.com/kVadrum/mcpwatchman"
24
+ )
25
+
26
+
27
+ if __name__ == "__main__":
28
+ cli()
@@ -0,0 +1 @@
1
+ """Typed API client shared by the CLI and internal tooling."""
@@ -0,0 +1 @@
1
+ """Database layer (Postgres + SQLAlchemy 2.x). See ADR-004."""
@@ -0,0 +1,14 @@
1
+ """SQLAlchemy 2.x models (scaffold).
2
+
3
+ The schema — servers, server_versions, scan_runs, findings, evidence, scores,
4
+ score_history, and supporting tables — lands with the first Alembic migration.
5
+ Requires the `workers` extra.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ from sqlalchemy.orm import DeclarativeBase
11
+
12
+
13
+ class Base(DeclarativeBase):
14
+ """Declarative base for all mcpwatchman models."""
@@ -0,0 +1 @@
1
+ """Crawler, scanner, and scoring engine. Does not import from mcpwatchman.api."""
@@ -0,0 +1 @@
1
+ """Registry poller + scan-job enqueue."""
@@ -0,0 +1 @@
1
+ """Scan pipeline: source resolution, semgrep, osv-scanner, auth/transport/license detectors."""
@@ -0,0 +1 @@
1
+ """Scoring engine: per-axis scores, composite, evidence helpers. The single source of axis math."""
@@ -0,0 +1,54 @@
1
+ """Versioned axis weights for the mcpwatchman composite score.
2
+
3
+ These are PROVISIONAL starting values informed by the threat literature, not
4
+ calibration outputs. The composite they produce MUST NOT be published on public
5
+ surfaces until gold-set calibration validates the weights for a given
6
+ methodology version — until then the site, badges, and CLI render per-axis
7
+ scores only (the `COMPOSITE_PUBLISHED` gate below).
8
+
9
+ Weights are versioned in lockstep with the scoring methodology version, and
10
+ each version's weights sum to 1.0.
11
+ """
12
+
13
+ from __future__ import annotations
14
+
15
+ AXIS_WEIGHTS: dict[str, dict[str, float]] = {
16
+ "0.2.0": {
17
+ "code_safety": 0.30,
18
+ "auth_posture": 0.20,
19
+ "maintenance": 0.15,
20
+ "dependency_health": 0.20,
21
+ "transparency": 0.15,
22
+ },
23
+ }
24
+
25
+ CURRENT_METHODOLOGY_VERSION = "0.2.0"
26
+
27
+ # Per-version gate: composite is exposed publicly only once calibration passes.
28
+ COMPOSITE_PUBLISHED: dict[str, bool] = {
29
+ "0.2.0": False,
30
+ }
31
+
32
+
33
+ def weights_for(version: str = CURRENT_METHODOLOGY_VERSION) -> dict[str, float]:
34
+ """Return the axis-weight map for a methodology version."""
35
+ try:
36
+ return AXIS_WEIGHTS[version]
37
+ except KeyError:
38
+ known = ", ".join(sorted(AXIS_WEIGHTS))
39
+ raise ValueError(f"unknown methodology version {version!r}; known: {known}") from None
40
+
41
+
42
+ def composite_published(version: str = CURRENT_METHODOLOGY_VERSION) -> bool:
43
+ """Whether the composite may be shown publicly for this methodology version.
44
+
45
+ Fails CLOSED: an unknown version answers False rather than raising, so a
46
+ version added to AXIS_WEIGHTS and forgotten here cannot publish an
47
+ uncalibrated composite. The omission is caught in CI by test_weights.py
48
+ instead — safe in production, loud where a human is looking.
49
+
50
+ Exists so no caller indexes COMPOSITE_PUBLISHED directly: a caller supplying
51
+ its own `.get(version, True)` default would reintroduce exactly the failure
52
+ this gate prevents.
53
+ """
54
+ return COMPOSITE_PUBLISHED.get(version, False)
@@ -0,0 +1,53 @@
1
+ """Invariants for the versioned axis weights.
2
+
3
+ `weights.py` documents two rules and enforced neither: that each version's
4
+ weights sum to 1.0, and that the composite gate has an entry per version.
5
+ Both are hand-edited constants, so CI is the right place to hold them.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ import pytest
11
+
12
+ from mcpwatchman.workers.scoring.weights import (
13
+ AXIS_WEIGHTS,
14
+ COMPOSITE_PUBLISHED,
15
+ CURRENT_METHODOLOGY_VERSION,
16
+ composite_published,
17
+ weights_for,
18
+ )
19
+
20
+
21
+ @pytest.mark.parametrize("version", sorted(AXIS_WEIGHTS))
22
+ def test_weights_sum_to_one(version: str) -> None:
23
+ # Composites are compared across methodology versions; a version summing to
24
+ # 0.95 silently rescales every score rather than failing.
25
+ assert sum(AXIS_WEIGHTS[version].values()) == pytest.approx(1.0)
26
+
27
+
28
+ @pytest.mark.parametrize("version", sorted(AXIS_WEIGHTS))
29
+ def test_every_version_declares_its_composite_gate(version: str) -> None:
30
+ # The accessor fails closed, so a missing entry is safe but invisible.
31
+ # This is the thing that makes it visible.
32
+ assert version in COMPOSITE_PUBLISHED
33
+
34
+
35
+ def test_current_version_is_known() -> None:
36
+ assert CURRENT_METHODOLOGY_VERSION in AXIS_WEIGHTS
37
+
38
+
39
+ def test_composite_stays_unpublished_until_calibration() -> None:
40
+ # Pins the project's first "Do not": no composite on any public surface
41
+ # until gold-set calibration passes. When that legitimately flips, this
42
+ # test fails — which is the point. Flipping it should be deliberate and
43
+ # reviewed, not a constant edited in passing.
44
+ assert composite_published() is False
45
+
46
+
47
+ def test_unknown_version_fails_closed_not_loud() -> None:
48
+ assert composite_published("99.0.0") is False
49
+
50
+
51
+ def test_weights_for_names_the_known_versions() -> None:
52
+ with pytest.raises(ValueError, match="known: 0.2.0"):
53
+ weights_for("99.0.0")