cfr-mcp 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- cfr_mcp-0.1.0/.github/workflows/ci.yml +22 -0
- cfr_mcp-0.1.0/.gitignore +6 -0
- cfr_mcp-0.1.0/LICENSE +21 -0
- cfr_mcp-0.1.0/PKG-INFO +120 -0
- cfr_mcp-0.1.0/PLAN.md +172 -0
- cfr_mcp-0.1.0/README.md +93 -0
- cfr_mcp-0.1.0/docs/manual-test-log.md +100 -0
- cfr_mcp-0.1.0/pyproject.toml +47 -0
- cfr_mcp-0.1.0/src/cfr_mcp/__init__.py +0 -0
- cfr_mcp-0.1.0/src/cfr_mcp/cache.py +78 -0
- cfr_mcp-0.1.0/src/cfr_mcp/citations.py +168 -0
- cfr_mcp-0.1.0/src/cfr_mcp/client.py +262 -0
- cfr_mcp-0.1.0/src/cfr_mcp/server.py +397 -0
- cfr_mcp-0.1.0/src/cfr_mcp/xml_parse.py +243 -0
- cfr_mcp-0.1.0/tests/conftest.py +27 -0
- cfr_mcp-0.1.0/tests/fixtures/NOTES.md +93 -0
- cfr_mcp-0.1.0/tests/fixtures/agencies.json +1 -0
- cfr_mcp-0.1.0/tests/fixtures/appendix_40_261_I.xml +17 -0
- cfr_mcp-0.1.0/tests/fixtures/corrections.json +1 -0
- cfr_mcp-0.1.0/tests/fixtures/corrections_21.json +1 -0
- cfr_mcp-0.1.0/tests/fixtures/counts_hierarchy.json +1 -0
- cfr_mcp-0.1.0/tests/fixtures/part_1_2.xml +60 -0
- cfr_mcp-0.1.0/tests/fixtures/search_results.json +1 -0
- cfr_mcp-0.1.0/tests/fixtures/section_1_2_6.xml +5 -0
- cfr_mcp-0.1.0/tests/fixtures/section_21_101_9.xml +694 -0
- cfr_mcp-0.1.0/tests/fixtures/structure_title1.json +1 -0
- cfr_mcp-0.1.0/tests/fixtures/subpart_40_261_A.xml +857 -0
- cfr_mcp-0.1.0/tests/fixtures/titles.json +1 -0
- cfr_mcp-0.1.0/tests/fixtures/versions.json +1 -0
- cfr_mcp-0.1.0/tests/fixtures/versions_filtered.json +1 -0
- cfr_mcp-0.1.0/tests/test_citations.py +85 -0
- cfr_mcp-0.1.0/tests/test_client.py +95 -0
- cfr_mcp-0.1.0/tests/test_hardening.py +117 -0
- cfr_mcp-0.1.0/tests/test_tools.py +257 -0
- cfr_mcp-0.1.0/tests/test_xml_parse.py +101 -0
- cfr_mcp-0.1.0/uv.lock +1025 -0
|
@@ -0,0 +1,22 @@
|
|
|
1
|
+
name: CI
|
|
2
|
+
|
|
3
|
+
on:
|
|
4
|
+
push:
|
|
5
|
+
branches: [main]
|
|
6
|
+
pull_request:
|
|
7
|
+
|
|
8
|
+
jobs:
|
|
9
|
+
test:
|
|
10
|
+
runs-on: ubuntu-latest
|
|
11
|
+
strategy:
|
|
12
|
+
fail-fast: false
|
|
13
|
+
matrix:
|
|
14
|
+
python-version: ["3.11", "3.12", "3.13"]
|
|
15
|
+
steps:
|
|
16
|
+
- uses: actions/checkout@v4
|
|
17
|
+
- uses: astral-sh/setup-uv@v5
|
|
18
|
+
with:
|
|
19
|
+
python-version: ${{ matrix.python-version }}
|
|
20
|
+
- run: uv sync --extra dev
|
|
21
|
+
- run: uv run ruff check .
|
|
22
|
+
- run: uv run pytest -q
|
cfr_mcp-0.1.0/.gitignore
ADDED
cfr_mcp-0.1.0/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Ryan McCalla
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
cfr_mcp-0.1.0/PKG-INFO
ADDED
|
@@ -0,0 +1,120 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: cfr-mcp
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: MCP server for the US Code of Federal Regulations (eCFR)
|
|
5
|
+
Project-URL: Homepage, https://github.com/mccallar/cfr-mcp
|
|
6
|
+
Project-URL: Repository, https://github.com/mccallar/cfr-mcp
|
|
7
|
+
Project-URL: Issues, https://github.com/mccallar/cfr-mcp/issues
|
|
8
|
+
License: MIT
|
|
9
|
+
License-File: LICENSE
|
|
10
|
+
Keywords: cfr,compliance,ecfr,government,mcp,regulations
|
|
11
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
12
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
13
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
14
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
15
|
+
Classifier: Topic :: Text Processing :: Indexing
|
|
16
|
+
Requires-Python: >=3.11
|
|
17
|
+
Requires-Dist: httpx>=0.27
|
|
18
|
+
Requires-Dist: lxml>=5.0
|
|
19
|
+
Requires-Dist: mcp>=2.0
|
|
20
|
+
Requires-Dist: pydantic>=2.0
|
|
21
|
+
Provides-Extra: dev
|
|
22
|
+
Requires-Dist: pytest-asyncio>=0.23; extra == 'dev'
|
|
23
|
+
Requires-Dist: pytest>=8.0; extra == 'dev'
|
|
24
|
+
Requires-Dist: respx>=0.21; extra == 'dev'
|
|
25
|
+
Requires-Dist: ruff>=0.6; extra == 'dev'
|
|
26
|
+
Description-Content-Type: text/markdown
|
|
27
|
+
|
|
28
|
+
# cfr-mcp
|
|
29
|
+
|
|
30
|
+
[](https://github.com/mccallar/cfr-mcp/actions/workflows/ci.yml)
|
|
31
|
+
[](https://pypi.org/project/cfr-mcp/)
|
|
32
|
+
|
|
33
|
+
An MCP server that gives AI assistants access to the **US Code of Federal Regulations**.
|
|
34
|
+
|
|
35
|
+
Ask "what does 21 CFR 101.9 require?" or "has 40 CFR 261 changed since 2023?" and get
|
|
36
|
+
the actual regulation text, with citations, instead of a plausible-sounding guess.
|
|
37
|
+
|
|
38
|
+
Unofficial community project. Not affiliated with or endorsed by the Office of the
|
|
39
|
+
Federal Register, NARA, or GPO, and uses no government seals or logos.
|
|
40
|
+
|
|
41
|
+
## Install
|
|
42
|
+
|
|
43
|
+
Requires Python 3.11+. No API key — the eCFR API is open.
|
|
44
|
+
|
|
45
|
+
Claude Code:
|
|
46
|
+
|
|
47
|
+
```bash
|
|
48
|
+
claude mcp add cfr -- uvx cfr-mcp
|
|
49
|
+
```
|
|
50
|
+
|
|
51
|
+
Any other MCP client:
|
|
52
|
+
|
|
53
|
+
```json
|
|
54
|
+
{
|
|
55
|
+
"mcpServers": {
|
|
56
|
+
"cfr": {
|
|
57
|
+
"command": "uvx",
|
|
58
|
+
"args": ["cfr-mcp"]
|
|
59
|
+
}
|
|
60
|
+
}
|
|
61
|
+
}
|
|
62
|
+
```
|
|
63
|
+
|
|
64
|
+
## Tools
|
|
65
|
+
|
|
66
|
+
| Tool | What it does |
|
|
67
|
+
|---|---|
|
|
68
|
+
| `lookup_citation` | Text of a citation — `21 CFR 101.9`, `40 CFR 261.4(b)(1)`, `40 CFR Part 261 Subpart C` |
|
|
69
|
+
| `search_regulations` | Full-text search; returns citations and snippets, never bodies |
|
|
70
|
+
| `where_does_term_appear` | Which titles contain a term, with hit counts. Fetches no text |
|
|
71
|
+
| `browse_structure` | The hierarchy of a title or part, no text |
|
|
72
|
+
| `what_changed` | Amendment history and published corrections for a citation |
|
|
73
|
+
| `list_agencies` | Maps agency names to the CFR titles they administer |
|
|
74
|
+
|
|
75
|
+
Point-in-time works throughout: pass `date` as `YYYY-MM-DD` to read the CFR as it stood.
|
|
76
|
+
|
|
77
|
+
## Design notes
|
|
78
|
+
|
|
79
|
+
**Context budget is the whole game.** The eCFR `full` endpoint returns an entire
|
|
80
|
+
downloadable XML document for a title-level request. Every tool here caps output and
|
|
81
|
+
degrades to an outline rather than dumping text into the model's context. Title-level
|
|
82
|
+
XML requests are refused outright.
|
|
83
|
+
|
|
84
|
+
**Dates must be resolved, not assumed.** Versioner routes 404 on dates that aren't
|
|
85
|
+
valid issue dates for a title, so the client resolves through `titles.json` first
|
|
86
|
+
rather than passing today's date blindly.
|
|
87
|
+
|
|
88
|
+
**Caching is courtesy.** The eCFR publishes no rate limit and has no key to identify
|
|
89
|
+
callers politely, so the client caches to disk (historical dates forever, since
|
|
90
|
+
point-in-time content is immutable) and self-limits concurrency. Set `CFR_MCP_CACHE_DIR`
|
|
91
|
+
to relocate the cache.
|
|
92
|
+
|
|
93
|
+
## Legal
|
|
94
|
+
|
|
95
|
+
Regulation text is free to reproduce. **1 CFR 2.6** states that any person may reproduce
|
|
96
|
+
or republish material appearing in the Federal Register, with no restrictions on what is
|
|
97
|
+
reproduced, who reproduces it, or where. Federal government works are also outside
|
|
98
|
+
copyright under 17 U.S.C. §105.
|
|
99
|
+
|
|
100
|
+
**Incorporation by reference.** Some CFR sections incorporate private standards
|
|
101
|
+
(ASTM, NFPA, ASHRAE) whose copyright status after incorporation remains unsettled.
|
|
102
|
+
The eCFR does not contain the text of those standards and neither does this server —
|
|
103
|
+
it returns the citation only. Obtain standards from the issuing organization or the
|
|
104
|
+
Office of the Federal Register reading room.
|
|
105
|
+
|
|
106
|
+
**Status of the text.** The eCFR is authoritative but unofficial. Anyone relying on it
|
|
107
|
+
for legal research should verify against the current official CFR, the daily Federal
|
|
108
|
+
Register, and the List of CFR Sections Affected (LSA).
|
|
109
|
+
|
|
110
|
+
**Not legal advice.** This is a retrieval tool. It returns the text of regulations; it
|
|
111
|
+
does not tell you whether you are compliant with them.
|
|
112
|
+
|
|
113
|
+
## Development
|
|
114
|
+
|
|
115
|
+
```bash
|
|
116
|
+
uv sync --extra dev
|
|
117
|
+
uv run pytest
|
|
118
|
+
```
|
|
119
|
+
|
|
120
|
+
MIT licensed.
|
cfr_mcp-0.1.0/PLAN.md
ADDED
|
@@ -0,0 +1,172 @@
|
|
|
1
|
+
# PLAN.md — cfr-mcp: scaffold → published v0.1
|
|
2
|
+
|
|
3
|
+
Execution plan for Claude Code. Work phase by phase; do not start a phase until
|
|
4
|
+
the previous phase's acceptance criteria pass. Commit at the end of each phase.
|
|
5
|
+
|
|
6
|
+
## Context
|
|
7
|
+
|
|
8
|
+
The repo contains a working scaffold (~700 lines):
|
|
9
|
+
|
|
10
|
+
- `src/cfr_mcp/citations.py` — CFR citation parser. **Tested, 20 cases passing.**
|
|
11
|
+
- `src/cfr_mcp/client.py` — async httpx client. All endpoint paths in the
|
|
12
|
+
`ENDPOINTS` dict. **Never run against the live API.**
|
|
13
|
+
- `src/cfr_mcp/xml_parse.py` — eCFR XML → text, with size capping.
|
|
14
|
+
**Written from documented conventions, never seen real XML.**
|
|
15
|
+
- `src/cfr_mcp/cache.py` — disk cache. Historical dates cached forever.
|
|
16
|
+
- `src/cfr_mcp/server.py` — FastMCP server, six tools. **JSON key names for
|
|
17
|
+
search/structure/versions/agencies responses are educated guesses.**
|
|
18
|
+
- `tests/test_citations.py` — parser tests (pytest).
|
|
19
|
+
|
|
20
|
+
The API: base `https://www.ecfr.gov`, no key. Services: admin
|
|
21
|
+
(`/api/admin/v1/...`), search (`/api/search/v1/...`), versioner
|
|
22
|
+
(`/api/versioner/v1/...`). Full endpoint list is in `client.py`.
|
|
23
|
+
|
|
24
|
+
## Invariants — never violate these
|
|
25
|
+
|
|
26
|
+
1. **Never request title-level XML.** `full_xml` refuses `is_title_only`
|
|
27
|
+
citations; keep it that way. Some titles are hundreds of MB.
|
|
28
|
+
2. **Every tool output is capped.** Large content degrades to an outline with
|
|
29
|
+
instructions to drill down. No tool may return unbounded text.
|
|
30
|
+
3. **Dates are resolved, not assumed.** Versioner routes 404 on invalid issue
|
|
31
|
+
dates; always resolve current dates through `titles.json`.
|
|
32
|
+
4. **Cache failures never break lookups.** Cache is best-effort.
|
|
33
|
+
5. **Errors are returned as readable strings to the model**, not raised through
|
|
34
|
+
the MCP layer, so the assistant can self-correct (e.g. bad date → explain).
|
|
35
|
+
6. **Retrieval only.** No tool or description may imply compliance judgment or
|
|
36
|
+
legal advice. Keep the source disclaimer on content-bearing outputs.
|
|
37
|
+
|
|
38
|
+
## Phase 0 — Environment (15 min)
|
|
39
|
+
|
|
40
|
+
- [ ] `git init`, initial commit of the scaffold as-is.
|
|
41
|
+
- [ ] `uv sync --extra dev`
|
|
42
|
+
- [ ] `uv run pytest` → all citation tests pass.
|
|
43
|
+
- [ ] Add MIT `LICENSE` file (year, author name).
|
|
44
|
+
- [ ] In `client.py`, replace `YOURNAME`/`YOUR_EMAIL` in `USER_AGENT`
|
|
45
|
+
(ask the human for the GitHub username and contact email).
|
|
46
|
+
|
|
47
|
+
**Accept:** pytest green, clean `git status`.
|
|
48
|
+
|
|
49
|
+
## Phase 1 — Verify API reality, capture fixtures (1–2 h)
|
|
50
|
+
|
|
51
|
+
Fetch real responses and save them under `tests/fixtures/`. Use small targets.
|
|
52
|
+
|
|
53
|
+
- [ ] `GET /api/versioner/v1/titles.json` → `fixtures/titles.json`.
|
|
54
|
+
Confirm field names `number`, `latest_issue_date`, `up_to_date_as_of`.
|
|
55
|
+
Fix `latest_date_for_title()` if they differ.
|
|
56
|
+
- [ ] `GET /api/versioner/v1/full/{date}/title-1.xml?part=2§ion=2.6`
|
|
57
|
+
(Title 1 is tiny) → `fixtures/section_1_2_6.xml`.
|
|
58
|
+
- [ ] Same for a mid-size section: `title-21 ... part=101§ion=101.9`
|
|
59
|
+
→ `fixtures/section_21_101_9.xml`.
|
|
60
|
+
- [ ] A whole small part: `title-1 ... part=2` → `fixtures/part_1_2.xml`.
|
|
61
|
+
- [ ] An appendix and a subpart request; confirm param names
|
|
62
|
+
(`appendix`, `subpart`) are what the API expects. Fix
|
|
63
|
+
`Citation.as_params()` if not.
|
|
64
|
+
- [ ] `GET /api/search/v1/results?query=nutrition+labeling&per_page=3`
|
|
65
|
+
→ `fixtures/search_results.json`. Record the real shape of results,
|
|
66
|
+
hierarchy fields, excerpt field, meta/total.
|
|
67
|
+
- [ ] `GET /api/search/v1/counts/hierarchy?query=asbestos`
|
|
68
|
+
→ `fixtures/counts_hierarchy.json`.
|
|
69
|
+
- [ ] `GET /api/versioner/v1/structure/{date}/title-1.json`
|
|
70
|
+
→ `fixtures/structure_title1.json`.
|
|
71
|
+
- [ ] `GET /api/versioner/v1/versions/title-1.json` → `fixtures/versions.json`.
|
|
72
|
+
Confirm filter param names (`conditions[part]`,
|
|
73
|
+
`conditions[issue_date][gte]`) actually filter; note whether the
|
|
74
|
+
version entries include `amendment_date`, `issue_date`, `identifier`.
|
|
75
|
+
- [ ] `GET /api/admin/v1/agencies.json` → `fixtures/agencies.json`.
|
|
76
|
+
- [ ] `GET /api/admin/v1/corrections/title/1.json` → `fixtures/corrections.json`.
|
|
77
|
+
- [ ] Write `tests/fixtures/NOTES.md` documenting every place the real
|
|
78
|
+
response differs from what the code assumes.
|
|
79
|
+
|
|
80
|
+
**Accept:** all fixtures on disk; NOTES.md lists discrepancies (may be empty).
|
|
81
|
+
|
|
82
|
+
## Phase 2 — Make the code match reality (2–4 h)
|
|
83
|
+
|
|
84
|
+
- [ ] Fix `xml_parse._build()` against the XML fixtures. Verify: DIV nesting,
|
|
85
|
+
`TYPE`/`N` attributes, `HEAD` headings, paragraph elements, and that
|
|
86
|
+
heading number-stripping works on real headings.
|
|
87
|
+
- [ ] Verify `extract_paragraphs` finds `(b)(1)` trails in real section text
|
|
88
|
+
(fixture 21 CFR 101.9 has deep paragraph nesting — ideal test).
|
|
89
|
+
- [ ] Fix every guessed JSON key in `server.py` rendering loops
|
|
90
|
+
(search results, structure walk, versions, agencies, corrections).
|
|
91
|
+
- [ ] Write fixture-backed tests (use `respx` to mock httpx against fixtures):
|
|
92
|
+
- `tests/test_xml_parse.py` — parse each XML fixture; assert headings,
|
|
93
|
+
section numbers, non-empty text; assert `render_capped` degrades to an
|
|
94
|
+
outline when `max_chars` is tiny.
|
|
95
|
+
- `tests/test_client.py` — date resolution from titles fixture; 404 →
|
|
96
|
+
readable `ECFRError`; title-level XML refusal; cache hit skips HTTP.
|
|
97
|
+
- `tests/test_tools.py` — call each of the six tools with mocked
|
|
98
|
+
transport; assert output is a string, contains expected citation,
|
|
99
|
+
and stays under the cap.
|
|
100
|
+
- [ ] `uv run ruff check --fix .` and resolve remaining lint.
|
|
101
|
+
|
|
102
|
+
**Accept:** full pytest suite green offline (fixtures only, no network).
|
|
103
|
+
|
|
104
|
+
## Phase 3 — Live integration with Claude Code (1 h)
|
|
105
|
+
|
|
106
|
+
- [ ] Register locally: `claude mcp add cfr -- uv run cfr-mcp`
|
|
107
|
+
(or `uv run mcp dev src/cfr_mcp/server.py` for the inspector).
|
|
108
|
+
- [ ] Exercise each tool through the assistant and record results in
|
|
109
|
+
`docs/manual-test-log.md`:
|
|
110
|
+
1. "What does 21 CFR 101.9(c) require?" → correct paragraph text.
|
|
111
|
+
2. "Search the CFR for 'per- and polyfluoroalkyl'" → citations + snippets.
|
|
112
|
+
3. "Where does 'asbestos' appear in the CFR?" → hierarchy counts.
|
|
113
|
+
4. "Show the structure of 40 CFR Part 261" → outline, no text dump.
|
|
114
|
+
5. "Has 40 CFR 261.4 changed since 2023-01-01?" → amendment dates.
|
|
115
|
+
6. "Which CFR titles does the EPA administer?" → Title 40 (+ others).
|
|
116
|
+
7. Adversarial: "Give me all of Title 40" → graceful refusal with
|
|
117
|
+
guidance, not an error or a dump.
|
|
118
|
+
8. A paragraph that doesn't exist: "21 CFR 101.9(z)(9)" → falls back to
|
|
119
|
+
section text, does not fabricate.
|
|
120
|
+
- [ ] Fix anything awkward the model trips on (tool descriptions are part of
|
|
121
|
+
the product — iterate on them here).
|
|
122
|
+
|
|
123
|
+
**Accept:** all eight logged with sane transcripts.
|
|
124
|
+
|
|
125
|
+
## Phase 4 — Hardening (1–2 h)
|
|
126
|
+
|
|
127
|
+
- [ ] Title 35 is reserved: `lookup_citation("35 CFR 1.1")` must return a
|
|
128
|
+
clear "reserved title" message. Add test.
|
|
129
|
+
- [ ] Very large part (e.g. 40 CFR 63): confirm outline degradation and
|
|
130
|
+
acceptable latency; consider streaming/size guard on the raw download
|
|
131
|
+
if response exceeds ~5 MB.
|
|
132
|
+
- [ ] Date edge cases: pre-2017 dates (point-in-time floor), future dates,
|
|
133
|
+
malformed dates — each returns a helpful message. Tests.
|
|
134
|
+
- [ ] Concurrency: fire 10 parallel `lookup_citation` calls; semaphore holds,
|
|
135
|
+
nothing corrupts the cache (meta/body write order).
|
|
136
|
+
- [ ] Add `--version` / `-h` handling to `main()`.
|
|
137
|
+
|
|
138
|
+
**Accept:** suite green, edge cases covered.
|
|
139
|
+
|
|
140
|
+
## Phase 5 — Publish (1–2 h; human-in-the-loop)
|
|
141
|
+
|
|
142
|
+
- [ ] README final pass: verify install block, tool table, legal section
|
|
143
|
+
(1 CFR 2.6, IBR caveat, "authoritative but unofficial" disclaimer,
|
|
144
|
+
no seals/no-endorsement note).
|
|
145
|
+
- [ ] Create GitHub repo (human), push, add topics: `mcp`, `ecfr`, `cfr`,
|
|
146
|
+
`regulations`, `model-context-protocol`.
|
|
147
|
+
- [ ] CI: GitHub Actions workflow running ruff + pytest on 3.11/3.12/3.13.
|
|
148
|
+
- [ ] `uv build`; test the wheel in a clean venv: `uvx --from dist/*.whl cfr-mcp`
|
|
149
|
+
starts and responds to an MCP `initialize`.
|
|
150
|
+
- [ ] Publish: human creates PyPI account + token; `uv publish`.
|
|
151
|
+
- [ ] Verify end-to-end: `uvx cfr-mcp` from PyPI works in Claude Code.
|
|
152
|
+
- [ ] Submit to the MCP registry / community server lists (human approves
|
|
153
|
+
the listing text).
|
|
154
|
+
- [ ] Tag `v0.1.0`, GitHub release with a short changelog.
|
|
155
|
+
|
|
156
|
+
**Accept:** a stranger can go from README to a working `lookup_citation`
|
|
157
|
+
call in under five minutes.
|
|
158
|
+
|
|
159
|
+
## Explicitly out of scope for v0.1
|
|
160
|
+
|
|
161
|
+
- Hosted/remote transport (SSE/HTTP) — local stdio only.
|
|
162
|
+
- Any paid features, alerting, or UI.
|
|
163
|
+
- Fetching text of standards incorporated by reference — permanently out;
|
|
164
|
+
return citations only.
|
|
165
|
+
- State regulations, Federal Register documents beyond corrections.
|
|
166
|
+
|
|
167
|
+
## Backlog for v0.2 (do not build now)
|
|
168
|
+
|
|
169
|
+
- Federal Register cross-links in `what_changed` (which rule caused the change).
|
|
170
|
+
- Rendered side-by-side diffs between two dates.
|
|
171
|
+
- `search_suggestions` tool using `/api/search/v1/suggestions`.
|
|
172
|
+
- Prebuilt vertical part-bundles (food labeling, hazmat) as MCP resources.
|
cfr_mcp-0.1.0/README.md
ADDED
|
@@ -0,0 +1,93 @@
|
|
|
1
|
+
# cfr-mcp
|
|
2
|
+
|
|
3
|
+
[](https://github.com/mccallar/cfr-mcp/actions/workflows/ci.yml)
|
|
4
|
+
[](https://pypi.org/project/cfr-mcp/)
|
|
5
|
+
|
|
6
|
+
An MCP server that gives AI assistants access to the **US Code of Federal Regulations**.
|
|
7
|
+
|
|
8
|
+
Ask "what does 21 CFR 101.9 require?" or "has 40 CFR 261 changed since 2023?" and get
|
|
9
|
+
the actual regulation text, with citations, instead of a plausible-sounding guess.
|
|
10
|
+
|
|
11
|
+
Unofficial community project. Not affiliated with or endorsed by the Office of the
|
|
12
|
+
Federal Register, NARA, or GPO, and uses no government seals or logos.
|
|
13
|
+
|
|
14
|
+
## Install
|
|
15
|
+
|
|
16
|
+
Requires Python 3.11+. No API key — the eCFR API is open.
|
|
17
|
+
|
|
18
|
+
Claude Code:
|
|
19
|
+
|
|
20
|
+
```bash
|
|
21
|
+
claude mcp add cfr -- uvx cfr-mcp
|
|
22
|
+
```
|
|
23
|
+
|
|
24
|
+
Any other MCP client:
|
|
25
|
+
|
|
26
|
+
```json
|
|
27
|
+
{
|
|
28
|
+
"mcpServers": {
|
|
29
|
+
"cfr": {
|
|
30
|
+
"command": "uvx",
|
|
31
|
+
"args": ["cfr-mcp"]
|
|
32
|
+
}
|
|
33
|
+
}
|
|
34
|
+
}
|
|
35
|
+
```
|
|
36
|
+
|
|
37
|
+
## Tools
|
|
38
|
+
|
|
39
|
+
| Tool | What it does |
|
|
40
|
+
|---|---|
|
|
41
|
+
| `lookup_citation` | Text of a citation — `21 CFR 101.9`, `40 CFR 261.4(b)(1)`, `40 CFR Part 261 Subpart C` |
|
|
42
|
+
| `search_regulations` | Full-text search; returns citations and snippets, never bodies |
|
|
43
|
+
| `where_does_term_appear` | Which titles contain a term, with hit counts. Fetches no text |
|
|
44
|
+
| `browse_structure` | The hierarchy of a title or part, no text |
|
|
45
|
+
| `what_changed` | Amendment history and published corrections for a citation |
|
|
46
|
+
| `list_agencies` | Maps agency names to the CFR titles they administer |
|
|
47
|
+
|
|
48
|
+
Point-in-time works throughout: pass `date` as `YYYY-MM-DD` to read the CFR as it stood.
|
|
49
|
+
|
|
50
|
+
## Design notes
|
|
51
|
+
|
|
52
|
+
**Context budget is the whole game.** The eCFR `full` endpoint returns an entire
|
|
53
|
+
downloadable XML document for a title-level request. Every tool here caps output and
|
|
54
|
+
degrades to an outline rather than dumping text into the model's context. Title-level
|
|
55
|
+
XML requests are refused outright.
|
|
56
|
+
|
|
57
|
+
**Dates must be resolved, not assumed.** Versioner routes 404 on dates that aren't
|
|
58
|
+
valid issue dates for a title, so the client resolves through `titles.json` first
|
|
59
|
+
rather than passing today's date blindly.
|
|
60
|
+
|
|
61
|
+
**Caching is courtesy.** The eCFR publishes no rate limit and has no key to identify
|
|
62
|
+
callers politely, so the client caches to disk (historical dates forever, since
|
|
63
|
+
point-in-time content is immutable) and self-limits concurrency. Set `CFR_MCP_CACHE_DIR`
|
|
64
|
+
to relocate the cache.
|
|
65
|
+
|
|
66
|
+
## Legal
|
|
67
|
+
|
|
68
|
+
Regulation text is free to reproduce. **1 CFR 2.6** states that any person may reproduce
|
|
69
|
+
or republish material appearing in the Federal Register, with no restrictions on what is
|
|
70
|
+
reproduced, who reproduces it, or where. Federal government works are also outside
|
|
71
|
+
copyright under 17 U.S.C. §105.
|
|
72
|
+
|
|
73
|
+
**Incorporation by reference.** Some CFR sections incorporate private standards
|
|
74
|
+
(ASTM, NFPA, ASHRAE) whose copyright status after incorporation remains unsettled.
|
|
75
|
+
The eCFR does not contain the text of those standards and neither does this server —
|
|
76
|
+
it returns the citation only. Obtain standards from the issuing organization or the
|
|
77
|
+
Office of the Federal Register reading room.
|
|
78
|
+
|
|
79
|
+
**Status of the text.** The eCFR is authoritative but unofficial. Anyone relying on it
|
|
80
|
+
for legal research should verify against the current official CFR, the daily Federal
|
|
81
|
+
Register, and the List of CFR Sections Affected (LSA).
|
|
82
|
+
|
|
83
|
+
**Not legal advice.** This is a retrieval tool. It returns the text of regulations; it
|
|
84
|
+
does not tell you whether you are compliant with them.
|
|
85
|
+
|
|
86
|
+
## Development
|
|
87
|
+
|
|
88
|
+
```bash
|
|
89
|
+
uv sync --extra dev
|
|
90
|
+
uv run pytest
|
|
91
|
+
```
|
|
92
|
+
|
|
93
|
+
MIT licensed.
|
|
@@ -0,0 +1,100 @@
|
|
|
1
|
+
# Manual test log — Phase 3 live integration
|
|
2
|
+
|
|
3
|
+
2026-08-29, against the live eCFR API. Tools were exercised at the tool-call
|
|
4
|
+
surface (the exact async functions MCP dispatches to), with arguments chosen
|
|
5
|
+
as an assistant would choose them. Outputs below are summarized; sizes are
|
|
6
|
+
exact.
|
|
7
|
+
|
|
8
|
+
## 1. "What does 21 CFR 101.9(c) require?"
|
|
9
|
+
|
|
10
|
+
`lookup_citation("21 CFR 101.9(c)")`
|
|
11
|
+
|
|
12
|
+
- First run returned the full (c) block: correct text, but **35,281 chars —
|
|
13
|
+
violated the output cap invariant**. Fixed: an oversized paragraph now
|
|
14
|
+
returns its opening text plus a sub-paragraph list. Re-run: 3,340 chars,
|
|
15
|
+
begins with the correct "(c) The declaration of nutrition information…"
|
|
16
|
+
text and ends "Sub-paragraphs available: (1)…(9). Request a deeper
|
|
17
|
+
citation like 101.9(c)(1)". ✅
|
|
18
|
+
|
|
19
|
+
## 2. "Search the CFR for 'per- and polyfluoroalkyl'"
|
|
20
|
+
|
|
21
|
+
`search_regulations("per- and polyfluoroalkyl", limit=5)`
|
|
22
|
+
|
|
23
|
+
- 1,268 chars. "72 result(s)… showing 5". Real citations (40 CFR 141.901,
|
|
24
|
+
705.1, 372.29, 705.3) with headings and clean snippets, no HTML. ✅
|
|
25
|
+
|
|
26
|
+
## 3. "Where does 'asbestos' appear in the CFR?"
|
|
27
|
+
|
|
28
|
+
`where_does_term_appear("asbestos")`
|
|
29
|
+
|
|
30
|
+
- 4,616 chars. Title-level hit counts with top-3 parts nested under each
|
|
31
|
+
(e.g. Title 16 → Part 1304 Ban of Consumer Patching Compounds…). First
|
|
32
|
+
run leaked `<strong>` tags inside part headings (the counts endpoint
|
|
33
|
+
highlights the query term); fixed with the same HTML stripping used for
|
|
34
|
+
search. ✅
|
|
35
|
+
|
|
36
|
+
## 4. "Show the structure of 40 CFR Part 261"
|
|
37
|
+
|
|
38
|
+
`browse_structure(40, part="261")`
|
|
39
|
+
|
|
40
|
+
- 6,378 chars. Subparts A–DD with section lines, appendices I–IX at the
|
|
41
|
+
bottom, zero body text. ✅
|
|
42
|
+
|
|
43
|
+
## 5. "Has 40 CFR 261.4 changed since 2023-01-01?"
|
|
44
|
+
|
|
45
|
+
`what_changed("40 CFR 261.4", since="2023-01-01")`
|
|
46
|
+
|
|
47
|
+
- 1,358 chars. Eleven amendment entries 2023-03-29 → 2025-03-21 with
|
|
48
|
+
amendment and issue dates, plus part-261 published corrections with FR
|
|
49
|
+
citations. Filter params verified on the wire
|
|
50
|
+
(`part=261&issue_date[gte]=2023-01-01`). ✅
|
|
51
|
+
|
|
52
|
+
## 6. "Which CFR titles does the EPA administer?"
|
|
53
|
+
|
|
54
|
+
`list_agencies("environmental protection")`
|
|
55
|
+
|
|
56
|
+
- 93 chars: "Environmental Protection Agency
|
|
57
|
+
[environmental-protection-agency] — Title(s) 2, 5, 40, 41, 48". First run
|
|
58
|
+
sorted titles as strings ("2, 40, 41, 48, 5"); fixed to numeric sort. ✅
|
|
59
|
+
|
|
60
|
+
## 7. Adversarial: "Give me all of Title 40"
|
|
61
|
+
|
|
62
|
+
`lookup_citation("Title 40")`
|
|
63
|
+
|
|
64
|
+
- 124 chars, zero network calls: "'Title 40' names a whole CFR title, which
|
|
65
|
+
is far too large to retrieve. Use browse_structure to navigate it, or
|
|
66
|
+
name a part." Graceful refusal with guidance. ✅
|
|
67
|
+
|
|
68
|
+
## 8. Nonexistent paragraph: "21 CFR 101.9(z)(9)"
|
|
69
|
+
|
|
70
|
+
`lookup_citation("21 CFR 101.9(z)(9)")`
|
|
71
|
+
|
|
72
|
+
- First run fell back to a degenerate outline (a childless section's
|
|
73
|
+
"outline" was just its heading, twice) and never said the paragraph was
|
|
74
|
+
missing. Fixed: now 192 chars — "Paragraph (z)(9) does not exist in
|
|
75
|
+
21 CFR 101.9. Top-level paragraphs present: (a)…(l). Request the whole
|
|
76
|
+
section or one of those paragraphs." No fabrication. ✅
|
|
77
|
+
|
|
78
|
+
## Fixes made during this phase
|
|
79
|
+
|
|
80
|
+
1. Output cap now applies to paragraph-narrowed text (scenario 1); oversized
|
|
81
|
+
paragraphs degrade to intro + sub-paragraph list.
|
|
82
|
+
2. `where_does_term_appear` strips HTML from headings (scenario 3).
|
|
83
|
+
3. `list_agencies` sorts titles numerically (scenario 6).
|
|
84
|
+
4. Missing paragraphs get an explicit not-found message listing the
|
|
85
|
+
paragraphs that do exist (scenario 8).
|
|
86
|
+
5. `render_capped` on a huge childless section now outlines its top-level
|
|
87
|
+
paragraphs instead of repeating the heading.
|
|
88
|
+
|
|
89
|
+
All five have regression tests in `tests/`.
|
|
90
|
+
|
|
91
|
+
## Phase 4 live checks
|
|
92
|
+
|
|
93
|
+
- `lookup_citation("40 CFR Part 63")` (one of the largest CFR parts): the
|
|
94
|
+
streaming byte guard aborted the download at 20 MB after 18.1s with
|
|
95
|
+
"Request a smaller unit — a subpart or a section — or use
|
|
96
|
+
browse_structure to navigate this part." No multi-hundred-MB download,
|
|
97
|
+
no dump. ✅
|
|
98
|
+
- `lookup_citation("40 CFR Part 261")` (mid-size, 1.5M chars of text,
|
|
99
|
+
~6 MB XML): 1.7s, degraded to an 8,009-char subpart/section outline with
|
|
100
|
+
per-node sizes. ✅
|
|
@@ -0,0 +1,47 @@
|
|
|
1
|
+
[project]
|
|
2
|
+
name = "cfr-mcp"
|
|
3
|
+
version = "0.1.0"
|
|
4
|
+
description = "MCP server for the US Code of Federal Regulations (eCFR)"
|
|
5
|
+
readme = "README.md"
|
|
6
|
+
requires-python = ">=3.11"
|
|
7
|
+
license = { text = "MIT" }
|
|
8
|
+
keywords = ["mcp", "cfr", "ecfr", "regulations", "government", "compliance"]
|
|
9
|
+
classifiers = [
|
|
10
|
+
"License :: OSI Approved :: MIT License",
|
|
11
|
+
"Programming Language :: Python :: 3.11",
|
|
12
|
+
"Programming Language :: Python :: 3.12",
|
|
13
|
+
"Programming Language :: Python :: 3.13",
|
|
14
|
+
"Topic :: Text Processing :: Indexing",
|
|
15
|
+
]
|
|
16
|
+
dependencies = [
|
|
17
|
+
"mcp>=2.0",
|
|
18
|
+
"httpx>=0.27",
|
|
19
|
+
"lxml>=5.0",
|
|
20
|
+
"pydantic>=2.0",
|
|
21
|
+
]
|
|
22
|
+
|
|
23
|
+
[project.optional-dependencies]
|
|
24
|
+
dev = ["pytest>=8.0", "pytest-asyncio>=0.23", "respx>=0.21", "ruff>=0.6"]
|
|
25
|
+
|
|
26
|
+
[project.scripts]
|
|
27
|
+
cfr-mcp = "cfr_mcp.server:main"
|
|
28
|
+
|
|
29
|
+
[project.urls]
|
|
30
|
+
Homepage = "https://github.com/mccallar/cfr-mcp"
|
|
31
|
+
Repository = "https://github.com/mccallar/cfr-mcp"
|
|
32
|
+
Issues = "https://github.com/mccallar/cfr-mcp/issues"
|
|
33
|
+
|
|
34
|
+
[build-system]
|
|
35
|
+
requires = ["hatchling"]
|
|
36
|
+
build-backend = "hatchling.build"
|
|
37
|
+
|
|
38
|
+
[tool.hatch.build.targets.wheel]
|
|
39
|
+
packages = ["src/cfr_mcp"]
|
|
40
|
+
|
|
41
|
+
[tool.pytest.ini_options]
|
|
42
|
+
pythonpath = ["src"]
|
|
43
|
+
testpaths = ["tests"]
|
|
44
|
+
asyncio_mode = "auto"
|
|
45
|
+
|
|
46
|
+
[tool.ruff]
|
|
47
|
+
line-length = 100
|
|
File without changes
|
|
@@ -0,0 +1,78 @@
|
|
|
1
|
+
"""Disk cache for eCFR responses.
|
|
2
|
+
|
|
3
|
+
eCFR updates at most daily, and historical point-in-time content never changes
|
|
4
|
+
at all. Caching is the whole reason this server can be a good citizen against
|
|
5
|
+
an API that publishes no rate limit and has no key to identify us politely.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import hashlib
|
|
11
|
+
import json
|
|
12
|
+
import os
|
|
13
|
+
import time
|
|
14
|
+
from pathlib import Path
|
|
15
|
+
from typing import Any
|
|
16
|
+
|
|
17
|
+
DEFAULT_TTL = 24 * 60 * 60 # eCFR updates daily
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def _default_dir() -> Path:
|
|
21
|
+
env = os.environ.get("CFR_MCP_CACHE_DIR")
|
|
22
|
+
if env:
|
|
23
|
+
return Path(env)
|
|
24
|
+
base = os.environ.get("XDG_CACHE_HOME") or (Path.home() / ".cache")
|
|
25
|
+
return Path(base) / "cfr-mcp"
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
class Cache:
|
|
29
|
+
def __init__(self, directory: Path | None = None, ttl: int = DEFAULT_TTL) -> None:
|
|
30
|
+
self.dir = directory or _default_dir()
|
|
31
|
+
self.ttl = ttl
|
|
32
|
+
self.dir.mkdir(parents=True, exist_ok=True)
|
|
33
|
+
|
|
34
|
+
def key(self, path: str, params: dict[str, Any] | None = None) -> str:
|
|
35
|
+
blob = path + "?" + json.dumps(params or {}, sort_keys=True)
|
|
36
|
+
return hashlib.sha256(blob.encode()).hexdigest()[:32]
|
|
37
|
+
|
|
38
|
+
def _paths(self, key: str) -> tuple[Path, Path]:
|
|
39
|
+
return self.dir / f"{key}.body", self.dir / f"{key}.meta"
|
|
40
|
+
|
|
41
|
+
def get(self, key: str) -> str | None:
|
|
42
|
+
body_path, meta_path = self._paths(key)
|
|
43
|
+
if not (body_path.exists() and meta_path.exists()):
|
|
44
|
+
return None
|
|
45
|
+
try:
|
|
46
|
+
meta = json.loads(meta_path.read_text())
|
|
47
|
+
except (json.JSONDecodeError, OSError):
|
|
48
|
+
return None
|
|
49
|
+
if (
|
|
50
|
+
not meta.get("immutable")
|
|
51
|
+
and time.time() - meta.get("stored_at", 0) > self.ttl
|
|
52
|
+
):
|
|
53
|
+
return None
|
|
54
|
+
try:
|
|
55
|
+
return body_path.read_text(encoding="utf-8")
|
|
56
|
+
except OSError:
|
|
57
|
+
return None
|
|
58
|
+
|
|
59
|
+
def set(self, key: str, body: str, *, immutable: bool = False) -> None:
|
|
60
|
+
body_path, meta_path = self._paths(key)
|
|
61
|
+
try:
|
|
62
|
+
# Write body first; a torn write leaves no meta and reads as a miss.
|
|
63
|
+
body_path.write_text(body, encoding="utf-8")
|
|
64
|
+
meta_path.write_text(
|
|
65
|
+
json.dumps({"stored_at": time.time(), "immutable": immutable})
|
|
66
|
+
)
|
|
67
|
+
except OSError:
|
|
68
|
+
pass # cache failures must never break a lookup
|
|
69
|
+
|
|
70
|
+
def clear(self) -> int:
|
|
71
|
+
removed = 0
|
|
72
|
+
for f in self.dir.glob("*.*"):
|
|
73
|
+
try:
|
|
74
|
+
f.unlink()
|
|
75
|
+
removed += 1
|
|
76
|
+
except OSError:
|
|
77
|
+
pass
|
|
78
|
+
return removed // 2
|