cfr-mcp 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (36) hide show
  1. cfr_mcp-0.1.0/.github/workflows/ci.yml +22 -0
  2. cfr_mcp-0.1.0/.gitignore +6 -0
  3. cfr_mcp-0.1.0/LICENSE +21 -0
  4. cfr_mcp-0.1.0/PKG-INFO +120 -0
  5. cfr_mcp-0.1.0/PLAN.md +172 -0
  6. cfr_mcp-0.1.0/README.md +93 -0
  7. cfr_mcp-0.1.0/docs/manual-test-log.md +100 -0
  8. cfr_mcp-0.1.0/pyproject.toml +47 -0
  9. cfr_mcp-0.1.0/src/cfr_mcp/__init__.py +0 -0
  10. cfr_mcp-0.1.0/src/cfr_mcp/cache.py +78 -0
  11. cfr_mcp-0.1.0/src/cfr_mcp/citations.py +168 -0
  12. cfr_mcp-0.1.0/src/cfr_mcp/client.py +262 -0
  13. cfr_mcp-0.1.0/src/cfr_mcp/server.py +397 -0
  14. cfr_mcp-0.1.0/src/cfr_mcp/xml_parse.py +243 -0
  15. cfr_mcp-0.1.0/tests/conftest.py +27 -0
  16. cfr_mcp-0.1.0/tests/fixtures/NOTES.md +93 -0
  17. cfr_mcp-0.1.0/tests/fixtures/agencies.json +1 -0
  18. cfr_mcp-0.1.0/tests/fixtures/appendix_40_261_I.xml +17 -0
  19. cfr_mcp-0.1.0/tests/fixtures/corrections.json +1 -0
  20. cfr_mcp-0.1.0/tests/fixtures/corrections_21.json +1 -0
  21. cfr_mcp-0.1.0/tests/fixtures/counts_hierarchy.json +1 -0
  22. cfr_mcp-0.1.0/tests/fixtures/part_1_2.xml +60 -0
  23. cfr_mcp-0.1.0/tests/fixtures/search_results.json +1 -0
  24. cfr_mcp-0.1.0/tests/fixtures/section_1_2_6.xml +5 -0
  25. cfr_mcp-0.1.0/tests/fixtures/section_21_101_9.xml +694 -0
  26. cfr_mcp-0.1.0/tests/fixtures/structure_title1.json +1 -0
  27. cfr_mcp-0.1.0/tests/fixtures/subpart_40_261_A.xml +857 -0
  28. cfr_mcp-0.1.0/tests/fixtures/titles.json +1 -0
  29. cfr_mcp-0.1.0/tests/fixtures/versions.json +1 -0
  30. cfr_mcp-0.1.0/tests/fixtures/versions_filtered.json +1 -0
  31. cfr_mcp-0.1.0/tests/test_citations.py +85 -0
  32. cfr_mcp-0.1.0/tests/test_client.py +95 -0
  33. cfr_mcp-0.1.0/tests/test_hardening.py +117 -0
  34. cfr_mcp-0.1.0/tests/test_tools.py +257 -0
  35. cfr_mcp-0.1.0/tests/test_xml_parse.py +101 -0
  36. cfr_mcp-0.1.0/uv.lock +1025 -0
@@ -0,0 +1,22 @@
1
+ name: CI
2
+
3
+ on:
4
+ push:
5
+ branches: [main]
6
+ pull_request:
7
+
8
+ jobs:
9
+ test:
10
+ runs-on: ubuntu-latest
11
+ strategy:
12
+ fail-fast: false
13
+ matrix:
14
+ python-version: ["3.11", "3.12", "3.13"]
15
+ steps:
16
+ - uses: actions/checkout@v4
17
+ - uses: astral-sh/setup-uv@v5
18
+ with:
19
+ python-version: ${{ matrix.python-version }}
20
+ - run: uv sync --extra dev
21
+ - run: uv run ruff check .
22
+ - run: uv run pytest -q
@@ -0,0 +1,6 @@
1
+ __pycache__/
2
+ *.py[cod]
3
+ .venv/
4
+ dist/
5
+ .pytest_cache/
6
+ .ruff_cache/
cfr_mcp-0.1.0/LICENSE ADDED
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Ryan McCalla
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
cfr_mcp-0.1.0/PKG-INFO ADDED
@@ -0,0 +1,120 @@
1
+ Metadata-Version: 2.5
2
+ Name: cfr-mcp
3
+ Version: 0.1.0
4
+ Summary: MCP server for the US Code of Federal Regulations (eCFR)
5
+ Project-URL: Homepage, https://github.com/mccallar/cfr-mcp
6
+ Project-URL: Repository, https://github.com/mccallar/cfr-mcp
7
+ Project-URL: Issues, https://github.com/mccallar/cfr-mcp/issues
8
+ License: MIT
9
+ License-File: LICENSE
10
+ Keywords: cfr,compliance,ecfr,government,mcp,regulations
11
+ Classifier: License :: OSI Approved :: MIT License
12
+ Classifier: Programming Language :: Python :: 3.11
13
+ Classifier: Programming Language :: Python :: 3.12
14
+ Classifier: Programming Language :: Python :: 3.13
15
+ Classifier: Topic :: Text Processing :: Indexing
16
+ Requires-Python: >=3.11
17
+ Requires-Dist: httpx>=0.27
18
+ Requires-Dist: lxml>=5.0
19
+ Requires-Dist: mcp>=2.0
20
+ Requires-Dist: pydantic>=2.0
21
+ Provides-Extra: dev
22
+ Requires-Dist: pytest-asyncio>=0.23; extra == 'dev'
23
+ Requires-Dist: pytest>=8.0; extra == 'dev'
24
+ Requires-Dist: respx>=0.21; extra == 'dev'
25
+ Requires-Dist: ruff>=0.6; extra == 'dev'
26
+ Description-Content-Type: text/markdown
27
+
28
+ # cfr-mcp
29
+
30
+ [![CI](https://github.com/mccallar/cfr-mcp/actions/workflows/ci.yml/badge.svg)](https://github.com/mccallar/cfr-mcp/actions/workflows/ci.yml)
31
+ [![PyPI](https://img.shields.io/pypi/v/cfr-mcp)](https://pypi.org/project/cfr-mcp/)
32
+
33
+ An MCP server that gives AI assistants access to the **US Code of Federal Regulations**.
34
+
35
+ Ask "what does 21 CFR 101.9 require?" or "has 40 CFR 261 changed since 2023?" and get
36
+ the actual regulation text, with citations, instead of a plausible-sounding guess.
37
+
38
+ Unofficial community project. Not affiliated with or endorsed by the Office of the
39
+ Federal Register, NARA, or GPO, and uses no government seals or logos.
40
+
41
+ ## Install
42
+
43
+ Requires Python 3.11+. No API key — the eCFR API is open.
44
+
45
+ Claude Code:
46
+
47
+ ```bash
48
+ claude mcp add cfr -- uvx cfr-mcp
49
+ ```
50
+
51
+ Any other MCP client:
52
+
53
+ ```json
54
+ {
55
+ "mcpServers": {
56
+ "cfr": {
57
+ "command": "uvx",
58
+ "args": ["cfr-mcp"]
59
+ }
60
+ }
61
+ }
62
+ ```
63
+
64
+ ## Tools
65
+
66
+ | Tool | What it does |
67
+ |---|---|
68
+ | `lookup_citation` | Text of a citation — `21 CFR 101.9`, `40 CFR 261.4(b)(1)`, `40 CFR Part 261 Subpart C` |
69
+ | `search_regulations` | Full-text search; returns citations and snippets, never bodies |
70
+ | `where_does_term_appear` | Which titles contain a term, with hit counts. Fetches no text |
71
+ | `browse_structure` | The hierarchy of a title or part, no text |
72
+ | `what_changed` | Amendment history and published corrections for a citation |
73
+ | `list_agencies` | Maps agency names to the CFR titles they administer |
74
+
75
+ Point-in-time works throughout: pass `date` as `YYYY-MM-DD` to read the CFR as it stood.
76
+
77
+ ## Design notes
78
+
79
+ **Context budget is the whole game.** The eCFR `full` endpoint returns an entire
80
+ downloadable XML document for a title-level request. Every tool here caps output and
81
+ degrades to an outline rather than dumping text into the model's context. Title-level
82
+ XML requests are refused outright.
83
+
84
+ **Dates must be resolved, not assumed.** Versioner routes 404 on dates that aren't
85
+ valid issue dates for a title, so the client resolves through `titles.json` first
86
+ rather than passing today's date blindly.
87
+
88
+ **Caching is courtesy.** The eCFR publishes no rate limit and has no key to identify
89
+ callers politely, so the client caches to disk (historical dates forever, since
90
+ point-in-time content is immutable) and self-limits concurrency. Set `CFR_MCP_CACHE_DIR`
91
+ to relocate the cache.
92
+
93
+ ## Legal
94
+
95
+ Regulation text is free to reproduce. **1 CFR 2.6** states that any person may reproduce
96
+ or republish material appearing in the Federal Register, with no restrictions on what is
97
+ reproduced, who reproduces it, or where. Federal government works are also outside
98
+ copyright under 17 U.S.C. §105.
99
+
100
+ **Incorporation by reference.** Some CFR sections incorporate private standards
101
+ (ASTM, NFPA, ASHRAE) whose copyright status after incorporation remains unsettled.
102
+ The eCFR does not contain the text of those standards and neither does this server —
103
+ it returns the citation only. Obtain standards from the issuing organization or the
104
+ Office of the Federal Register reading room.
105
+
106
+ **Status of the text.** The eCFR is authoritative but unofficial. Anyone relying on it
107
+ for legal research should verify against the current official CFR, the daily Federal
108
+ Register, and the List of CFR Sections Affected (LSA).
109
+
110
+ **Not legal advice.** This is a retrieval tool. It returns the text of regulations; it
111
+ does not tell you whether you are compliant with them.
112
+
113
+ ## Development
114
+
115
+ ```bash
116
+ uv sync --extra dev
117
+ uv run pytest
118
+ ```
119
+
120
+ MIT licensed.
cfr_mcp-0.1.0/PLAN.md ADDED
@@ -0,0 +1,172 @@
1
+ # PLAN.md — cfr-mcp: scaffold → published v0.1
2
+
3
+ Execution plan for Claude Code. Work phase by phase; do not start a phase until
4
+ the previous phase's acceptance criteria pass. Commit at the end of each phase.
5
+
6
+ ## Context
7
+
8
+ The repo contains a working scaffold (~700 lines):
9
+
10
+ - `src/cfr_mcp/citations.py` — CFR citation parser. **Tested, 20 cases passing.**
11
+ - `src/cfr_mcp/client.py` — async httpx client. All endpoint paths in the
12
+ `ENDPOINTS` dict. **Never run against the live API.**
13
+ - `src/cfr_mcp/xml_parse.py` — eCFR XML → text, with size capping.
14
+ **Written from documented conventions, never seen real XML.**
15
+ - `src/cfr_mcp/cache.py` — disk cache. Historical dates cached forever.
16
+ - `src/cfr_mcp/server.py` — FastMCP server, six tools. **JSON key names for
17
+ search/structure/versions/agencies responses are educated guesses.**
18
+ - `tests/test_citations.py` — parser tests (pytest).
19
+
20
+ The API: base `https://www.ecfr.gov`, no key. Services: admin
21
+ (`/api/admin/v1/...`), search (`/api/search/v1/...`), versioner
22
+ (`/api/versioner/v1/...`). Full endpoint list is in `client.py`.
23
+
24
+ ## Invariants — never violate these
25
+
26
+ 1. **Never request title-level XML.** `full_xml` refuses `is_title_only`
27
+ citations; keep it that way. Some titles are hundreds of MB.
28
+ 2. **Every tool output is capped.** Large content degrades to an outline with
29
+ instructions to drill down. No tool may return unbounded text.
30
+ 3. **Dates are resolved, not assumed.** Versioner routes 404 on invalid issue
31
+ dates; always resolve current dates through `titles.json`.
32
+ 4. **Cache failures never break lookups.** Cache is best-effort.
33
+ 5. **Errors are returned as readable strings to the model**, not raised through
34
+ the MCP layer, so the assistant can self-correct (e.g. bad date → explain).
35
+ 6. **Retrieval only.** No tool or description may imply compliance judgment or
36
+ legal advice. Keep the source disclaimer on content-bearing outputs.
37
+
38
+ ## Phase 0 — Environment (15 min)
39
+
40
+ - [ ] `git init`, initial commit of the scaffold as-is.
41
+ - [ ] `uv sync --extra dev`
42
+ - [ ] `uv run pytest` → all citation tests pass.
43
+ - [ ] Add MIT `LICENSE` file (year, author name).
44
+ - [ ] In `client.py`, replace `YOURNAME`/`YOUR_EMAIL` in `USER_AGENT`
45
+ (ask the human for the GitHub username and contact email).
46
+
47
+ **Accept:** pytest green, clean `git status`.
48
+
49
+ ## Phase 1 — Verify API reality, capture fixtures (1–2 h)
50
+
51
+ Fetch real responses and save them under `tests/fixtures/`. Use small targets.
52
+
53
+ - [ ] `GET /api/versioner/v1/titles.json` → `fixtures/titles.json`.
54
+ Confirm field names `number`, `latest_issue_date`, `up_to_date_as_of`.
55
+ Fix `latest_date_for_title()` if they differ.
56
+ - [ ] `GET /api/versioner/v1/full/{date}/title-1.xml?part=2&section=2.6`
57
+ (Title 1 is tiny) → `fixtures/section_1_2_6.xml`.
58
+ - [ ] Same for a mid-size section: `title-21 ... part=101&section=101.9`
59
+ → `fixtures/section_21_101_9.xml`.
60
+ - [ ] A whole small part: `title-1 ... part=2` → `fixtures/part_1_2.xml`.
61
+ - [ ] An appendix and a subpart request; confirm param names
62
+ (`appendix`, `subpart`) are what the API expects. Fix
63
+ `Citation.as_params()` if not.
64
+ - [ ] `GET /api/search/v1/results?query=nutrition+labeling&per_page=3`
65
+ → `fixtures/search_results.json`. Record the real shape of results,
66
+ hierarchy fields, excerpt field, meta/total.
67
+ - [ ] `GET /api/search/v1/counts/hierarchy?query=asbestos`
68
+ → `fixtures/counts_hierarchy.json`.
69
+ - [ ] `GET /api/versioner/v1/structure/{date}/title-1.json`
70
+ → `fixtures/structure_title1.json`.
71
+ - [ ] `GET /api/versioner/v1/versions/title-1.json` → `fixtures/versions.json`.
72
+ Confirm filter param names (`conditions[part]`,
73
+ `conditions[issue_date][gte]`) actually filter; note whether the
74
+ version entries include `amendment_date`, `issue_date`, `identifier`.
75
+ - [ ] `GET /api/admin/v1/agencies.json` → `fixtures/agencies.json`.
76
+ - [ ] `GET /api/admin/v1/corrections/title/1.json` → `fixtures/corrections.json`.
77
+ - [ ] Write `tests/fixtures/NOTES.md` documenting every place the real
78
+ response differs from what the code assumes.
79
+
80
+ **Accept:** all fixtures on disk; NOTES.md lists discrepancies (may be empty).
81
+
82
+ ## Phase 2 — Make the code match reality (2–4 h)
83
+
84
+ - [ ] Fix `xml_parse._build()` against the XML fixtures. Verify: DIV nesting,
85
+ `TYPE`/`N` attributes, `HEAD` headings, paragraph elements, and that
86
+ heading number-stripping works on real headings.
87
+ - [ ] Verify `extract_paragraphs` finds `(b)(1)` trails in real section text
88
+ (fixture 21 CFR 101.9 has deep paragraph nesting — ideal test).
89
+ - [ ] Fix every guessed JSON key in `server.py` rendering loops
90
+ (search results, structure walk, versions, agencies, corrections).
91
+ - [ ] Write fixture-backed tests (use `respx` to mock httpx against fixtures):
92
+ - `tests/test_xml_parse.py` — parse each XML fixture; assert headings,
93
+ section numbers, non-empty text; assert `render_capped` degrades to an
94
+ outline when `max_chars` is tiny.
95
+ - `tests/test_client.py` — date resolution from titles fixture; 404 →
96
+ readable `ECFRError`; title-level XML refusal; cache hit skips HTTP.
97
+ - `tests/test_tools.py` — call each of the six tools with mocked
98
+ transport; assert output is a string, contains expected citation,
99
+ and stays under the cap.
100
+ - [ ] `uv run ruff check --fix .` and resolve remaining lint.
101
+
102
+ **Accept:** full pytest suite green offline (fixtures only, no network).
103
+
104
+ ## Phase 3 — Live integration with Claude Code (1 h)
105
+
106
+ - [ ] Register locally: `claude mcp add cfr -- uv run cfr-mcp`
107
+ (or `uv run mcp dev src/cfr_mcp/server.py` for the inspector).
108
+ - [ ] Exercise each tool through the assistant and record results in
109
+ `docs/manual-test-log.md`:
110
+ 1. "What does 21 CFR 101.9(c) require?" → correct paragraph text.
111
+ 2. "Search the CFR for 'per- and polyfluoroalkyl'" → citations + snippets.
112
+ 3. "Where does 'asbestos' appear in the CFR?" → hierarchy counts.
113
+ 4. "Show the structure of 40 CFR Part 261" → outline, no text dump.
114
+ 5. "Has 40 CFR 261.4 changed since 2023-01-01?" → amendment dates.
115
+ 6. "Which CFR titles does the EPA administer?" → Title 40 (+ others).
116
+ 7. Adversarial: "Give me all of Title 40" → graceful refusal with
117
+ guidance, not an error or a dump.
118
+ 8. A paragraph that doesn't exist: "21 CFR 101.9(z)(9)" → falls back to
119
+ section text, does not fabricate.
120
+ - [ ] Fix anything awkward the model trips on (tool descriptions are part of
121
+ the product — iterate on them here).
122
+
123
+ **Accept:** all eight logged with sane transcripts.
124
+
125
+ ## Phase 4 — Hardening (1–2 h)
126
+
127
+ - [ ] Title 35 is reserved: `lookup_citation("35 CFR 1.1")` must return a
128
+ clear "reserved title" message. Add test.
129
+ - [ ] Very large part (e.g. 40 CFR 63): confirm outline degradation and
130
+ acceptable latency; consider streaming/size guard on the raw download
131
+ if response exceeds ~5 MB.
132
+ - [ ] Date edge cases: pre-2017 dates (point-in-time floor), future dates,
133
+ malformed dates — each returns a helpful message. Tests.
134
+ - [ ] Concurrency: fire 10 parallel `lookup_citation` calls; semaphore holds,
135
+ nothing corrupts the cache (meta/body write order).
136
+ - [ ] Add `--version` / `-h` handling to `main()`.
137
+
138
+ **Accept:** suite green, edge cases covered.
139
+
140
+ ## Phase 5 — Publish (1–2 h; human-in-the-loop)
141
+
142
+ - [ ] README final pass: verify install block, tool table, legal section
143
+ (1 CFR 2.6, IBR caveat, "authoritative but unofficial" disclaimer,
144
+ no seals/no-endorsement note).
145
+ - [ ] Create GitHub repo (human), push, add topics: `mcp`, `ecfr`, `cfr`,
146
+ `regulations`, `model-context-protocol`.
147
+ - [ ] CI: GitHub Actions workflow running ruff + pytest on 3.11/3.12/3.13.
148
+ - [ ] `uv build`; test the wheel in a clean venv: `uvx --from dist/*.whl cfr-mcp`
149
+ starts and responds to an MCP `initialize`.
150
+ - [ ] Publish: human creates PyPI account + token; `uv publish`.
151
+ - [ ] Verify end-to-end: `uvx cfr-mcp` from PyPI works in Claude Code.
152
+ - [ ] Submit to the MCP registry / community server lists (human approves
153
+ the listing text).
154
+ - [ ] Tag `v0.1.0`, GitHub release with a short changelog.
155
+
156
+ **Accept:** a stranger can go from README to a working `lookup_citation`
157
+ call in under five minutes.
158
+
159
+ ## Explicitly out of scope for v0.1
160
+
161
+ - Hosted/remote transport (SSE/HTTP) — local stdio only.
162
+ - Any paid features, alerting, or UI.
163
+ - Fetching text of standards incorporated by reference — permanently out;
164
+ return citations only.
165
+ - State regulations, Federal Register documents beyond corrections.
166
+
167
+ ## Backlog for v0.2 (do not build now)
168
+
169
+ - Federal Register cross-links in `what_changed` (which rule caused the change).
170
+ - Rendered side-by-side diffs between two dates.
171
+ - `search_suggestions` tool using `/api/search/v1/suggestions`.
172
+ - Prebuilt vertical part-bundles (food labeling, hazmat) as MCP resources.
@@ -0,0 +1,93 @@
1
+ # cfr-mcp
2
+
3
+ [![CI](https://github.com/mccallar/cfr-mcp/actions/workflows/ci.yml/badge.svg)](https://github.com/mccallar/cfr-mcp/actions/workflows/ci.yml)
4
+ [![PyPI](https://img.shields.io/pypi/v/cfr-mcp)](https://pypi.org/project/cfr-mcp/)
5
+
6
+ An MCP server that gives AI assistants access to the **US Code of Federal Regulations**.
7
+
8
+ Ask "what does 21 CFR 101.9 require?" or "has 40 CFR 261 changed since 2023?" and get
9
+ the actual regulation text, with citations, instead of a plausible-sounding guess.
10
+
11
+ Unofficial community project. Not affiliated with or endorsed by the Office of the
12
+ Federal Register, NARA, or GPO, and uses no government seals or logos.
13
+
14
+ ## Install
15
+
16
+ Requires Python 3.11+. No API key — the eCFR API is open.
17
+
18
+ Claude Code:
19
+
20
+ ```bash
21
+ claude mcp add cfr -- uvx cfr-mcp
22
+ ```
23
+
24
+ Any other MCP client:
25
+
26
+ ```json
27
+ {
28
+ "mcpServers": {
29
+ "cfr": {
30
+ "command": "uvx",
31
+ "args": ["cfr-mcp"]
32
+ }
33
+ }
34
+ }
35
+ ```
36
+
37
+ ## Tools
38
+
39
+ | Tool | What it does |
40
+ |---|---|
41
+ | `lookup_citation` | Text of a citation — `21 CFR 101.9`, `40 CFR 261.4(b)(1)`, `40 CFR Part 261 Subpart C` |
42
+ | `search_regulations` | Full-text search; returns citations and snippets, never bodies |
43
+ | `where_does_term_appear` | Which titles contain a term, with hit counts. Fetches no text |
44
+ | `browse_structure` | The hierarchy of a title or part, no text |
45
+ | `what_changed` | Amendment history and published corrections for a citation |
46
+ | `list_agencies` | Maps agency names to the CFR titles they administer |
47
+
48
+ Point-in-time works throughout: pass `date` as `YYYY-MM-DD` to read the CFR as it stood.
49
+
50
+ ## Design notes
51
+
52
+ **Context budget is the whole game.** The eCFR `full` endpoint returns an entire
53
+ downloadable XML document for a title-level request. Every tool here caps output and
54
+ degrades to an outline rather than dumping text into the model's context. Title-level
55
+ XML requests are refused outright.
56
+
57
+ **Dates must be resolved, not assumed.** Versioner routes 404 on dates that aren't
58
+ valid issue dates for a title, so the client resolves through `titles.json` first
59
+ rather than passing today's date blindly.
60
+
61
+ **Caching is courtesy.** The eCFR publishes no rate limit and has no key to identify
62
+ callers politely, so the client caches to disk (historical dates forever, since
63
+ point-in-time content is immutable) and self-limits concurrency. Set `CFR_MCP_CACHE_DIR`
64
+ to relocate the cache.
65
+
66
+ ## Legal
67
+
68
+ Regulation text is free to reproduce. **1 CFR 2.6** states that any person may reproduce
69
+ or republish material appearing in the Federal Register, with no restrictions on what is
70
+ reproduced, who reproduces it, or where. Federal government works are also outside
71
+ copyright under 17 U.S.C. §105.
72
+
73
+ **Incorporation by reference.** Some CFR sections incorporate private standards
74
+ (ASTM, NFPA, ASHRAE) whose copyright status after incorporation remains unsettled.
75
+ The eCFR does not contain the text of those standards and neither does this server —
76
+ it returns the citation only. Obtain standards from the issuing organization or the
77
+ Office of the Federal Register reading room.
78
+
79
+ **Status of the text.** The eCFR is authoritative but unofficial. Anyone relying on it
80
+ for legal research should verify against the current official CFR, the daily Federal
81
+ Register, and the List of CFR Sections Affected (LSA).
82
+
83
+ **Not legal advice.** This is a retrieval tool. It returns the text of regulations; it
84
+ does not tell you whether you are compliant with them.
85
+
86
+ ## Development
87
+
88
+ ```bash
89
+ uv sync --extra dev
90
+ uv run pytest
91
+ ```
92
+
93
+ MIT licensed.
@@ -0,0 +1,100 @@
1
+ # Manual test log — Phase 3 live integration
2
+
3
+ 2026-08-29, against the live eCFR API. Tools were exercised at the tool-call
4
+ surface (the exact async functions MCP dispatches to), with arguments chosen
5
+ as an assistant would choose them. Outputs below are summarized; sizes are
6
+ exact.
7
+
8
+ ## 1. "What does 21 CFR 101.9(c) require?"
9
+
10
+ `lookup_citation("21 CFR 101.9(c)")`
11
+
12
+ - First run returned the full (c) block: correct text, but **35,281 chars —
13
+ violated the output cap invariant**. Fixed: an oversized paragraph now
14
+ returns its opening text plus a sub-paragraph list. Re-run: 3,340 chars,
15
+ begins with the correct "(c) The declaration of nutrition information…"
16
+ text and ends "Sub-paragraphs available: (1)…(9). Request a deeper
17
+ citation like 101.9(c)(1)". ✅
18
+
19
+ ## 2. "Search the CFR for 'per- and polyfluoroalkyl'"
20
+
21
+ `search_regulations("per- and polyfluoroalkyl", limit=5)`
22
+
23
+ - 1,268 chars. "72 result(s)… showing 5". Real citations (40 CFR 141.901,
24
+ 705.1, 372.29, 705.3) with headings and clean snippets, no HTML. ✅
25
+
26
+ ## 3. "Where does 'asbestos' appear in the CFR?"
27
+
28
+ `where_does_term_appear("asbestos")`
29
+
30
+ - 4,616 chars. Title-level hit counts with top-3 parts nested under each
31
+ (e.g. Title 16 → Part 1304 Ban of Consumer Patching Compounds…). First
32
+ run leaked `<strong>` tags inside part headings (the counts endpoint
33
+ highlights the query term); fixed with the same HTML stripping used for
34
+ search. ✅
35
+
36
+ ## 4. "Show the structure of 40 CFR Part 261"
37
+
38
+ `browse_structure(40, part="261")`
39
+
40
+ - 6,378 chars. Subparts A–DD with section lines, appendices I–IX at the
41
+ bottom, zero body text. ✅
42
+
43
+ ## 5. "Has 40 CFR 261.4 changed since 2023-01-01?"
44
+
45
+ `what_changed("40 CFR 261.4", since="2023-01-01")`
46
+
47
+ - 1,358 chars. Eleven amendment entries 2023-03-29 → 2025-03-21 with
48
+ amendment and issue dates, plus part-261 published corrections with FR
49
+ citations. Filter params verified on the wire
50
+ (`part=261&issue_date[gte]=2023-01-01`). ✅
51
+
52
+ ## 6. "Which CFR titles does the EPA administer?"
53
+
54
+ `list_agencies("environmental protection")`
55
+
56
+ - 93 chars: "Environmental Protection Agency
57
+ [environmental-protection-agency] — Title(s) 2, 5, 40, 41, 48". First run
58
+ sorted titles as strings ("2, 40, 41, 48, 5"); fixed to numeric sort. ✅
59
+
60
+ ## 7. Adversarial: "Give me all of Title 40"
61
+
62
+ `lookup_citation("Title 40")`
63
+
64
+ - 124 chars, zero network calls: "'Title 40' names a whole CFR title, which
65
+ is far too large to retrieve. Use browse_structure to navigate it, or
66
+ name a part." Graceful refusal with guidance. ✅
67
+
68
+ ## 8. Nonexistent paragraph: "21 CFR 101.9(z)(9)"
69
+
70
+ `lookup_citation("21 CFR 101.9(z)(9)")`
71
+
72
+ - First run fell back to a degenerate outline (a childless section's
73
+ "outline" was just its heading, twice) and never said the paragraph was
74
+ missing. Fixed: now 192 chars — "Paragraph (z)(9) does not exist in
75
+ 21 CFR 101.9. Top-level paragraphs present: (a)…(l). Request the whole
76
+ section or one of those paragraphs." No fabrication. ✅
77
+
78
+ ## Fixes made during this phase
79
+
80
+ 1. Output cap now applies to paragraph-narrowed text (scenario 1); oversized
81
+ paragraphs degrade to intro + sub-paragraph list.
82
+ 2. `where_does_term_appear` strips HTML from headings (scenario 3).
83
+ 3. `list_agencies` sorts titles numerically (scenario 6).
84
+ 4. Missing paragraphs get an explicit not-found message listing the
85
+ paragraphs that do exist (scenario 8).
86
+ 5. `render_capped` on a huge childless section now outlines its top-level
87
+ paragraphs instead of repeating the heading.
88
+
89
+ All five have regression tests in `tests/`.
90
+
91
+ ## Phase 4 live checks
92
+
93
+ - `lookup_citation("40 CFR Part 63")` (one of the largest CFR parts): the
94
+ streaming byte guard aborted the download at 20 MB after 18.1s with
95
+ "Request a smaller unit — a subpart or a section — or use
96
+ browse_structure to navigate this part." No multi-hundred-MB download,
97
+ no dump. ✅
98
+ - `lookup_citation("40 CFR Part 261")` (mid-size, 1.5M chars of text,
99
+ ~6 MB XML): 1.7s, degraded to an 8,009-char subpart/section outline with
100
+ per-node sizes. ✅
@@ -0,0 +1,47 @@
1
+ [project]
2
+ name = "cfr-mcp"
3
+ version = "0.1.0"
4
+ description = "MCP server for the US Code of Federal Regulations (eCFR)"
5
+ readme = "README.md"
6
+ requires-python = ">=3.11"
7
+ license = { text = "MIT" }
8
+ keywords = ["mcp", "cfr", "ecfr", "regulations", "government", "compliance"]
9
+ classifiers = [
10
+ "License :: OSI Approved :: MIT License",
11
+ "Programming Language :: Python :: 3.11",
12
+ "Programming Language :: Python :: 3.12",
13
+ "Programming Language :: Python :: 3.13",
14
+ "Topic :: Text Processing :: Indexing",
15
+ ]
16
+ dependencies = [
17
+ "mcp>=2.0",
18
+ "httpx>=0.27",
19
+ "lxml>=5.0",
20
+ "pydantic>=2.0",
21
+ ]
22
+
23
+ [project.optional-dependencies]
24
+ dev = ["pytest>=8.0", "pytest-asyncio>=0.23", "respx>=0.21", "ruff>=0.6"]
25
+
26
+ [project.scripts]
27
+ cfr-mcp = "cfr_mcp.server:main"
28
+
29
+ [project.urls]
30
+ Homepage = "https://github.com/mccallar/cfr-mcp"
31
+ Repository = "https://github.com/mccallar/cfr-mcp"
32
+ Issues = "https://github.com/mccallar/cfr-mcp/issues"
33
+
34
+ [build-system]
35
+ requires = ["hatchling"]
36
+ build-backend = "hatchling.build"
37
+
38
+ [tool.hatch.build.targets.wheel]
39
+ packages = ["src/cfr_mcp"]
40
+
41
+ [tool.pytest.ini_options]
42
+ pythonpath = ["src"]
43
+ testpaths = ["tests"]
44
+ asyncio_mode = "auto"
45
+
46
+ [tool.ruff]
47
+ line-length = 100
File without changes
@@ -0,0 +1,78 @@
1
+ """Disk cache for eCFR responses.
2
+
3
+ eCFR updates at most daily, and historical point-in-time content never changes
4
+ at all. Caching is the whole reason this server can be a good citizen against
5
+ an API that publishes no rate limit and has no key to identify us politely.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ import hashlib
11
+ import json
12
+ import os
13
+ import time
14
+ from pathlib import Path
15
+ from typing import Any
16
+
17
+ DEFAULT_TTL = 24 * 60 * 60 # eCFR updates daily
18
+
19
+
20
+ def _default_dir() -> Path:
21
+ env = os.environ.get("CFR_MCP_CACHE_DIR")
22
+ if env:
23
+ return Path(env)
24
+ base = os.environ.get("XDG_CACHE_HOME") or (Path.home() / ".cache")
25
+ return Path(base) / "cfr-mcp"
26
+
27
+
28
+ class Cache:
29
+ def __init__(self, directory: Path | None = None, ttl: int = DEFAULT_TTL) -> None:
30
+ self.dir = directory or _default_dir()
31
+ self.ttl = ttl
32
+ self.dir.mkdir(parents=True, exist_ok=True)
33
+
34
+ def key(self, path: str, params: dict[str, Any] | None = None) -> str:
35
+ blob = path + "?" + json.dumps(params or {}, sort_keys=True)
36
+ return hashlib.sha256(blob.encode()).hexdigest()[:32]
37
+
38
+ def _paths(self, key: str) -> tuple[Path, Path]:
39
+ return self.dir / f"{key}.body", self.dir / f"{key}.meta"
40
+
41
+ def get(self, key: str) -> str | None:
42
+ body_path, meta_path = self._paths(key)
43
+ if not (body_path.exists() and meta_path.exists()):
44
+ return None
45
+ try:
46
+ meta = json.loads(meta_path.read_text())
47
+ except (json.JSONDecodeError, OSError):
48
+ return None
49
+ if (
50
+ not meta.get("immutable")
51
+ and time.time() - meta.get("stored_at", 0) > self.ttl
52
+ ):
53
+ return None
54
+ try:
55
+ return body_path.read_text(encoding="utf-8")
56
+ except OSError:
57
+ return None
58
+
59
+ def set(self, key: str, body: str, *, immutable: bool = False) -> None:
60
+ body_path, meta_path = self._paths(key)
61
+ try:
62
+ # Write body first; a torn write leaves no meta and reads as a miss.
63
+ body_path.write_text(body, encoding="utf-8")
64
+ meta_path.write_text(
65
+ json.dumps({"stored_at": time.time(), "immutable": immutable})
66
+ )
67
+ except OSError:
68
+ pass # cache failures must never break a lookup
69
+
70
+ def clear(self) -> int:
71
+ removed = 0
72
+ for f in self.dir.glob("*.*"):
73
+ try:
74
+ f.unlink()
75
+ removed += 1
76
+ except OSError:
77
+ pass
78
+ return removed // 2