contentdm-mcp 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (77) hide show
  1. contentdm_mcp-0.1.0/.env.example +34 -0
  2. contentdm_mcp-0.1.0/.gitignore +39 -0
  3. contentdm_mcp-0.1.0/CHANGELOG.md +52 -0
  4. contentdm_mcp-0.1.0/CONTRIBUTING.md +117 -0
  5. contentdm_mcp-0.1.0/LICENSE +21 -0
  6. contentdm_mcp-0.1.0/PKG-INFO +318 -0
  7. contentdm_mcp-0.1.0/README.md +281 -0
  8. contentdm_mcp-0.1.0/SECURITY.md +80 -0
  9. contentdm_mcp-0.1.0/docs/API-NOTES.md +230 -0
  10. contentdm_mcp-0.1.0/docs/DESIGN.md +198 -0
  11. contentdm_mcp-0.1.0/pyproject.toml +83 -0
  12. contentdm_mcp-0.1.0/server.json +52 -0
  13. contentdm_mcp-0.1.0/src/contentdm_mcp/__init__.py +11 -0
  14. contentdm_mcp-0.1.0/src/contentdm_mcp/__main__.py +13 -0
  15. contentdm_mcp-0.1.0/src/contentdm_mcp/adapters/__init__.py +54 -0
  16. contentdm_mcp-0.1.0/src/contentdm_mcp/adapters/base.py +186 -0
  17. contentdm_mcp-0.1.0/src/contentdm_mcp/adapters/classic.py +476 -0
  18. contentdm_mcp-0.1.0/src/contentdm_mcp/client.py +811 -0
  19. contentdm_mcp-0.1.0/src/contentdm_mcp/config.py +159 -0
  20. contentdm_mcp-0.1.0/src/contentdm_mcp/instances.py +393 -0
  21. contentdm_mcp-0.1.0/src/contentdm_mcp/instances.yaml +422 -0
  22. contentdm_mcp-0.1.0/src/contentdm_mcp/server.py +1179 -0
  23. contentdm_mcp-0.1.0/src/contentdm_mcp/shape.py +161 -0
  24. contentdm_mcp-0.1.0/tests/__init__.py +0 -0
  25. contentdm_mcp-0.1.0/tests/conftest.py +263 -0
  26. contentdm_mcp-0.1.0/tests/fixtures/al_bad_alias.html +1 -0
  27. contentdm_mcp-0.1.0/tests/fixtures/al_collection_list.json +50 -0
  28. contentdm_mcp-0.1.0/tests/fixtures/al_compound_16539.json +70 -0
  29. contentdm_mcp-0.1.0/tests/fixtures/al_getparent_page.json +3 -0
  30. contentdm_mcp-0.1.0/tests/fixtures/al_getparent_top.json +3 -0
  31. contentdm_mcp-0.1.0/tests/fixtures/al_iiif_400_pdf_page.txt +6 -0
  32. contentdm_mcp-0.1.0/tests/fixtures/al_iiif_501_audio.txt +6 -0
  33. contentdm_mcp-0.1.0/tests/fixtures/al_iiif_info_16527_pdf.json +104 -0
  34. contentdm_mcp-0.1.0/tests/fixtures/al_item_16538_transcript.json +44 -0
  35. contentdm_mcp-0.1.0/tests/fixtures/al_item_16539_compound.json +44 -0
  36. contentdm_mcp-0.1.0/tests/fixtures/al_item_570_audio.json +41 -0
  37. contentdm_mcp-0.1.0/tests/fixtures/al_item_not_found.json +5 -0
  38. contentdm_mcp-0.1.0/tests/fixtures/al_not_compound.json +4 -0
  39. contentdm_mcp-0.1.0/tests/fixtures/al_query_all_empty.json +8 -0
  40. contentdm_mcp-0.1.0/tests/fixtures/al_query_voices_semmes.json +39 -0
  41. contentdm_mcp-0.1.0/tests/fixtures/al_records_fields.json +467 -0
  42. contentdm_mcp-0.1.0/tests/fixtures/al_voices_16527_p2_64px.jpg +0 -0
  43. contentdm_mcp-0.1.0/tests/fixtures/al_voices_fields.json +512 -0
  44. contentdm_mcp-0.1.0/tests/fixtures/ga_collection_list.json +20 -0
  45. contentdm_mcp-0.1.0/tests/fixtures/ga_deaths_fields.json +497 -0
  46. contentdm_mcp-0.1.0/tests/fixtures/ga_item_300730_death.json +43 -0
  47. contentdm_mcp-0.1.0/tests/fixtures/mo_cohist_fields.json +407 -0
  48. contentdm_mcp-0.1.0/tests/fixtures/mo_collection_list.json +8 -0
  49. contentdm_mcp-0.1.0/tests/fixtures/mo_compound_96244_monograph.json +603 -0
  50. contentdm_mcp-0.1.0/tests/fixtures/oh_collection_list.json +20 -0
  51. contentdm_mcp-0.1.0/tests/fixtures/oh_exponent_fields.json +512 -0
  52. contentdm_mcp-0.1.0/tests/fixtures/oh_item_21598_newspaper_page.json +44 -0
  53. contentdm_mcp-0.1.0/tests/fixtures/sd_collection_list_empty.json +1 -0
  54. contentdm_mcp-0.1.0/tests/fixtures/tn_collection_list.json +20 -0
  55. contentdm_mcp-0.1.0/tests/fixtures/tn_compound_1206616.json +2340 -0
  56. contentdm_mcp-0.1.0/tests/fixtures/tn_deaths_fields.json +467 -0
  57. contentdm_mcp-0.1.0/tests/fixtures/tn_getparent_page.json +3 -0
  58. contentdm_mcp-0.1.0/tests/fixtures/tn_iiif_info_1206609.json +87 -0
  59. contentdm_mcp-0.1.0/tests/fixtures/tn_item_1206609_page.json +41 -0
  60. contentdm_mcp-0.1.0/tests/fixtures/tn_item_1206616_parent.json +41 -0
  61. contentdm_mcp-0.1.0/tests/fixtures/tn_query_in_compound.json +17 -0
  62. contentdm_mcp-0.1.0/tests/fixtures/tn_query_york_objects.json +49 -0
  63. contentdm_mcp-0.1.0/tests/fixtures/tn_query_york_pages.json +49 -0
  64. contentdm_mcp-0.1.0/tests/fixtures/tool_schema.json +57 -0
  65. contentdm_mcp-0.1.0/tests/live_check.py +234 -0
  66. contentdm_mcp-0.1.0/tests/regen_tool_snapshot.py +29 -0
  67. contentdm_mcp-0.1.0/tests/test_classic.py +302 -0
  68. contentdm_mcp-0.1.0/tests/test_client.py +316 -0
  69. contentdm_mcp-0.1.0/tests/test_config.py +120 -0
  70. contentdm_mcp-0.1.0/tests/test_entrypoint.py +55 -0
  71. contentdm_mcp-0.1.0/tests/test_instances.py +268 -0
  72. contentdm_mcp-0.1.0/tests/test_registry_listing.py +62 -0
  73. contentdm_mcp-0.1.0/tests/test_security.py +590 -0
  74. contentdm_mcp-0.1.0/tests/test_server.py +422 -0
  75. contentdm_mcp-0.1.0/tests/test_shape.py +108 -0
  76. contentdm_mcp-0.1.0/tests/test_tool_contract.py +229 -0
  77. contentdm_mcp-0.1.0/uv.lock +985 -0
@@ -0,0 +1,34 @@
1
+ # Copy to .env to change a default. Nothing here is required: CONTENTdm's web
2
+ # services need no key and no account.
3
+ #
4
+ # The server reads .env from the directory it is started in -- the
5
+ # repository root, when launched with `uv --directory /path/to/contentdm-mcp`.
6
+ # Variables already set in the environment take precedence.
7
+
8
+ # Optional: an email address or URL added to the User-Agent, so an
9
+ # institution can reach you if your use ever causes it trouble.
10
+ # CONTENTDM_CONTACT=you@example.org
11
+
12
+ # Optional: directory for the on-disk response cache.
13
+ # CONTENTDM_CACHE_DIR=~/.cache/contentdm-mcp
14
+
15
+ # Optional: HTTP timeout in seconds for one request (default 30; image
16
+ # downloads get 120 to read).
17
+ # CONTENTDM_TIMEOUT=30
18
+
19
+ # Optional: least seconds between two requests to one site (default 1, never
20
+ # below 0.5).
21
+ # CONTENTDM_MIN_INTERVAL=1
22
+
23
+ # Optional: more CONTENTdm sites a tool may reach. Without this, a tool reaches
24
+ # the curated sites and any address under contentdm.oclc.org, and nothing
25
+ # else. Either https base URLs separated by commas, or the absolute path of a
26
+ # YAML file whose entries are shaped like instances.yaml's (only base_url is
27
+ # required). A model cannot add to this list; only you can.
28
+ # CONTENTDM_EXTRA_INSTANCES=https://digital.example.edu
29
+ # CONTENTDM_EXTRA_INSTANCES=/Users/you/contentdm-sites.yaml
30
+
31
+ # Optional: an existing folder that every get_image destination must lie
32
+ # inside. get_image never writes under ~/Library otherwise; set this to save
33
+ # into an iCloud Drive folder, which lives there.
34
+ # CONTENTDM_DOWNLOAD_DIR="~/Library/Mobile Documents/com~apple~CloudDocs/Genealogy"
@@ -0,0 +1,39 @@
1
+ # Settings. .env holds local overrides; .env.example is the committed template.
2
+ .env
3
+ .env.*
4
+ !.env.example
5
+ .envrc
6
+
7
+ # Response cache
8
+ .cache/
9
+
10
+ # Downloaded page images are never part of this repository. The one
11
+ # recorded test image is the exception.
12
+ *.jpg
13
+ !tests/fixtures/*.jpg
14
+ *.jpeg
15
+ *.png
16
+ *.tif
17
+ *.tiff
18
+ *.pdf
19
+
20
+ # Python
21
+ __pycache__/
22
+ *.py[cod]
23
+ .venv/
24
+ venv/
25
+ build/
26
+ dist/
27
+ *.egg-info/
28
+
29
+ # Tool caches
30
+ .pytest_cache/
31
+ .ruff_cache/
32
+ .mypy_cache/
33
+ .coverage
34
+ htmlcov/
35
+
36
+ # Editors and OS
37
+ .vscode/
38
+ .idea/
39
+ .DS_Store
@@ -0,0 +1,52 @@
1
+ # Changelog
2
+
3
+ All notable changes to this project are recorded here. The format follows
4
+ [Keep a Changelog](https://keepachangelog.com/en/1.1.0/), and the project uses
5
+ [semantic versioning](https://semver.org/). The tool surface is the public
6
+ interface: renaming or removing a tool or a parameter is a major release, and
7
+ adding one is a minor release. Before 1.0, a minor release may do either.
8
+ Changes to `instances.yaml` alone (a site added, moved or re-checked) are
9
+ patch releases.
10
+
11
+ ## [Unreleased]
12
+
13
+ ## [0.1.0] — 2026-10-06
14
+
15
+ First release.
16
+
17
+ ### Added
18
+
19
+ - Eight tools over classic CONTENTdm's `dmwebservices` API and IIIF image
20
+ service: `list_instances`, `list_collections`, `get_collection`, `search`,
21
+ `get_item`, `get_pages`, `get_image` (writes one new file, never
22
+ overwrites) and `cache_status`.
23
+ - A host policy for arguments a model writes: a tool reaches the curated
24
+ sites, any address under `contentdm.oclc.org` (where every classic site
25
+ answers as `cdmNNNNN`), and sites the operator lists in
26
+ `CONTENTDM_EXTRA_INSTANCES`. Any other host is refused before it is looked
27
+ up, with a message naming both ways forward.
28
+ - Every connection checked where it is made: a name that leads to a private,
29
+ loopback, link-local, CGNAT, multicast, reserved or unspecified address,
30
+ IPv4 or IPv6, is refused, and the connection goes to the address checked,
31
+ so redirects and DNS rebinding are covered. JSON answers are capped at
32
+ 10 MB and images at 60 MB as they stream.
33
+ - `get_image` saves only a real image of the format its `.jpg` or `.jpeg`
34
+ name says, never a hidden file or one under `~/Library`, and only inside
35
+ `CONTENTDM_DOWNLOAD_DIR` when that is set.
36
+ - A curated `instances.yaml` of 18 state archive, state library and
37
+ university sites holding state or county records, each verified live on
38
+ 2026-10-06, and 7 entries for institutions that have left CONTENTdm
39
+ (North Carolina to Quartex, Pennsylvania's POWER Library to Islandora,
40
+ South Dakota and Idaho to Preservica, Montana to Recollect, Wisconsin and
41
+ Arizona to their own sites).
42
+ - An adapter layer between the tools and the API, so the new CONTENTdm or
43
+ Quartex can be added without changing the tools.
44
+ - `search` across several sites side by side, reporting which answered,
45
+ failed, timed out or were not searched; `page_hits` for the page inside a
46
+ volume where a text match is.
47
+ - Every result carries the item's public address and a citation core
48
+ (institution, collection, title, identifier, address); `get_item` adds the
49
+ institution's own "cite as" line and rights statement where it has them.
50
+ - A polite client: one request at a time per site at least a second apart,
51
+ identical calls joined, answers cached (collection lists and fields 30
52
+ days, items 7, searches 1), one retry on 429 and 5xx.
@@ -0,0 +1,117 @@
1
+ # Contributing
2
+
3
+ Issues and pull requests are welcome. This file says how the project is put
4
+ together and what a change is expected to carry.
5
+
6
+ ## Setting up
7
+
8
+ ```bash
9
+ git clone https://github.com/ianderso/contentdm-mcp
10
+ cd contentdm-mcp
11
+ uv sync --extra dev
12
+ ```
13
+
14
+ Before sending a change, run what CI runs:
15
+
16
+ ```bash
17
+ uv run ruff check .
18
+ uv run ruff format --check .
19
+ uv run pytest
20
+ ```
21
+
22
+ The suite is mocked with [respx](https://lundberg.github.io/respx/) against
23
+ recorded CONTENTdm responses. It must never touch a live site: each one is an
24
+ institution's own server, and CI should not depend on any of them being up.
25
+
26
+ ## Where things live
27
+
28
+ | Path | What it holds |
29
+ | --- | --- |
30
+ | `src/contentdm_mcp/server.py` | The tools. Their docstrings and `Field` descriptions *are* the published tool descriptions and schema. |
31
+ | `src/contentdm_mcp/adapters/base.py` | The interface every platform adapter provides, and the records it returns. |
32
+ | `src/contentdm_mcp/adapters/classic.py` | Classic CONTENTdm: the `dmwebservices` calls, IIIF, and item addresses. |
33
+ | `src/contentdm_mcp/client.py` | The cached, paced HTTP client, the host allowlist, the redirect rule, the public-address check made at connect time, and the size caps. |
34
+ | `src/contentdm_mcp/instances.py` | Loading the curated list and the operator's, and the host policy an address given as an instance must pass. |
35
+ | `src/contentdm_mcp/instances.yaml` | The curated list itself. |
36
+ | `src/contentdm_mcp/shape.py` | Turning records into results: kinds, labels, the citation core. |
37
+ | `src/contentdm_mcp/config.py` | Settings from the environment and `.env`. |
38
+ | `docs/API-NOTES.md` | What the API was observed to do, and when. |
39
+ | `docs/DESIGN.md` | Why the server is shaped the way it is, and what is out of scope by decision. |
40
+ | `tests/fixtures/` | Recorded responses and the tool-schema snapshot. |
41
+ | `tests/test_tool_contract.py` | Tests over the tool surface as a client sees it. |
42
+ | `tests/test_security.py` | Every refusal against a steered model: hosts, addresses, sizes, files. |
43
+ | `tests/live_check.py` | The one script that talks to live sites, run by hand. Not collected. |
44
+
45
+ ## What a change carries
46
+
47
+ **A test that fails without it.** Bug fixes especially: reproduce the bug as a
48
+ test first.
49
+
50
+ **Descriptions written for the model.** A tool's docstring is what a model
51
+ reads when choosing and calling it. The combined descriptions have a ceiling
52
+ (`DESCRIPTION_BUDGET` in `tests/test_tool_contract.py`), because they are sent
53
+ on every session. Raise it deliberately, in a pull request of its own. They
54
+ are dedented at import, so the budget measures the same text on every Python.
55
+
56
+ **The evidence distinction, kept.** A transcript, OCR text or index entry is a
57
+ lead to the page image, never the record. Tools that return text say so; a
58
+ contract test enforces it.
59
+
60
+ **A citation core on every item.** Any result that names an item carries its
61
+ public address and the institution, collection, title and identifier.
62
+
63
+ **A structured result, never an exception.** Every tool catches its failures
64
+ and returns an `error` envelope. A sweep test calls every tool with every site
65
+ failing and fails if one raises.
66
+
67
+ **Nothing that writes to a site, and nothing that works around a bot check.**
68
+ See [docs/DESIGN.md](docs/DESIGN.md#out-of-scope-by-decision).
69
+
70
+ **Nothing that widens what an argument can reach.** Treat every argument as
71
+ written by someone else's text. A new kind of host goes through the policy in
72
+ `instances.resolve`, a new place to write through `server._destination`, and
73
+ each comes with a test in `tests/test_security.py` showing that the refusals
74
+ still come before the harm. See
75
+ [docs/DESIGN.md](docs/DESIGN.md#which-hosts-a-tool-reaches).
76
+
77
+ **A new fixture recorded, not invented,** when a change depends on how a site
78
+ answers, with what was observed and when added to `docs/API-NOTES.md`. Keep
79
+ fixtures to historical records.
80
+
81
+ **The snapshot, when the surface changes.** Renaming or adding a tool or a
82
+ parameter fails the snapshot test on purpose. Regenerate it with
83
+ `uv run python -m tests.regen_tool_snapshot`, update the README tables, and add
84
+ a `CHANGELOG.md` entry.
85
+
86
+ ## Adding a site to the curated list
87
+
88
+ 1. Call `{base}/digital/bl/dmwebservices/index.php?q=dmGetCollectionList/json`
89
+ yourself. Record the number of collections and the day.
90
+ 2. Prefer state archives, state libraries and statewide networks. Add a
91
+ university or public library only when it holds state or county records,
92
+ and say which collections in `holds`.
93
+ 3. Read a few items' rights fields and put what they say about reuse in
94
+ `terms`.
95
+ 4. Check the public host with Python (`uv run python -c "import httpx;
96
+ httpx.get('https://host/')"`). If its certificate chain is incomplete, add
97
+ the site's `cdmNNNNN.contentdm.oclc.org` address as `api_base`: the number
98
+ is in the `path` of every collection in the list.
99
+ 5. Run `uv run python -m tests.live_check`: it calls every entry.
100
+
101
+ A site that has left CONTENTdm stays in the list with `status: moved` and a
102
+ `moved_to`, so a search can say where it went.
103
+
104
+ ## Adding a platform
105
+
106
+ Write an adapter in `src/contentdm_mcp/adapters/` that implements
107
+ `adapters.base.Adapter`, register it in `ADAPTERS`, record its answers as
108
+ fixtures, and change the affected entries' `platform`. The tools should not
109
+ need to change; if they do, say why in the pull request.
110
+
111
+ ## Releasing
112
+
113
+ A maintainer bumps `__version__` in `src/contentdm_mcp/__init__.py` and
114
+ both versions in `server.json`, moves the changelog's Unreleased entries under
115
+ the new version, and publishes a GitHub release tagged `v<version>`. The
116
+ release workflow builds the tag, publishes to PyPI by Trusted Publishing, and
117
+ lists the version in the MCP Registry.
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Ian Anderson
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,318 @@
1
+ Metadata-Version: 2.5
2
+ Name: contentdm-mcp
3
+ Version: 0.1.0
4
+ Summary: MCP server for the record images and transcripts that US state archives and state libraries publish on OCLC's CONTENTdm: search one instance or a curated list of them, read an item and its pages, and download a page image. For genealogy and history.
5
+ Project-URL: Homepage, https://github.com/ianderso/contentdm-mcp
6
+ Project-URL: Repository, https://github.com/ianderso/contentdm-mcp
7
+ Project-URL: Issues, https://github.com/ianderso/contentdm-mcp/issues
8
+ Project-URL: Changelog, https://github.com/ianderso/contentdm-mcp/blob/main/CHANGELOG.md
9
+ Author: Ian Anderson
10
+ License-Expression: MIT
11
+ License-File: LICENSE
12
+ Keywords: archives,contentdm,digital-collections,family-history,genealogy,iiif,mcp,research,state-archives
13
+ Classifier: Development Status :: 4 - Beta
14
+ Classifier: Intended Audience :: End Users/Desktop
15
+ Classifier: Intended Audience :: Science/Research
16
+ Classifier: Operating System :: OS Independent
17
+ Classifier: Programming Language :: Python :: 3
18
+ Classifier: Programming Language :: Python :: 3 :: Only
19
+ Classifier: Programming Language :: Python :: 3.11
20
+ Classifier: Programming Language :: Python :: 3.12
21
+ Classifier: Programming Language :: Python :: 3.13
22
+ Classifier: Topic :: Sociology :: Genealogy
23
+ Classifier: Topic :: Sociology :: History
24
+ Requires-Python: >=3.11
25
+ Requires-Dist: httpcore>=1.0
26
+ Requires-Dist: httpx<1,>=0.27
27
+ Requires-Dist: mcp<3,>=2.0.0
28
+ Requires-Dist: pydantic>=2.6
29
+ Requires-Dist: python-dotenv>=1.0
30
+ Requires-Dist: pyyaml>=6.0.1
31
+ Provides-Extra: dev
32
+ Requires-Dist: pytest-asyncio>=0.24; extra == 'dev'
33
+ Requires-Dist: pytest>=8.2; extra == 'dev'
34
+ Requires-Dist: respx>=0.21; extra == 'dev'
35
+ Requires-Dist: ruff>=0.9; extra == 'dev'
36
+ Description-Content-Type: text/markdown
37
+
38
+ # contentdm-mcp
39
+
40
+ [![CI](https://github.com/ianderso/contentdm-mcp/actions/workflows/ci.yml/badge.svg)](https://github.com/ianderso/contentdm-mcp/actions/workflows/ci.yml)
41
+ [![PyPI](https://img.shields.io/pypi/v/contentdm-mcp)](https://pypi.org/project/contentdm-mcp/)
42
+
43
+ <!-- mcp-name: io.github.ianderso/contentdm-mcp -->
44
+
45
+ An [MCP](https://modelcontextprotocol.io) server for the **record images
46
+ that US state archives and state libraries publish on CONTENTdm**: death
47
+ certificates and their annual indexes, Confederate pension files, voter
48
+ registers, prison registers, county court and estate papers, letters and
49
+ diaries, with whatever transcript or OCR text the institution added.
50
+
51
+ [CONTENTdm](https://www.oclc.org/en/contentdm.html) is OCLC's hosted
52
+ digital-collections platform. The Alabama Department of Archives and History,
53
+ the Georgia Archives' Virtual Vault, the Tennessee State Library and Archives,
54
+ Ohio Memory, the Illinois Digital Archives, the Indiana State Library,
55
+ Missouri Digital Heritage and many more run on it, and every one of them
56
+ answers the same keyless JSON API and IIIF image service. The server ships a
57
+ curated list of those sites, searches one or several of them at once, reads
58
+ an item and its pages, and downloads a page image to read.
59
+
60
+ It works the way a careful genealogist does. **The page image is the
61
+ evidence.** A transcript or OCR text is a lead that says which page to read,
62
+ and an index card is a finding aid to the record behind it. Every result
63
+ carries the holding institution's item address and a citation core
64
+ (institution, collection, title, identifier, address), because the item page
65
+ is what gets cited, not this server.
66
+
67
+ Nothing here writes anywhere, and nothing here keeps a family tree. It sits
68
+ well beside [dpla-catalog-mcp](https://github.com/ianderso/dpla-catalog-mcp),
69
+ which finds an item across hundreds of collections and hands you the
70
+ institution's address (`get_item` takes that address directly), and
71
+ [nara-catalog-mcp](https://github.com/ianderso/nara-catalog-mcp) for federal
72
+ records.
73
+
74
+ This is an independent project. It is not affiliated with, endorsed by, or
75
+ supported by OCLC or any of the institutions whose sites it reads.
76
+
77
+ ## Tools
78
+
79
+ The server publishes eight tools. All but `get_image` are read-only;
80
+ `get_image` writes one new file and never overwrites one.
81
+
82
+ **Finding**
83
+
84
+ | Tool | Purpose |
85
+ | --- | --- |
86
+ | `list_instances` | The curated sites: who runs each, what it holds for a genealogist, any reuse terms, and the day it was checked. Sites that have left CONTENTdm are listed with where they went. Makes no request. |
87
+ | `list_collections` | A site's collections and the alias each is searched by. |
88
+ | `get_collection` | One collection's fields: what `search` can name in `field`, and which field holds the transcript or OCR text. |
89
+ | `search` | Search one site, or several side by side. Says which sites answered, failed or timed out, so a negative is bounded. `page_hits` returns the page inside a volume where a text match sits. |
90
+
91
+ **Reading**
92
+
93
+ | Tool | Purpose |
94
+ | --- | --- |
95
+ | `get_item` | One item: metadata, transcript or OCR text, the institution's own "cite as" line and rights statement, and the citation core. For a page, which object and page number it is. Takes an item address, such as DPLA's `isShownAt`, on a site it may reach. |
96
+ | `get_pages` | A compound object's pages in order, with `query` to mark the pages that match. |
97
+ | `get_image` | Download one page image through the site's IIIF service, sized to fit, including any page of a multi-page PDF. Saves a new `.jpg`, never a hidden file or one under `~/Library`. |
98
+ | `cache_status` | This session's requests, by site, and cache use. Makes no request. |
99
+
100
+ ## The curated sites
101
+
102
+ `list_instances` gives the whole list. On 2026-10-06, 18 sites answered, and 7
103
+ entries record institutions that have left CONTENTdm, so a search can say
104
+ where they went:
105
+
106
+ | State | Site | Status |
107
+ | --- | --- | --- |
108
+ | AL | Alabama Department of Archives and History | supported |
109
+ | AK | Alaska's Digital Archives | supported |
110
+ | AR | Arkansas State Archives | supported |
111
+ | CT | Connecticut State Library | supported |
112
+ | DE | Delaware Division of Libraries, with the Delaware Public Archives | supported |
113
+ | GA | Georgia Archives, Virtual Vault | supported |
114
+ | IL | Illinois Digital Archives | supported |
115
+ | IN | Indiana State Library (Indiana Memory); Ball State University | supported |
116
+ | KY | Kentucky Digital Library | supported |
117
+ | MO | Missouri Digital Heritage | supported |
118
+ | ND | Digital Horizons | supported |
119
+ | NM | New Mexico Digital Collections, with the State Records Center and Archives | supported |
120
+ | NV | Nevada State Library, Archives and Public Records | supported |
121
+ | OH | Ohio Memory | supported |
122
+ | OK | Oklahoma Department of Libraries, with the Oklahoma State Archives | supported |
123
+ | TN | Tennessee State Library and Archives (TeVA) | supported |
124
+ | WA | Washington Rural Heritage | supported |
125
+ | AZ | Arizona Memory Project | left CONTENTdm; behind a bot check |
126
+ | ID, SD | State archives' digital collections | moved to Preservica |
127
+ | MT | Montana Memory Project | moved to Recollect |
128
+ | NC | State Archives of North Carolina | moved to Quartex |
129
+ | PA | POWER Library | moved to Islandora |
130
+ | WI | Wisconsin Historical Society | moved to its own site |
131
+
132
+ Any other CONTENTdm site works too, at its `cdmNNNNN.contentdm.oclc.org`
133
+ address: pass `https://cdmNNNNN.contentdm.oclc.org` as `instance`. Every
134
+ classic site answers there, whatever its own domain, and the source of its
135
+ pages names the number (`"cdmServerUrl": "serverNNNNN.contentdm.oclc.org"`).
136
+ There are more than 675 of them, and no registry. To use a site by its own
137
+ domain, add it to `CONTENTDM_EXTRA_INSTANCES`; a tool argument cannot, for the
138
+ reason under [Security](#security).
139
+
140
+ ## Setup
141
+
142
+ You need Python 3.11 or later and [uv](https://docs.astral.sh/uv/). There is
143
+ no key to request.
144
+
145
+ **Without cloning.** `uvx` fetches it from PyPI and runs it in one step:
146
+
147
+ ```bash
148
+ uvx contentdm-mcp
149
+ ```
150
+
151
+ **From a clone**, which is what you want if you will change it:
152
+
153
+ ```bash
154
+ git clone https://github.com/ianderso/contentdm-mcp
155
+ cd contentdm-mcp
156
+ uv sync
157
+ uv run contentdm-mcp # stdio server, usually launched by the client
158
+ ```
159
+
160
+ Either way the server speaks MCP over stdio, so you will normally let an MCP
161
+ client start it rather than run it by hand.
162
+
163
+ ### Claude Desktop
164
+
165
+ ```json
166
+ {
167
+ "mcpServers": {
168
+ "contentdm": {
169
+ "command": "uvx",
170
+ "args": ["contentdm-mcp"]
171
+ }
172
+ }
173
+ }
174
+ ```
175
+
176
+ A desktop app does not always inherit your shell's `PATH`. If the server fails
177
+ to start because `uvx` cannot be found, give the full path that `which uvx`
178
+ prints as the `command`.
179
+
180
+ ### Claude Code
181
+
182
+ ```bash
183
+ claude mcp add contentdm -- uvx contentdm-mcp
184
+ ```
185
+
186
+ ## Configuration
187
+
188
+ Nothing is required. A `.env` file in the directory the server starts in
189
+ supplies anything the environment does not; only that directory is read.
190
+
191
+ | Variable | Meaning |
192
+ | --- | --- |
193
+ | `CONTENTDM_CACHE_DIR` | Response cache directory. Default `~/.cache/contentdm-mcp`. |
194
+ | `CONTENTDM_TIMEOUT` | HTTP timeout in seconds for one request. Default 30. Image downloads get 120 to read. |
195
+ | `CONTENTDM_MIN_INTERVAL` | Least seconds between two requests to one site. Default 1, and never below 0.5. |
196
+ | `CONTENTDM_CONTACT` | An email address or URL added to the User-Agent, so an institution can reach you if your use causes trouble. Optional, and courteous. |
197
+ | `CONTENTDM_EXTRA_INSTANCES` | More sites a tool may reach by their own domain: https base URLs separated by commas, or the absolute path of a YAML file whose entries are shaped like [`instances.yaml`](src/contentdm_mcp/instances.yaml)'s (only `base_url` is required). `list_instances` lists them; a search of every site leaves them out. |
198
+ | `CONTENTDM_DOWNLOAD_DIR` | An existing folder. When set, `get_image` saves only inside it. Set it to save into an iCloud Drive folder, which lives under `~/Library`. |
199
+
200
+ An unusable value is reported on the first tool call as a `not_configured`
201
+ result naming the variable.
202
+
203
+ ## Being a good guest
204
+
205
+ Each site is one institution's server. The client sends one request at a time
206
+ to each site, at least a second apart; a search across several sites runs
207
+ them side by side, six at a time, so no site sees more than one caller. Two
208
+ identical calls in flight share one request. Collection lists and field
209
+ definitions are cached for 30 days, items for 7, searches for a day. A 429, a
210
+ 5xx or a dropped connection gets one retry, honouring `Retry-After`. The
211
+ User-Agent names the package, its version and this repository.
212
+
213
+ `robots.txt` on these sites disallows the website's search pages, not the
214
+ API this server uses. Each institution's own terms govern what you do with
215
+ its images: `list_instances` and `get_item` pass on what the sites say, and
216
+ several ask for permission before publication.
217
+
218
+ ## How to read what comes back
219
+
220
+ - **A search is only as wide as `answered`.** `failed`, `timed_out` and
221
+ `not_searched` list the sites a negative does not cover. A site that
222
+ redirects, asks for a bot check or lists no collections has usually moved.
223
+ - **Search covers metadata and text fields, never the image.** Many record
224
+ images have no text at all, so a name can be on a page that no search finds.
225
+ Browse by collection, or by the index volume, instead.
226
+ - **A compound object's text is on its pages.** Volumes, files and newspaper
227
+ issues are compound objects; their own transcript field is usually empty.
228
+ Use `page_hits=true`, or `get_pages` with a `query`, to find the page.
229
+ - **Field names are per collection.** "Name of Father" is `namea` in one
230
+ collection and `father` in another. `get_collection` says which nick holds
231
+ what, and which field is the transcript: that is marked by type, because
232
+ the name differs from site to site.
233
+ - **"Transcript" can be machine OCR.** The Tennessee vital-records indexes
234
+ carry OCR under that label. Read the image.
235
+ - **`exact` is a phrase search** (the words together, in order), not
236
+ whole-field equality. `all` needs every word, `any` one of them.
237
+ - **A PDF item can hold many pages.** Its IIIF image shows the first;
238
+ `get_image` with `pdf_page` reaches the others.
239
+ - **`page` is a position in the object,** counted from 1, not the number
240
+ printed on the scan: the entry for Sgt. Alvin C. York is on page 461 of
241
+ volume 8 of Tennessee's 1960-1964 death index, a scan named `54784_460` and
242
+ stamped 457. Cite the printed number too when it differs.
243
+ - **Cite the item page.** Use the `citation` each result carries, and the
244
+ institution's own `cite_as` line where `get_item` finds one (Georgia,
245
+ Oklahoma and Alaska items have them). Record the identifier
246
+ (`alias:pointer`) so the item can be found again.
247
+
248
+ ## Deliberately not here
249
+
250
+ - **Writing to any site.** CONTENTdm's write functions need a staff login.
251
+ - **Sites on other platforms.** The new CONTENTdm (launched 2026-09-22) has
252
+ no published public API, and Quartex's is undocumented. The tools speak to
253
+ an adapter layer, so either can be added without changing them; see
254
+ [docs/DESIGN.md](docs/DESIGN.md).
255
+ - **Working around bot checks.** A site that answers with a challenge is
256
+ reported as blocked and left alone.
257
+
258
+ ## Security
259
+
260
+ Tool arguments are written by a model, and the model reads text this server
261
+ does not control: titles, transcripts, DPLA records, web pages. The server
262
+ assumes that text can steer the model, and limits what a steered model can
263
+ make it do.
264
+
265
+ - **Which hosts.** A tool reaches the curated sites, any address under
266
+ `contentdm.oclc.org` (OCLC's own servers), and the sites you list in
267
+ `CONTENTDM_EXTRA_INSTANCES`. Any other host is refused before it is even
268
+ looked up, because a DNS query for `secret.attacker.example` would already
269
+ deliver the name; the refusal says how to reach the site instead.
270
+ - **Which addresses.** Each connection is checked where it is made: a name
271
+ that leads to a private, loopback, link-local, CGNAT, multicast, reserved or
272
+ unspecified address, IPv4 or IPv6, is refused, and the connection goes to
273
+ the address that was checked. Redirects are followed only between an
274
+ instance's own hosts, and are checked again. Proxy settings in the
275
+ environment are not used.
276
+ - **How much.** A JSON answer over 10 MB, or an image over 60 MB, is refused
277
+ as it streams in.
278
+ - **Which files.** `get_image` creates one new file and never overwrites one.
279
+ The bytes must be an image, judged by their first bytes rather than the
280
+ Content-Type, and of the format the file's suffix names: `.jpg` or `.jpeg`.
281
+ Never a hidden file or folder, never under `~/Library`, and with
282
+ `CONTENTDM_DOWNLOAD_DIR` set, never outside it, all judged after links are
283
+ resolved. A refused download leaves nothing on disk.
284
+ - **Arguments are validated** (collection aliases, field nicks, numeric
285
+ pointers) before they reach a request, and search words are stripped of the
286
+ API's own syntax characters.
287
+ - **Site text is untrusted.** Titles, descriptions and transcripts reach the
288
+ model verbatim. The server's instructions tell the model to treat that text
289
+ as material to weigh, never as instructions; the model still decides, so
290
+ review what it proposes to do.
291
+
292
+ To report a vulnerability, see [SECURITY.md](SECURITY.md).
293
+
294
+ ## Development
295
+
296
+ ```bash
297
+ uv sync --extra dev
298
+ uv run pytest # mocked with respx; never touches a site
299
+ uv run ruff check .
300
+ uv run ruff format --check .
301
+ uv run python -m tests.live_check # paced calls to the live sites
302
+ ```
303
+
304
+ The live check asks the sites what the recorded fixtures cannot: whether
305
+ their answers still have the shape the server reads, and whether every
306
+ curated site still answers. See [CONTRIBUTING.md](CONTRIBUTING.md) for how the
307
+ suite is organised, [docs/API-NOTES.md](docs/API-NOTES.md) for what was
308
+ observed of the API and when, and [docs/DESIGN.md](docs/DESIGN.md) for why the
309
+ server is shaped this way.
310
+
311
+ ## Credits
312
+
313
+ The images and descriptions belong to the institutions that publish them.
314
+ CONTENTdm is a product of [OCLC](https://www.oclc.org).
315
+
316
+ ## License
317
+
318
+ [MIT](LICENSE).