contentdm-mcp 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- contentdm_mcp-0.1.0/.env.example +34 -0
- contentdm_mcp-0.1.0/.gitignore +39 -0
- contentdm_mcp-0.1.0/CHANGELOG.md +52 -0
- contentdm_mcp-0.1.0/CONTRIBUTING.md +117 -0
- contentdm_mcp-0.1.0/LICENSE +21 -0
- contentdm_mcp-0.1.0/PKG-INFO +318 -0
- contentdm_mcp-0.1.0/README.md +281 -0
- contentdm_mcp-0.1.0/SECURITY.md +80 -0
- contentdm_mcp-0.1.0/docs/API-NOTES.md +230 -0
- contentdm_mcp-0.1.0/docs/DESIGN.md +198 -0
- contentdm_mcp-0.1.0/pyproject.toml +83 -0
- contentdm_mcp-0.1.0/server.json +52 -0
- contentdm_mcp-0.1.0/src/contentdm_mcp/__init__.py +11 -0
- contentdm_mcp-0.1.0/src/contentdm_mcp/__main__.py +13 -0
- contentdm_mcp-0.1.0/src/contentdm_mcp/adapters/__init__.py +54 -0
- contentdm_mcp-0.1.0/src/contentdm_mcp/adapters/base.py +186 -0
- contentdm_mcp-0.1.0/src/contentdm_mcp/adapters/classic.py +476 -0
- contentdm_mcp-0.1.0/src/contentdm_mcp/client.py +811 -0
- contentdm_mcp-0.1.0/src/contentdm_mcp/config.py +159 -0
- contentdm_mcp-0.1.0/src/contentdm_mcp/instances.py +393 -0
- contentdm_mcp-0.1.0/src/contentdm_mcp/instances.yaml +422 -0
- contentdm_mcp-0.1.0/src/contentdm_mcp/server.py +1179 -0
- contentdm_mcp-0.1.0/src/contentdm_mcp/shape.py +161 -0
- contentdm_mcp-0.1.0/tests/__init__.py +0 -0
- contentdm_mcp-0.1.0/tests/conftest.py +263 -0
- contentdm_mcp-0.1.0/tests/fixtures/al_bad_alias.html +1 -0
- contentdm_mcp-0.1.0/tests/fixtures/al_collection_list.json +50 -0
- contentdm_mcp-0.1.0/tests/fixtures/al_compound_16539.json +70 -0
- contentdm_mcp-0.1.0/tests/fixtures/al_getparent_page.json +3 -0
- contentdm_mcp-0.1.0/tests/fixtures/al_getparent_top.json +3 -0
- contentdm_mcp-0.1.0/tests/fixtures/al_iiif_400_pdf_page.txt +6 -0
- contentdm_mcp-0.1.0/tests/fixtures/al_iiif_501_audio.txt +6 -0
- contentdm_mcp-0.1.0/tests/fixtures/al_iiif_info_16527_pdf.json +104 -0
- contentdm_mcp-0.1.0/tests/fixtures/al_item_16538_transcript.json +44 -0
- contentdm_mcp-0.1.0/tests/fixtures/al_item_16539_compound.json +44 -0
- contentdm_mcp-0.1.0/tests/fixtures/al_item_570_audio.json +41 -0
- contentdm_mcp-0.1.0/tests/fixtures/al_item_not_found.json +5 -0
- contentdm_mcp-0.1.0/tests/fixtures/al_not_compound.json +4 -0
- contentdm_mcp-0.1.0/tests/fixtures/al_query_all_empty.json +8 -0
- contentdm_mcp-0.1.0/tests/fixtures/al_query_voices_semmes.json +39 -0
- contentdm_mcp-0.1.0/tests/fixtures/al_records_fields.json +467 -0
- contentdm_mcp-0.1.0/tests/fixtures/al_voices_16527_p2_64px.jpg +0 -0
- contentdm_mcp-0.1.0/tests/fixtures/al_voices_fields.json +512 -0
- contentdm_mcp-0.1.0/tests/fixtures/ga_collection_list.json +20 -0
- contentdm_mcp-0.1.0/tests/fixtures/ga_deaths_fields.json +497 -0
- contentdm_mcp-0.1.0/tests/fixtures/ga_item_300730_death.json +43 -0
- contentdm_mcp-0.1.0/tests/fixtures/mo_cohist_fields.json +407 -0
- contentdm_mcp-0.1.0/tests/fixtures/mo_collection_list.json +8 -0
- contentdm_mcp-0.1.0/tests/fixtures/mo_compound_96244_monograph.json +603 -0
- contentdm_mcp-0.1.0/tests/fixtures/oh_collection_list.json +20 -0
- contentdm_mcp-0.1.0/tests/fixtures/oh_exponent_fields.json +512 -0
- contentdm_mcp-0.1.0/tests/fixtures/oh_item_21598_newspaper_page.json +44 -0
- contentdm_mcp-0.1.0/tests/fixtures/sd_collection_list_empty.json +1 -0
- contentdm_mcp-0.1.0/tests/fixtures/tn_collection_list.json +20 -0
- contentdm_mcp-0.1.0/tests/fixtures/tn_compound_1206616.json +2340 -0
- contentdm_mcp-0.1.0/tests/fixtures/tn_deaths_fields.json +467 -0
- contentdm_mcp-0.1.0/tests/fixtures/tn_getparent_page.json +3 -0
- contentdm_mcp-0.1.0/tests/fixtures/tn_iiif_info_1206609.json +87 -0
- contentdm_mcp-0.1.0/tests/fixtures/tn_item_1206609_page.json +41 -0
- contentdm_mcp-0.1.0/tests/fixtures/tn_item_1206616_parent.json +41 -0
- contentdm_mcp-0.1.0/tests/fixtures/tn_query_in_compound.json +17 -0
- contentdm_mcp-0.1.0/tests/fixtures/tn_query_york_objects.json +49 -0
- contentdm_mcp-0.1.0/tests/fixtures/tn_query_york_pages.json +49 -0
- contentdm_mcp-0.1.0/tests/fixtures/tool_schema.json +57 -0
- contentdm_mcp-0.1.0/tests/live_check.py +234 -0
- contentdm_mcp-0.1.0/tests/regen_tool_snapshot.py +29 -0
- contentdm_mcp-0.1.0/tests/test_classic.py +302 -0
- contentdm_mcp-0.1.0/tests/test_client.py +316 -0
- contentdm_mcp-0.1.0/tests/test_config.py +120 -0
- contentdm_mcp-0.1.0/tests/test_entrypoint.py +55 -0
- contentdm_mcp-0.1.0/tests/test_instances.py +268 -0
- contentdm_mcp-0.1.0/tests/test_registry_listing.py +62 -0
- contentdm_mcp-0.1.0/tests/test_security.py +590 -0
- contentdm_mcp-0.1.0/tests/test_server.py +422 -0
- contentdm_mcp-0.1.0/tests/test_shape.py +108 -0
- contentdm_mcp-0.1.0/tests/test_tool_contract.py +229 -0
- contentdm_mcp-0.1.0/uv.lock +985 -0
|
@@ -0,0 +1,34 @@
|
|
|
1
|
+
# Copy to .env to change a default. Nothing here is required: CONTENTdm's web
|
|
2
|
+
# services need no key and no account.
|
|
3
|
+
#
|
|
4
|
+
# The server reads .env from the directory it is started in -- the
|
|
5
|
+
# repository root, when launched with `uv --directory /path/to/contentdm-mcp`.
|
|
6
|
+
# Variables already set in the environment take precedence.
|
|
7
|
+
|
|
8
|
+
# Optional: an email address or URL added to the User-Agent, so an
|
|
9
|
+
# institution can reach you if your use ever causes it trouble.
|
|
10
|
+
# CONTENTDM_CONTACT=you@example.org
|
|
11
|
+
|
|
12
|
+
# Optional: directory for the on-disk response cache.
|
|
13
|
+
# CONTENTDM_CACHE_DIR=~/.cache/contentdm-mcp
|
|
14
|
+
|
|
15
|
+
# Optional: HTTP timeout in seconds for one request (default 30; image
|
|
16
|
+
# downloads get 120 to read).
|
|
17
|
+
# CONTENTDM_TIMEOUT=30
|
|
18
|
+
|
|
19
|
+
# Optional: least seconds between two requests to one site (default 1, never
|
|
20
|
+
# below 0.5).
|
|
21
|
+
# CONTENTDM_MIN_INTERVAL=1
|
|
22
|
+
|
|
23
|
+
# Optional: more CONTENTdm sites a tool may reach. Without this, a tool reaches
|
|
24
|
+
# the curated sites and any address under contentdm.oclc.org, and nothing
|
|
25
|
+
# else. Either https base URLs separated by commas, or the absolute path of a
|
|
26
|
+
# YAML file whose entries are shaped like instances.yaml's (only base_url is
|
|
27
|
+
# required). A model cannot add to this list; only you can.
|
|
28
|
+
# CONTENTDM_EXTRA_INSTANCES=https://digital.example.edu
|
|
29
|
+
# CONTENTDM_EXTRA_INSTANCES=/Users/you/contentdm-sites.yaml
|
|
30
|
+
|
|
31
|
+
# Optional: an existing folder that every get_image destination must lie
|
|
32
|
+
# inside. get_image never writes under ~/Library otherwise; set this to save
|
|
33
|
+
# into an iCloud Drive folder, which lives there.
|
|
34
|
+
# CONTENTDM_DOWNLOAD_DIR="~/Library/Mobile Documents/com~apple~CloudDocs/Genealogy"
|
|
@@ -0,0 +1,39 @@
|
|
|
1
|
+
# Settings. .env holds local overrides; .env.example is the committed template.
|
|
2
|
+
.env
|
|
3
|
+
.env.*
|
|
4
|
+
!.env.example
|
|
5
|
+
.envrc
|
|
6
|
+
|
|
7
|
+
# Response cache
|
|
8
|
+
.cache/
|
|
9
|
+
|
|
10
|
+
# Downloaded page images are never part of this repository. The one
|
|
11
|
+
# recorded test image is the exception.
|
|
12
|
+
*.jpg
|
|
13
|
+
!tests/fixtures/*.jpg
|
|
14
|
+
*.jpeg
|
|
15
|
+
*.png
|
|
16
|
+
*.tif
|
|
17
|
+
*.tiff
|
|
18
|
+
*.pdf
|
|
19
|
+
|
|
20
|
+
# Python
|
|
21
|
+
__pycache__/
|
|
22
|
+
*.py[cod]
|
|
23
|
+
.venv/
|
|
24
|
+
venv/
|
|
25
|
+
build/
|
|
26
|
+
dist/
|
|
27
|
+
*.egg-info/
|
|
28
|
+
|
|
29
|
+
# Tool caches
|
|
30
|
+
.pytest_cache/
|
|
31
|
+
.ruff_cache/
|
|
32
|
+
.mypy_cache/
|
|
33
|
+
.coverage
|
|
34
|
+
htmlcov/
|
|
35
|
+
|
|
36
|
+
# Editors and OS
|
|
37
|
+
.vscode/
|
|
38
|
+
.idea/
|
|
39
|
+
.DS_Store
|
|
@@ -0,0 +1,52 @@
|
|
|
1
|
+
# Changelog
|
|
2
|
+
|
|
3
|
+
All notable changes to this project are recorded here. The format follows
|
|
4
|
+
[Keep a Changelog](https://keepachangelog.com/en/1.1.0/), and the project uses
|
|
5
|
+
[semantic versioning](https://semver.org/). The tool surface is the public
|
|
6
|
+
interface: renaming or removing a tool or a parameter is a major release, and
|
|
7
|
+
adding one is a minor release. Before 1.0, a minor release may do either.
|
|
8
|
+
Changes to `instances.yaml` alone (a site added, moved or re-checked) are
|
|
9
|
+
patch releases.
|
|
10
|
+
|
|
11
|
+
## [Unreleased]
|
|
12
|
+
|
|
13
|
+
## [0.1.0] — 2026-10-06
|
|
14
|
+
|
|
15
|
+
First release.
|
|
16
|
+
|
|
17
|
+
### Added
|
|
18
|
+
|
|
19
|
+
- Eight tools over classic CONTENTdm's `dmwebservices` API and IIIF image
|
|
20
|
+
service: `list_instances`, `list_collections`, `get_collection`, `search`,
|
|
21
|
+
`get_item`, `get_pages`, `get_image` (writes one new file, never
|
|
22
|
+
overwrites) and `cache_status`.
|
|
23
|
+
- A host policy for arguments a model writes: a tool reaches the curated
|
|
24
|
+
sites, any address under `contentdm.oclc.org` (where every classic site
|
|
25
|
+
answers as `cdmNNNNN`), and sites the operator lists in
|
|
26
|
+
`CONTENTDM_EXTRA_INSTANCES`. Any other host is refused before it is looked
|
|
27
|
+
up, with a message naming both ways forward.
|
|
28
|
+
- Every connection checked where it is made: a name that leads to a private,
|
|
29
|
+
loopback, link-local, CGNAT, multicast, reserved or unspecified address,
|
|
30
|
+
IPv4 or IPv6, is refused, and the connection goes to the address checked,
|
|
31
|
+
so redirects and DNS rebinding are covered. JSON answers are capped at
|
|
32
|
+
10 MB and images at 60 MB as they stream.
|
|
33
|
+
- `get_image` saves only a real image of the format its `.jpg` or `.jpeg`
|
|
34
|
+
name says, never a hidden file or one under `~/Library`, and only inside
|
|
35
|
+
`CONTENTDM_DOWNLOAD_DIR` when that is set.
|
|
36
|
+
- A curated `instances.yaml` of 18 state archive, state library and
|
|
37
|
+
university sites holding state or county records, each verified live on
|
|
38
|
+
2026-10-06, and 7 entries for institutions that have left CONTENTdm
|
|
39
|
+
(North Carolina to Quartex, Pennsylvania's POWER Library to Islandora,
|
|
40
|
+
South Dakota and Idaho to Preservica, Montana to Recollect, Wisconsin and
|
|
41
|
+
Arizona to their own sites).
|
|
42
|
+
- An adapter layer between the tools and the API, so the new CONTENTdm or
|
|
43
|
+
Quartex can be added without changing the tools.
|
|
44
|
+
- `search` across several sites side by side, reporting which answered,
|
|
45
|
+
failed, timed out or were not searched; `page_hits` for the page inside a
|
|
46
|
+
volume where a text match is.
|
|
47
|
+
- Every result carries the item's public address and a citation core
|
|
48
|
+
(institution, collection, title, identifier, address); `get_item` adds the
|
|
49
|
+
institution's own "cite as" line and rights statement where it has them.
|
|
50
|
+
- A polite client: one request at a time per site at least a second apart,
|
|
51
|
+
identical calls joined, answers cached (collection lists and fields 30
|
|
52
|
+
days, items 7, searches 1), one retry on 429 and 5xx.
|
|
@@ -0,0 +1,117 @@
|
|
|
1
|
+
# Contributing
|
|
2
|
+
|
|
3
|
+
Issues and pull requests are welcome. This file says how the project is put
|
|
4
|
+
together and what a change is expected to carry.
|
|
5
|
+
|
|
6
|
+
## Setting up
|
|
7
|
+
|
|
8
|
+
```bash
|
|
9
|
+
git clone https://github.com/ianderso/contentdm-mcp
|
|
10
|
+
cd contentdm-mcp
|
|
11
|
+
uv sync --extra dev
|
|
12
|
+
```
|
|
13
|
+
|
|
14
|
+
Before sending a change, run what CI runs:
|
|
15
|
+
|
|
16
|
+
```bash
|
|
17
|
+
uv run ruff check .
|
|
18
|
+
uv run ruff format --check .
|
|
19
|
+
uv run pytest
|
|
20
|
+
```
|
|
21
|
+
|
|
22
|
+
The suite is mocked with [respx](https://lundberg.github.io/respx/) against
|
|
23
|
+
recorded CONTENTdm responses. It must never touch a live site: each one is an
|
|
24
|
+
institution's own server, and CI should not depend on any of them being up.
|
|
25
|
+
|
|
26
|
+
## Where things live
|
|
27
|
+
|
|
28
|
+
| Path | What it holds |
|
|
29
|
+
| --- | --- |
|
|
30
|
+
| `src/contentdm_mcp/server.py` | The tools. Their docstrings and `Field` descriptions *are* the published tool descriptions and schema. |
|
|
31
|
+
| `src/contentdm_mcp/adapters/base.py` | The interface every platform adapter provides, and the records it returns. |
|
|
32
|
+
| `src/contentdm_mcp/adapters/classic.py` | Classic CONTENTdm: the `dmwebservices` calls, IIIF, and item addresses. |
|
|
33
|
+
| `src/contentdm_mcp/client.py` | The cached, paced HTTP client, the host allowlist, the redirect rule, the public-address check made at connect time, and the size caps. |
|
|
34
|
+
| `src/contentdm_mcp/instances.py` | Loading the curated list and the operator's, and the host policy an address given as an instance must pass. |
|
|
35
|
+
| `src/contentdm_mcp/instances.yaml` | The curated list itself. |
|
|
36
|
+
| `src/contentdm_mcp/shape.py` | Turning records into results: kinds, labels, the citation core. |
|
|
37
|
+
| `src/contentdm_mcp/config.py` | Settings from the environment and `.env`. |
|
|
38
|
+
| `docs/API-NOTES.md` | What the API was observed to do, and when. |
|
|
39
|
+
| `docs/DESIGN.md` | Why the server is shaped the way it is, and what is out of scope by decision. |
|
|
40
|
+
| `tests/fixtures/` | Recorded responses and the tool-schema snapshot. |
|
|
41
|
+
| `tests/test_tool_contract.py` | Tests over the tool surface as a client sees it. |
|
|
42
|
+
| `tests/test_security.py` | Every refusal against a steered model: hosts, addresses, sizes, files. |
|
|
43
|
+
| `tests/live_check.py` | The one script that talks to live sites, run by hand. Not collected. |
|
|
44
|
+
|
|
45
|
+
## What a change carries
|
|
46
|
+
|
|
47
|
+
**A test that fails without it.** Bug fixes especially: reproduce the bug as a
|
|
48
|
+
test first.
|
|
49
|
+
|
|
50
|
+
**Descriptions written for the model.** A tool's docstring is what a model
|
|
51
|
+
reads when choosing and calling it. The combined descriptions have a ceiling
|
|
52
|
+
(`DESCRIPTION_BUDGET` in `tests/test_tool_contract.py`), because they are sent
|
|
53
|
+
on every session. Raise it deliberately, in a pull request of its own. They
|
|
54
|
+
are dedented at import, so the budget measures the same text on every Python.
|
|
55
|
+
|
|
56
|
+
**The evidence distinction, kept.** A transcript, OCR text or index entry is a
|
|
57
|
+
lead to the page image, never the record. Tools that return text say so; a
|
|
58
|
+
contract test enforces it.
|
|
59
|
+
|
|
60
|
+
**A citation core on every item.** Any result that names an item carries its
|
|
61
|
+
public address and the institution, collection, title and identifier.
|
|
62
|
+
|
|
63
|
+
**A structured result, never an exception.** Every tool catches its failures
|
|
64
|
+
and returns an `error` envelope. A sweep test calls every tool with every site
|
|
65
|
+
failing and fails if one raises.
|
|
66
|
+
|
|
67
|
+
**Nothing that writes to a site, and nothing that works around a bot check.**
|
|
68
|
+
See [docs/DESIGN.md](docs/DESIGN.md#out-of-scope-by-decision).
|
|
69
|
+
|
|
70
|
+
**Nothing that widens what an argument can reach.** Treat every argument as
|
|
71
|
+
written by someone else's text. A new kind of host goes through the policy in
|
|
72
|
+
`instances.resolve`, a new place to write through `server._destination`, and
|
|
73
|
+
each comes with a test in `tests/test_security.py` showing that the refusals
|
|
74
|
+
still come before the harm. See
|
|
75
|
+
[docs/DESIGN.md](docs/DESIGN.md#which-hosts-a-tool-reaches).
|
|
76
|
+
|
|
77
|
+
**A new fixture recorded, not invented,** when a change depends on how a site
|
|
78
|
+
answers, with what was observed and when added to `docs/API-NOTES.md`. Keep
|
|
79
|
+
fixtures to historical records.
|
|
80
|
+
|
|
81
|
+
**The snapshot, when the surface changes.** Renaming or adding a tool or a
|
|
82
|
+
parameter fails the snapshot test on purpose. Regenerate it with
|
|
83
|
+
`uv run python -m tests.regen_tool_snapshot`, update the README tables, and add
|
|
84
|
+
a `CHANGELOG.md` entry.
|
|
85
|
+
|
|
86
|
+
## Adding a site to the curated list
|
|
87
|
+
|
|
88
|
+
1. Call `{base}/digital/bl/dmwebservices/index.php?q=dmGetCollectionList/json`
|
|
89
|
+
yourself. Record the number of collections and the day.
|
|
90
|
+
2. Prefer state archives, state libraries and statewide networks. Add a
|
|
91
|
+
university or public library only when it holds state or county records,
|
|
92
|
+
and say which collections in `holds`.
|
|
93
|
+
3. Read a few items' rights fields and put what they say about reuse in
|
|
94
|
+
`terms`.
|
|
95
|
+
4. Check the public host with Python (`uv run python -c "import httpx;
|
|
96
|
+
httpx.get('https://host/')"`). If its certificate chain is incomplete, add
|
|
97
|
+
the site's `cdmNNNNN.contentdm.oclc.org` address as `api_base`: the number
|
|
98
|
+
is in the `path` of every collection in the list.
|
|
99
|
+
5. Run `uv run python -m tests.live_check`: it calls every entry.
|
|
100
|
+
|
|
101
|
+
A site that has left CONTENTdm stays in the list with `status: moved` and a
|
|
102
|
+
`moved_to`, so a search can say where it went.
|
|
103
|
+
|
|
104
|
+
## Adding a platform
|
|
105
|
+
|
|
106
|
+
Write an adapter in `src/contentdm_mcp/adapters/` that implements
|
|
107
|
+
`adapters.base.Adapter`, register it in `ADAPTERS`, record its answers as
|
|
108
|
+
fixtures, and change the affected entries' `platform`. The tools should not
|
|
109
|
+
need to change; if they do, say why in the pull request.
|
|
110
|
+
|
|
111
|
+
## Releasing
|
|
112
|
+
|
|
113
|
+
A maintainer bumps `__version__` in `src/contentdm_mcp/__init__.py` and
|
|
114
|
+
both versions in `server.json`, moves the changelog's Unreleased entries under
|
|
115
|
+
the new version, and publishes a GitHub release tagged `v<version>`. The
|
|
116
|
+
release workflow builds the tag, publishes to PyPI by Trusted Publishing, and
|
|
117
|
+
lists the version in the MCP Registry.
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Ian Anderson
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,318 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: contentdm-mcp
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: MCP server for the record images and transcripts that US state archives and state libraries publish on OCLC's CONTENTdm: search one instance or a curated list of them, read an item and its pages, and download a page image. For genealogy and history.
|
|
5
|
+
Project-URL: Homepage, https://github.com/ianderso/contentdm-mcp
|
|
6
|
+
Project-URL: Repository, https://github.com/ianderso/contentdm-mcp
|
|
7
|
+
Project-URL: Issues, https://github.com/ianderso/contentdm-mcp/issues
|
|
8
|
+
Project-URL: Changelog, https://github.com/ianderso/contentdm-mcp/blob/main/CHANGELOG.md
|
|
9
|
+
Author: Ian Anderson
|
|
10
|
+
License-Expression: MIT
|
|
11
|
+
License-File: LICENSE
|
|
12
|
+
Keywords: archives,contentdm,digital-collections,family-history,genealogy,iiif,mcp,research,state-archives
|
|
13
|
+
Classifier: Development Status :: 4 - Beta
|
|
14
|
+
Classifier: Intended Audience :: End Users/Desktop
|
|
15
|
+
Classifier: Intended Audience :: Science/Research
|
|
16
|
+
Classifier: Operating System :: OS Independent
|
|
17
|
+
Classifier: Programming Language :: Python :: 3
|
|
18
|
+
Classifier: Programming Language :: Python :: 3 :: Only
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
20
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
21
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
22
|
+
Classifier: Topic :: Sociology :: Genealogy
|
|
23
|
+
Classifier: Topic :: Sociology :: History
|
|
24
|
+
Requires-Python: >=3.11
|
|
25
|
+
Requires-Dist: httpcore>=1.0
|
|
26
|
+
Requires-Dist: httpx<1,>=0.27
|
|
27
|
+
Requires-Dist: mcp<3,>=2.0.0
|
|
28
|
+
Requires-Dist: pydantic>=2.6
|
|
29
|
+
Requires-Dist: python-dotenv>=1.0
|
|
30
|
+
Requires-Dist: pyyaml>=6.0.1
|
|
31
|
+
Provides-Extra: dev
|
|
32
|
+
Requires-Dist: pytest-asyncio>=0.24; extra == 'dev'
|
|
33
|
+
Requires-Dist: pytest>=8.2; extra == 'dev'
|
|
34
|
+
Requires-Dist: respx>=0.21; extra == 'dev'
|
|
35
|
+
Requires-Dist: ruff>=0.9; extra == 'dev'
|
|
36
|
+
Description-Content-Type: text/markdown
|
|
37
|
+
|
|
38
|
+
# contentdm-mcp
|
|
39
|
+
|
|
40
|
+
[](https://github.com/ianderso/contentdm-mcp/actions/workflows/ci.yml)
|
|
41
|
+
[](https://pypi.org/project/contentdm-mcp/)
|
|
42
|
+
|
|
43
|
+
<!-- mcp-name: io.github.ianderso/contentdm-mcp -->
|
|
44
|
+
|
|
45
|
+
An [MCP](https://modelcontextprotocol.io) server for the **record images
|
|
46
|
+
that US state archives and state libraries publish on CONTENTdm**: death
|
|
47
|
+
certificates and their annual indexes, Confederate pension files, voter
|
|
48
|
+
registers, prison registers, county court and estate papers, letters and
|
|
49
|
+
diaries, with whatever transcript or OCR text the institution added.
|
|
50
|
+
|
|
51
|
+
[CONTENTdm](https://www.oclc.org/en/contentdm.html) is OCLC's hosted
|
|
52
|
+
digital-collections platform. The Alabama Department of Archives and History,
|
|
53
|
+
the Georgia Archives' Virtual Vault, the Tennessee State Library and Archives,
|
|
54
|
+
Ohio Memory, the Illinois Digital Archives, the Indiana State Library,
|
|
55
|
+
Missouri Digital Heritage and many more run on it, and every one of them
|
|
56
|
+
answers the same keyless JSON API and IIIF image service. The server ships a
|
|
57
|
+
curated list of those sites, searches one or several of them at once, reads
|
|
58
|
+
an item and its pages, and downloads a page image to read.
|
|
59
|
+
|
|
60
|
+
It works the way a careful genealogist does. **The page image is the
|
|
61
|
+
evidence.** A transcript or OCR text is a lead that says which page to read,
|
|
62
|
+
and an index card is a finding aid to the record behind it. Every result
|
|
63
|
+
carries the holding institution's item address and a citation core
|
|
64
|
+
(institution, collection, title, identifier, address), because the item page
|
|
65
|
+
is what gets cited, not this server.
|
|
66
|
+
|
|
67
|
+
Nothing here writes anywhere, and nothing here keeps a family tree. It sits
|
|
68
|
+
well beside [dpla-catalog-mcp](https://github.com/ianderso/dpla-catalog-mcp),
|
|
69
|
+
which finds an item across hundreds of collections and hands you the
|
|
70
|
+
institution's address (`get_item` takes that address directly), and
|
|
71
|
+
[nara-catalog-mcp](https://github.com/ianderso/nara-catalog-mcp) for federal
|
|
72
|
+
records.
|
|
73
|
+
|
|
74
|
+
This is an independent project. It is not affiliated with, endorsed by, or
|
|
75
|
+
supported by OCLC or any of the institutions whose sites it reads.
|
|
76
|
+
|
|
77
|
+
## Tools
|
|
78
|
+
|
|
79
|
+
The server publishes eight tools. All but `get_image` are read-only;
|
|
80
|
+
`get_image` writes one new file and never overwrites one.
|
|
81
|
+
|
|
82
|
+
**Finding**
|
|
83
|
+
|
|
84
|
+
| Tool | Purpose |
|
|
85
|
+
| --- | --- |
|
|
86
|
+
| `list_instances` | The curated sites: who runs each, what it holds for a genealogist, any reuse terms, and the day it was checked. Sites that have left CONTENTdm are listed with where they went. Makes no request. |
|
|
87
|
+
| `list_collections` | A site's collections and the alias each is searched by. |
|
|
88
|
+
| `get_collection` | One collection's fields: what `search` can name in `field`, and which field holds the transcript or OCR text. |
|
|
89
|
+
| `search` | Search one site, or several side by side. Says which sites answered, failed or timed out, so a negative is bounded. `page_hits` returns the page inside a volume where a text match sits. |
|
|
90
|
+
|
|
91
|
+
**Reading**
|
|
92
|
+
|
|
93
|
+
| Tool | Purpose |
|
|
94
|
+
| --- | --- |
|
|
95
|
+
| `get_item` | One item: metadata, transcript or OCR text, the institution's own "cite as" line and rights statement, and the citation core. For a page, which object and page number it is. Takes an item address, such as DPLA's `isShownAt`, on a site it may reach. |
|
|
96
|
+
| `get_pages` | A compound object's pages in order, with `query` to mark the pages that match. |
|
|
97
|
+
| `get_image` | Download one page image through the site's IIIF service, sized to fit, including any page of a multi-page PDF. Saves a new `.jpg`, never a hidden file or one under `~/Library`. |
|
|
98
|
+
| `cache_status` | This session's requests, by site, and cache use. Makes no request. |
|
|
99
|
+
|
|
100
|
+
## The curated sites
|
|
101
|
+
|
|
102
|
+
`list_instances` gives the whole list. On 2026-10-06, 18 sites answered, and 7
|
|
103
|
+
entries record institutions that have left CONTENTdm, so a search can say
|
|
104
|
+
where they went:
|
|
105
|
+
|
|
106
|
+
| State | Site | Status |
|
|
107
|
+
| --- | --- | --- |
|
|
108
|
+
| AL | Alabama Department of Archives and History | supported |
|
|
109
|
+
| AK | Alaska's Digital Archives | supported |
|
|
110
|
+
| AR | Arkansas State Archives | supported |
|
|
111
|
+
| CT | Connecticut State Library | supported |
|
|
112
|
+
| DE | Delaware Division of Libraries, with the Delaware Public Archives | supported |
|
|
113
|
+
| GA | Georgia Archives, Virtual Vault | supported |
|
|
114
|
+
| IL | Illinois Digital Archives | supported |
|
|
115
|
+
| IN | Indiana State Library (Indiana Memory); Ball State University | supported |
|
|
116
|
+
| KY | Kentucky Digital Library | supported |
|
|
117
|
+
| MO | Missouri Digital Heritage | supported |
|
|
118
|
+
| ND | Digital Horizons | supported |
|
|
119
|
+
| NM | New Mexico Digital Collections, with the State Records Center and Archives | supported |
|
|
120
|
+
| NV | Nevada State Library, Archives and Public Records | supported |
|
|
121
|
+
| OH | Ohio Memory | supported |
|
|
122
|
+
| OK | Oklahoma Department of Libraries, with the Oklahoma State Archives | supported |
|
|
123
|
+
| TN | Tennessee State Library and Archives (TeVA) | supported |
|
|
124
|
+
| WA | Washington Rural Heritage | supported |
|
|
125
|
+
| AZ | Arizona Memory Project | left CONTENTdm; behind a bot check |
|
|
126
|
+
| ID, SD | State archives' digital collections | moved to Preservica |
|
|
127
|
+
| MT | Montana Memory Project | moved to Recollect |
|
|
128
|
+
| NC | State Archives of North Carolina | moved to Quartex |
|
|
129
|
+
| PA | POWER Library | moved to Islandora |
|
|
130
|
+
| WI | Wisconsin Historical Society | moved to its own site |
|
|
131
|
+
|
|
132
|
+
Any other CONTENTdm site works too, at its `cdmNNNNN.contentdm.oclc.org`
|
|
133
|
+
address: pass `https://cdmNNNNN.contentdm.oclc.org` as `instance`. Every
|
|
134
|
+
classic site answers there, whatever its own domain, and the source of its
|
|
135
|
+
pages names the number (`"cdmServerUrl": "serverNNNNN.contentdm.oclc.org"`).
|
|
136
|
+
There are more than 675 of them, and no registry. To use a site by its own
|
|
137
|
+
domain, add it to `CONTENTDM_EXTRA_INSTANCES`; a tool argument cannot, for the
|
|
138
|
+
reason under [Security](#security).
|
|
139
|
+
|
|
140
|
+
## Setup
|
|
141
|
+
|
|
142
|
+
You need Python 3.11 or later and [uv](https://docs.astral.sh/uv/). There is
|
|
143
|
+
no key to request.
|
|
144
|
+
|
|
145
|
+
**Without cloning.** `uvx` fetches it from PyPI and runs it in one step:
|
|
146
|
+
|
|
147
|
+
```bash
|
|
148
|
+
uvx contentdm-mcp
|
|
149
|
+
```
|
|
150
|
+
|
|
151
|
+
**From a clone**, which is what you want if you will change it:
|
|
152
|
+
|
|
153
|
+
```bash
|
|
154
|
+
git clone https://github.com/ianderso/contentdm-mcp
|
|
155
|
+
cd contentdm-mcp
|
|
156
|
+
uv sync
|
|
157
|
+
uv run contentdm-mcp # stdio server, usually launched by the client
|
|
158
|
+
```
|
|
159
|
+
|
|
160
|
+
Either way the server speaks MCP over stdio, so you will normally let an MCP
|
|
161
|
+
client start it rather than run it by hand.
|
|
162
|
+
|
|
163
|
+
### Claude Desktop
|
|
164
|
+
|
|
165
|
+
```json
|
|
166
|
+
{
|
|
167
|
+
"mcpServers": {
|
|
168
|
+
"contentdm": {
|
|
169
|
+
"command": "uvx",
|
|
170
|
+
"args": ["contentdm-mcp"]
|
|
171
|
+
}
|
|
172
|
+
}
|
|
173
|
+
}
|
|
174
|
+
```
|
|
175
|
+
|
|
176
|
+
A desktop app does not always inherit your shell's `PATH`. If the server fails
|
|
177
|
+
to start because `uvx` cannot be found, give the full path that `which uvx`
|
|
178
|
+
prints as the `command`.
|
|
179
|
+
|
|
180
|
+
### Claude Code
|
|
181
|
+
|
|
182
|
+
```bash
|
|
183
|
+
claude mcp add contentdm -- uvx contentdm-mcp
|
|
184
|
+
```
|
|
185
|
+
|
|
186
|
+
## Configuration
|
|
187
|
+
|
|
188
|
+
Nothing is required. A `.env` file in the directory the server starts in
|
|
189
|
+
supplies anything the environment does not; only that directory is read.
|
|
190
|
+
|
|
191
|
+
| Variable | Meaning |
|
|
192
|
+
| --- | --- |
|
|
193
|
+
| `CONTENTDM_CACHE_DIR` | Response cache directory. Default `~/.cache/contentdm-mcp`. |
|
|
194
|
+
| `CONTENTDM_TIMEOUT` | HTTP timeout in seconds for one request. Default 30. Image downloads get 120 to read. |
|
|
195
|
+
| `CONTENTDM_MIN_INTERVAL` | Least seconds between two requests to one site. Default 1, and never below 0.5. |
|
|
196
|
+
| `CONTENTDM_CONTACT` | An email address or URL added to the User-Agent, so an institution can reach you if your use causes trouble. Optional, and courteous. |
|
|
197
|
+
| `CONTENTDM_EXTRA_INSTANCES` | More sites a tool may reach by their own domain: https base URLs separated by commas, or the absolute path of a YAML file whose entries are shaped like [`instances.yaml`](src/contentdm_mcp/instances.yaml)'s (only `base_url` is required). `list_instances` lists them; a search of every site leaves them out. |
|
|
198
|
+
| `CONTENTDM_DOWNLOAD_DIR` | An existing folder. When set, `get_image` saves only inside it. Set it to save into an iCloud Drive folder, which lives under `~/Library`. |
|
|
199
|
+
|
|
200
|
+
An unusable value is reported on the first tool call as a `not_configured`
|
|
201
|
+
result naming the variable.
|
|
202
|
+
|
|
203
|
+
## Being a good guest
|
|
204
|
+
|
|
205
|
+
Each site is one institution's server. The client sends one request at a time
|
|
206
|
+
to each site, at least a second apart; a search across several sites runs
|
|
207
|
+
them side by side, six at a time, so no site sees more than one caller. Two
|
|
208
|
+
identical calls in flight share one request. Collection lists and field
|
|
209
|
+
definitions are cached for 30 days, items for 7, searches for a day. A 429, a
|
|
210
|
+
5xx or a dropped connection gets one retry, honouring `Retry-After`. The
|
|
211
|
+
User-Agent names the package, its version and this repository.
|
|
212
|
+
|
|
213
|
+
`robots.txt` on these sites disallows the website's search pages, not the
|
|
214
|
+
API this server uses. Each institution's own terms govern what you do with
|
|
215
|
+
its images: `list_instances` and `get_item` pass on what the sites say, and
|
|
216
|
+
several ask for permission before publication.
|
|
217
|
+
|
|
218
|
+
## How to read what comes back
|
|
219
|
+
|
|
220
|
+
- **A search is only as wide as `answered`.** `failed`, `timed_out` and
|
|
221
|
+
`not_searched` list the sites a negative does not cover. A site that
|
|
222
|
+
redirects, asks for a bot check or lists no collections has usually moved.
|
|
223
|
+
- **Search covers metadata and text fields, never the image.** Many record
|
|
224
|
+
images have no text at all, so a name can be on a page that no search finds.
|
|
225
|
+
Browse by collection, or by the index volume, instead.
|
|
226
|
+
- **A compound object's text is on its pages.** Volumes, files and newspaper
|
|
227
|
+
issues are compound objects; their own transcript field is usually empty.
|
|
228
|
+
Use `page_hits=true`, or `get_pages` with a `query`, to find the page.
|
|
229
|
+
- **Field names are per collection.** "Name of Father" is `namea` in one
|
|
230
|
+
collection and `father` in another. `get_collection` says which nick holds
|
|
231
|
+
what, and which field is the transcript: that is marked by type, because
|
|
232
|
+
the name differs from site to site.
|
|
233
|
+
- **"Transcript" can be machine OCR.** The Tennessee vital-records indexes
|
|
234
|
+
carry OCR under that label. Read the image.
|
|
235
|
+
- **`exact` is a phrase search** (the words together, in order), not
|
|
236
|
+
whole-field equality. `all` needs every word, `any` one of them.
|
|
237
|
+
- **A PDF item can hold many pages.** Its IIIF image shows the first;
|
|
238
|
+
`get_image` with `pdf_page` reaches the others.
|
|
239
|
+
- **`page` is a position in the object,** counted from 1, not the number
|
|
240
|
+
printed on the scan: the entry for Sgt. Alvin C. York is on page 461 of
|
|
241
|
+
volume 8 of Tennessee's 1960-1964 death index, a scan named `54784_460` and
|
|
242
|
+
stamped 457. Cite the printed number too when it differs.
|
|
243
|
+
- **Cite the item page.** Use the `citation` each result carries, and the
|
|
244
|
+
institution's own `cite_as` line where `get_item` finds one (Georgia,
|
|
245
|
+
Oklahoma and Alaska items have them). Record the identifier
|
|
246
|
+
(`alias:pointer`) so the item can be found again.
|
|
247
|
+
|
|
248
|
+
## Deliberately not here
|
|
249
|
+
|
|
250
|
+
- **Writing to any site.** CONTENTdm's write functions need a staff login.
|
|
251
|
+
- **Sites on other platforms.** The new CONTENTdm (launched 2026-09-22) has
|
|
252
|
+
no published public API, and Quartex's is undocumented. The tools speak to
|
|
253
|
+
an adapter layer, so either can be added without changing them; see
|
|
254
|
+
[docs/DESIGN.md](docs/DESIGN.md).
|
|
255
|
+
- **Working around bot checks.** A site that answers with a challenge is
|
|
256
|
+
reported as blocked and left alone.
|
|
257
|
+
|
|
258
|
+
## Security
|
|
259
|
+
|
|
260
|
+
Tool arguments are written by a model, and the model reads text this server
|
|
261
|
+
does not control: titles, transcripts, DPLA records, web pages. The server
|
|
262
|
+
assumes that text can steer the model, and limits what a steered model can
|
|
263
|
+
make it do.
|
|
264
|
+
|
|
265
|
+
- **Which hosts.** A tool reaches the curated sites, any address under
|
|
266
|
+
`contentdm.oclc.org` (OCLC's own servers), and the sites you list in
|
|
267
|
+
`CONTENTDM_EXTRA_INSTANCES`. Any other host is refused before it is even
|
|
268
|
+
looked up, because a DNS query for `secret.attacker.example` would already
|
|
269
|
+
deliver the name; the refusal says how to reach the site instead.
|
|
270
|
+
- **Which addresses.** Each connection is checked where it is made: a name
|
|
271
|
+
that leads to a private, loopback, link-local, CGNAT, multicast, reserved or
|
|
272
|
+
unspecified address, IPv4 or IPv6, is refused, and the connection goes to
|
|
273
|
+
the address that was checked. Redirects are followed only between an
|
|
274
|
+
instance's own hosts, and are checked again. Proxy settings in the
|
|
275
|
+
environment are not used.
|
|
276
|
+
- **How much.** A JSON answer over 10 MB, or an image over 60 MB, is refused
|
|
277
|
+
as it streams in.
|
|
278
|
+
- **Which files.** `get_image` creates one new file and never overwrites one.
|
|
279
|
+
The bytes must be an image, judged by their first bytes rather than the
|
|
280
|
+
Content-Type, and of the format the file's suffix names: `.jpg` or `.jpeg`.
|
|
281
|
+
Never a hidden file or folder, never under `~/Library`, and with
|
|
282
|
+
`CONTENTDM_DOWNLOAD_DIR` set, never outside it, all judged after links are
|
|
283
|
+
resolved. A refused download leaves nothing on disk.
|
|
284
|
+
- **Arguments are validated** (collection aliases, field nicks, numeric
|
|
285
|
+
pointers) before they reach a request, and search words are stripped of the
|
|
286
|
+
API's own syntax characters.
|
|
287
|
+
- **Site text is untrusted.** Titles, descriptions and transcripts reach the
|
|
288
|
+
model verbatim. The server's instructions tell the model to treat that text
|
|
289
|
+
as material to weigh, never as instructions; the model still decides, so
|
|
290
|
+
review what it proposes to do.
|
|
291
|
+
|
|
292
|
+
To report a vulnerability, see [SECURITY.md](SECURITY.md).
|
|
293
|
+
|
|
294
|
+
## Development
|
|
295
|
+
|
|
296
|
+
```bash
|
|
297
|
+
uv sync --extra dev
|
|
298
|
+
uv run pytest # mocked with respx; never touches a site
|
|
299
|
+
uv run ruff check .
|
|
300
|
+
uv run ruff format --check .
|
|
301
|
+
uv run python -m tests.live_check # paced calls to the live sites
|
|
302
|
+
```
|
|
303
|
+
|
|
304
|
+
The live check asks the sites what the recorded fixtures cannot: whether
|
|
305
|
+
their answers still have the shape the server reads, and whether every
|
|
306
|
+
curated site still answers. See [CONTRIBUTING.md](CONTRIBUTING.md) for how the
|
|
307
|
+
suite is organised, [docs/API-NOTES.md](docs/API-NOTES.md) for what was
|
|
308
|
+
observed of the API and when, and [docs/DESIGN.md](docs/DESIGN.md) for why the
|
|
309
|
+
server is shaped this way.
|
|
310
|
+
|
|
311
|
+
## Credits
|
|
312
|
+
|
|
313
|
+
The images and descriptions belong to the institutions that publish them.
|
|
314
|
+
CONTENTdm is a product of [OCLC](https://www.oclc.org).
|
|
315
|
+
|
|
316
|
+
## License
|
|
317
|
+
|
|
318
|
+
[MIT](LICENSE).
|