siemens-docs-mcp 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (30) hide show
  1. siemens_docs_mcp-0.1.0/LICENSE +21 -0
  2. siemens_docs_mcp-0.1.0/PKG-INFO +180 -0
  3. siemens_docs_mcp-0.1.0/README.md +154 -0
  4. siemens_docs_mcp-0.1.0/pyproject.toml +48 -0
  5. siemens_docs_mcp-0.1.0/setup.cfg +4 -0
  6. siemens_docs_mcp-0.1.0/siemens_docs_mcp/__init__.py +1 -0
  7. siemens_docs_mcp-0.1.0/siemens_docs_mcp/catalog.py +254 -0
  8. siemens_docs_mcp-0.1.0/siemens_docs_mcp/cli.py +232 -0
  9. siemens_docs_mcp-0.1.0/siemens_docs_mcp/client.py +177 -0
  10. siemens_docs_mcp-0.1.0/siemens_docs_mcp/content.py +63 -0
  11. siemens_docs_mcp-0.1.0/siemens_docs_mcp/converter.py +383 -0
  12. siemens_docs_mcp-0.1.0/siemens_docs_mcp/server.py +130 -0
  13. siemens_docs_mcp-0.1.0/siemens_docs_mcp/toc.py +112 -0
  14. siemens_docs_mcp-0.1.0/siemens_docs_mcp/tools.py +167 -0
  15. siemens_docs_mcp-0.1.0/siemens_docs_mcp/writer.py +80 -0
  16. siemens_docs_mcp-0.1.0/siemens_docs_mcp.egg-info/PKG-INFO +180 -0
  17. siemens_docs_mcp-0.1.0/siemens_docs_mcp.egg-info/SOURCES.txt +28 -0
  18. siemens_docs_mcp-0.1.0/siemens_docs_mcp.egg-info/dependency_links.txt +1 -0
  19. siemens_docs_mcp-0.1.0/siemens_docs_mcp.egg-info/entry_points.txt +3 -0
  20. siemens_docs_mcp-0.1.0/siemens_docs_mcp.egg-info/requires.txt +6 -0
  21. siemens_docs_mcp-0.1.0/siemens_docs_mcp.egg-info/top_level.txt +1 -0
  22. siemens_docs_mcp-0.1.0/tests/test_catalog.py +267 -0
  23. siemens_docs_mcp-0.1.0/tests/test_cli.py +66 -0
  24. siemens_docs_mcp-0.1.0/tests/test_client.py +149 -0
  25. siemens_docs_mcp-0.1.0/tests/test_converter.py +44 -0
  26. siemens_docs_mcp-0.1.0/tests/test_live.py +20 -0
  27. siemens_docs_mcp-0.1.0/tests/test_packaging.py +7 -0
  28. siemens_docs_mcp-0.1.0/tests/test_server.py +80 -0
  29. siemens_docs_mcp-0.1.0/tests/test_toc_writer.py +49 -0
  30. siemens_docs_mcp-0.1.0/tests/test_tools.py +233 -0
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Łukasz Czarnacki
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,180 @@
1
+ Metadata-Version: 2.4
2
+ Name: siemens-docs-mcp
3
+ Version: 0.1.0
4
+ Summary: Search and read Siemens Fluid Topics documentation via MCP or export it to Markdown
5
+ Author: Łukasz Czarnacki
6
+ Project-URL: Homepage, https://github.com/Czarnak/siemens-docs-mcp
7
+ Project-URL: Issues, https://github.com/Czarnak/siemens-docs-mcp/issues
8
+ Keywords: mcp,siemens,tia-portal,fluid-topics,documentation
9
+ Classifier: Programming Language :: Python :: 3
10
+ Classifier: Programming Language :: Python :: 3.11
11
+ Classifier: Programming Language :: Python :: 3.12
12
+ Classifier: Programming Language :: Python :: 3.13
13
+ Classifier: Programming Language :: Python :: 3.14
14
+ Classifier: Operating System :: OS Independent
15
+ Classifier: Topic :: Documentation
16
+ Requires-Python: >=3.11
17
+ Description-Content-Type: text/markdown
18
+ License-File: LICENSE
19
+ Requires-Dist: httpx>=0.28.1
20
+ Requires-Dist: beautifulsoup4>=4.15.0
21
+ Requires-Dist: markdownify>=1.2.3
22
+ Requires-Dist: PyYAML>=6.0.3
23
+ Requires-Dist: lxml>=6.1.3
24
+ Requires-Dist: mcp<3,>=2.3.0
25
+ Dynamic: license-file
26
+
27
+ # Siemens Docs MCP
28
+
29
+ An MCP server that lets an AI assistant search and read Siemens documentation live — any publication on `docs.tia.siemens.cloud` (TIA Portal, STEP 7, WinCC Unified, Openness, …) and `docs.industrial-operations-x.siemens.cloud` (Industrial Operations X) — plus a CLI that exports a whole publication to a tree of Markdown files.
30
+
31
+ Built to solve the problem of documentation portals that only offer low-quality PDF exports or JavaScript-rendered web views, making the content difficult to search, reference, or feed to AI tools.
32
+
33
+ ---
34
+
35
+ ## Installation
36
+
37
+ ### PyPI
38
+
39
+ ```bash
40
+ pip install siemens-docs-mcp
41
+ ```
42
+
43
+ This installs two commands: `siemens-docs-mcp` (the MCP server) and `siemens-docs-export` (the Markdown exporter). Requires Python 3.11+.
44
+
45
+ ### From source
46
+
47
+ ```bash
48
+ git clone https://github.com/Czarnak/siemens-docs-mcp
49
+ cd siemens-docs-mcp
50
+ python -m venv .venv
51
+ source .venv/bin/activate # Windows: .venv\Scripts\activate
52
+ pip install -e .
53
+ # development (pytest, ruff, pip-audit; needs pip >= 25.1):
54
+ pip install -e . --group dev
55
+ ```
56
+
57
+ ---
58
+
59
+ ## MCP server
60
+
61
+ The server speaks MCP over stdio. Register it with Claude Code:
62
+
63
+ ```bash
64
+ # from PyPI, no manual install (needs uv)
65
+ claude mcp add siemens-docs -- uvx siemens-docs-mcp
66
+
67
+ # or an installed copy
68
+ claude mcp add siemens-docs -- siemens-docs-mcp
69
+ # from a source checkout: <repo>/.venv/Scripts/siemens-docs-mcp (Linux/macOS: <repo>/.venv/bin/siemens-docs-mcp)
70
+ ```
71
+
72
+ Pass settings with `-e`, e.g. `claude mcp add siemens-docs -e SIEMENS_DOCS_LOCALE=de-DE -- uvx siemens-docs-mcp`.
73
+
74
+ Tools:
75
+
76
+ | Tool | Purpose |
77
+ |---------------------|---------------------------------------------------------------------------|
78
+ | `search_docs` | Full-text search (filter by `product`, `version`, `locale`, `host`). |
79
+ | `read_page` | Read one page as Markdown; page through long pages with `offset`. |
80
+ | `get_toc` | Table of contents of a publication or of the subtree under a topic URL. |
81
+ | `list_publications` | List publications; discover valid `product` / `version` values. |
82
+
83
+ Any reader URL returned by a tool (or copied from the browser) is valid input to `read_page` and `get_toc`.
84
+
85
+ Environment variables:
86
+
87
+ | Variable | Default | Meaning |
88
+ |------------------------------|----------|------------------------------------------------------------------|
89
+ | `SIEMENS_DOCS_HOSTS` | (none) | Comma-separated extra hosts, appended to the built-in two (`docs.tia.siemens.cloud`, `docs.industrial-operations-x.siemens.cloud`). |
90
+ | `SIEMENS_DOCS_LOCALE` | `en-US` | Default locale for search and listings. |
91
+ | `SIEMENS_DOCS_MIN_INTERVAL` | `0.3` | Minimum seconds between requests to a host. |
92
+
93
+ The publication catalog and TOCs are cached in memory (not on disk), so the first call per host takes a few seconds.
94
+
95
+ ---
96
+
97
+ ## Markdown export (CLI)
98
+
99
+ Export a whole publication to Markdown — one file per page, folders mirroring the navigation hierarchy. Create a config with any reader URL of the publication (a topic URL exports the whole publication):
100
+
101
+ ```yaml
102
+ # my_docs.yaml
103
+ name: tia_openness_v21
104
+ url: "https://docs.tia.siemens.cloud/r/en-us/v21/tia-portal-openness-api-for-automation-of-engineering-workflows"
105
+ output_dir: "output/tia_openness_v21"
106
+ ```
107
+
108
+ ```bash
109
+ # Preview pages without writing files
110
+ siemens-docs-export my_docs.yaml --dry-run
111
+
112
+ # Export everything
113
+ siemens-docs-export my_docs.yaml
114
+
115
+ # Export a single page (for testing output quality)
116
+ siemens-docs-export my_docs.yaml --page cybersecurity-information
117
+
118
+ # Override the output directory / verbose logging
119
+ siemens-docs-export my_docs.yaml --output /tmp/docs --verbose
120
+ ```
121
+
122
+ A ready-made config lives in [`configs/`](https://github.com/Czarnak/siemens-docs-mcp/tree/main/configs) in the repository.
123
+
124
+ ### Output structure
125
+
126
+ ```
127
+ output/tia_openness_v21/
128
+ ├── index.md ← root page
129
+ ├── cybersecurity-information.md
130
+ ├── what-s-new-in-tia-portal-openness.md
131
+ ├── basics/
132
+ │ ├── basics.md
133
+ │ └── ...
134
+ ├── tia-portal-openness-api/
135
+ │ ├── tia-portal-openness-object/
136
+ │ │ └── ...
137
+ │ └── ...
138
+ └── ...
139
+ ```
140
+
141
+ ### Configuration reference
142
+
143
+ | Key | Required | Description |
144
+ |--------------|----------|-----------------------------------------------------------------------------|
145
+ | `name` | No | Human-readable label shown in log output. |
146
+ | `url` | Yes* | Any reader URL of the publication (resolved to its map automatically). |
147
+ | `api_base` | Yes* | Legacy: root URL of the Fluidtopics instance (use with `map_id`). |
148
+ | `map_id` | Yes* | Legacy: Fluidtopics map identifier (use with `api_base`). |
149
+ | `output_dir` | Yes | Directory where Markdown files will be written. |
150
+
151
+ \* Either `url`, or `api_base` + `map_id`. If both are present, `url` wins.
152
+
153
+ ---
154
+
155
+ ## How it works
156
+
157
+ Fluidtopics (the platform behind both portals) exposes a REST API that the browser SPA uses internally. This package calls that API directly — no browser automation:
158
+
159
+ 1. **Catalog** — `GET /api/khub/maps` lists every publication; reader URLs are resolved against it.
160
+ 2. **Search** — `POST /api/khub/clustered-search` with product/version/locale filters.
161
+ 3. **TOC** — `GET /api/khub/maps/{mapId}/pages` returns the navigation tree.
162
+ 4. **Content** — `GET /api/khub/maps/{mapId}/topics/{contentId}/content` returns raw HTML, converted to Markdown with [markdownify](https://github.com/matthewwithanm/python-markdownify).
163
+
164
+ Requests are throttled per host and retried once on 401/403/429/5xx.
165
+
166
+ > **Note:** Only Fluidtopics-based portals are supported. Other platforms (MadCap Flare, Paligo, etc.) would need a different adapter.
167
+
168
+ ---
169
+
170
+ ## Development
171
+
172
+ ```bash
173
+ python -m pytest -q # offline tests
174
+ python -m pytest -m live -q # live smoke tests against the real hosts
175
+ ruff check .
176
+ ```
177
+
178
+ ---
179
+
180
+ *Created with Claude AI*
@@ -0,0 +1,154 @@
1
+ # Siemens Docs MCP
2
+
3
+ An MCP server that lets an AI assistant search and read Siemens documentation live — any publication on `docs.tia.siemens.cloud` (TIA Portal, STEP 7, WinCC Unified, Openness, …) and `docs.industrial-operations-x.siemens.cloud` (Industrial Operations X) — plus a CLI that exports a whole publication to a tree of Markdown files.
4
+
5
+ Built to solve the problem of documentation portals that only offer low-quality PDF exports or JavaScript-rendered web views, making the content difficult to search, reference, or feed to AI tools.
6
+
7
+ ---
8
+
9
+ ## Installation
10
+
11
+ ### PyPI
12
+
13
+ ```bash
14
+ pip install siemens-docs-mcp
15
+ ```
16
+
17
+ This installs two commands: `siemens-docs-mcp` (the MCP server) and `siemens-docs-export` (the Markdown exporter). Requires Python 3.11+.
18
+
19
+ ### From source
20
+
21
+ ```bash
22
+ git clone https://github.com/Czarnak/siemens-docs-mcp
23
+ cd siemens-docs-mcp
24
+ python -m venv .venv
25
+ source .venv/bin/activate # Windows: .venv\Scripts\activate
26
+ pip install -e .
27
+ # development (pytest, ruff, pip-audit; needs pip >= 25.1):
28
+ pip install -e . --group dev
29
+ ```
30
+
31
+ ---
32
+
33
+ ## MCP server
34
+
35
+ The server speaks MCP over stdio. Register it with Claude Code:
36
+
37
+ ```bash
38
+ # from PyPI, no manual install (needs uv)
39
+ claude mcp add siemens-docs -- uvx siemens-docs-mcp
40
+
41
+ # or an installed copy
42
+ claude mcp add siemens-docs -- siemens-docs-mcp
43
+ # from a source checkout: <repo>/.venv/Scripts/siemens-docs-mcp (Linux/macOS: <repo>/.venv/bin/siemens-docs-mcp)
44
+ ```
45
+
46
+ Pass settings with `-e`, e.g. `claude mcp add siemens-docs -e SIEMENS_DOCS_LOCALE=de-DE -- uvx siemens-docs-mcp`.
47
+
48
+ Tools:
49
+
50
+ | Tool | Purpose |
51
+ |---------------------|---------------------------------------------------------------------------|
52
+ | `search_docs` | Full-text search (filter by `product`, `version`, `locale`, `host`). |
53
+ | `read_page` | Read one page as Markdown; page through long pages with `offset`. |
54
+ | `get_toc` | Table of contents of a publication or of the subtree under a topic URL. |
55
+ | `list_publications` | List publications; discover valid `product` / `version` values. |
56
+
57
+ Any reader URL returned by a tool (or copied from the browser) is valid input to `read_page` and `get_toc`.
58
+
59
+ Environment variables:
60
+
61
+ | Variable | Default | Meaning |
62
+ |------------------------------|----------|------------------------------------------------------------------|
63
+ | `SIEMENS_DOCS_HOSTS` | (none) | Comma-separated extra hosts, appended to the built-in two (`docs.tia.siemens.cloud`, `docs.industrial-operations-x.siemens.cloud`). |
64
+ | `SIEMENS_DOCS_LOCALE` | `en-US` | Default locale for search and listings. |
65
+ | `SIEMENS_DOCS_MIN_INTERVAL` | `0.3` | Minimum seconds between requests to a host. |
66
+
67
+ The publication catalog and TOCs are cached in memory (not on disk), so the first call per host takes a few seconds.
68
+
69
+ ---
70
+
71
+ ## Markdown export (CLI)
72
+
73
+ Export a whole publication to Markdown — one file per page, folders mirroring the navigation hierarchy. Create a config with any reader URL of the publication (a topic URL exports the whole publication):
74
+
75
+ ```yaml
76
+ # my_docs.yaml
77
+ name: tia_openness_v21
78
+ url: "https://docs.tia.siemens.cloud/r/en-us/v21/tia-portal-openness-api-for-automation-of-engineering-workflows"
79
+ output_dir: "output/tia_openness_v21"
80
+ ```
81
+
82
+ ```bash
83
+ # Preview pages without writing files
84
+ siemens-docs-export my_docs.yaml --dry-run
85
+
86
+ # Export everything
87
+ siemens-docs-export my_docs.yaml
88
+
89
+ # Export a single page (for testing output quality)
90
+ siemens-docs-export my_docs.yaml --page cybersecurity-information
91
+
92
+ # Override the output directory / verbose logging
93
+ siemens-docs-export my_docs.yaml --output /tmp/docs --verbose
94
+ ```
95
+
96
+ A ready-made config lives in [`configs/`](https://github.com/Czarnak/siemens-docs-mcp/tree/main/configs) in the repository.
97
+
98
+ ### Output structure
99
+
100
+ ```
101
+ output/tia_openness_v21/
102
+ ├── index.md ← root page
103
+ ├── cybersecurity-information.md
104
+ ├── what-s-new-in-tia-portal-openness.md
105
+ ├── basics/
106
+ │ ├── basics.md
107
+ │ └── ...
108
+ ├── tia-portal-openness-api/
109
+ │ ├── tia-portal-openness-object/
110
+ │ │ └── ...
111
+ │ └── ...
112
+ └── ...
113
+ ```
114
+
115
+ ### Configuration reference
116
+
117
+ | Key | Required | Description |
118
+ |--------------|----------|-----------------------------------------------------------------------------|
119
+ | `name` | No | Human-readable label shown in log output. |
120
+ | `url` | Yes* | Any reader URL of the publication (resolved to its map automatically). |
121
+ | `api_base` | Yes* | Legacy: root URL of the Fluidtopics instance (use with `map_id`). |
122
+ | `map_id` | Yes* | Legacy: Fluidtopics map identifier (use with `api_base`). |
123
+ | `output_dir` | Yes | Directory where Markdown files will be written. |
124
+
125
+ \* Either `url`, or `api_base` + `map_id`. If both are present, `url` wins.
126
+
127
+ ---
128
+
129
+ ## How it works
130
+
131
+ Fluidtopics (the platform behind both portals) exposes a REST API that the browser SPA uses internally. This package calls that API directly — no browser automation:
132
+
133
+ 1. **Catalog** — `GET /api/khub/maps` lists every publication; reader URLs are resolved against it.
134
+ 2. **Search** — `POST /api/khub/clustered-search` with product/version/locale filters.
135
+ 3. **TOC** — `GET /api/khub/maps/{mapId}/pages` returns the navigation tree.
136
+ 4. **Content** — `GET /api/khub/maps/{mapId}/topics/{contentId}/content` returns raw HTML, converted to Markdown with [markdownify](https://github.com/matthewwithanm/python-markdownify).
137
+
138
+ Requests are throttled per host and retried once on 401/403/429/5xx.
139
+
140
+ > **Note:** Only Fluidtopics-based portals are supported. Other platforms (MadCap Flare, Paligo, etc.) would need a different adapter.
141
+
142
+ ---
143
+
144
+ ## Development
145
+
146
+ ```bash
147
+ python -m pytest -q # offline tests
148
+ python -m pytest -m live -q # live smoke tests against the real hosts
149
+ ruff check .
150
+ ```
151
+
152
+ ---
153
+
154
+ *Created with Claude AI*
@@ -0,0 +1,48 @@
1
+ [build-system]
2
+ requires = ["setuptools>=84.0.0"]
3
+ build-backend = "setuptools.build_meta"
4
+
5
+ [project]
6
+ name = "siemens-docs-mcp"
7
+ version = "0.1.0"
8
+ description = "Search and read Siemens Fluid Topics documentation via MCP or export it to Markdown"
9
+ readme = "README.md"
10
+ requires-python = ">=3.11"
11
+ authors = [{ name = "Łukasz Czarnacki" }]
12
+ keywords = ["mcp", "siemens", "tia-portal", "fluid-topics", "documentation"]
13
+ classifiers = [
14
+ "Programming Language :: Python :: 3",
15
+ "Programming Language :: Python :: 3.11",
16
+ "Programming Language :: Python :: 3.12",
17
+ "Programming Language :: Python :: 3.13",
18
+ "Programming Language :: Python :: 3.14",
19
+ "Operating System :: OS Independent",
20
+ "Topic :: Documentation",
21
+ ]
22
+ dependencies = [
23
+ "httpx>=0.28.1",
24
+ "beautifulsoup4>=4.15.0",
25
+ "markdownify>=1.2.3",
26
+ "PyYAML>=6.0.3",
27
+ "lxml>=6.1.3",
28
+ "mcp>=2.3.0,<3",
29
+ ]
30
+
31
+ [project.urls]
32
+ Homepage = "https://github.com/Czarnak/siemens-docs-mcp"
33
+ Issues = "https://github.com/Czarnak/siemens-docs-mcp/issues"
34
+
35
+ [project.scripts]
36
+ siemens-docs-mcp = "siemens_docs_mcp.server:main"
37
+ siemens-docs-export = "siemens_docs_mcp.cli:main"
38
+
39
+ [dependency-groups]
40
+ dev = ["pytest>=9.1.1", "pytest-cov>=7.1.0", "ruff>=0.16.10", "pip-audit>=2.10.1"]
41
+
42
+ [tool.setuptools]
43
+ packages = ["siemens_docs_mcp"]
44
+
45
+ [tool.pytest.ini_options]
46
+ testpaths = ["tests"]
47
+ addopts = '-m "not live"'
48
+ markers = ["live: hits real Fluid Topics hosts (run with: pytest -m live)"]
@@ -0,0 +1,4 @@
1
+ [egg_info]
2
+ tag_build =
3
+ tag_date = 0
4
+
@@ -0,0 +1 @@
1
+ """siemens-docs-mcp — search, read and export Siemens Fluid Topics documentation."""
@@ -0,0 +1,254 @@
1
+ """Per-host catalog of publications (maps): listing, canonicalization, URL resolution, TOC cache."""
2
+ from __future__ import annotations
3
+
4
+ import difflib
5
+ import logging
6
+ import re
7
+ import threading
8
+ import time
9
+ from collections import Counter
10
+ from collections.abc import Callable, Sequence
11
+ from dataclasses import dataclass
12
+ from typing import Literal
13
+ from urllib.parse import unquote, urlsplit
14
+
15
+ from siemens_docs_mcp.client import FluidtopicsClient
16
+ from siemens_docs_mcp.toc import TocPage, iter_pages, parse_toc
17
+
18
+ log = logging.getLogger(__name__)
19
+
20
+ TIA_VERSION_KEY = "tia:SoftwareVersionFilter"
21
+ SW_VERSION_KEY = "SoftwareVersion"
22
+ _VERSION_KEYS = (TIA_VERSION_KEY, SW_VERSION_KEY)
23
+ _CONTENT_ID = re.compile(r"[\w~.-]+")
24
+
25
+
26
+ class CatalogError(ValueError):
27
+ """Raised for unknown hosts, unresolvable URLs, and unknown filter values."""
28
+
29
+
30
+ @dataclass(frozen=True)
31
+ class MapInfo:
32
+ id: str
33
+ title: str
34
+ locale: str
35
+ product: str
36
+ version: str
37
+ version_key: str
38
+ pretty_url: str
39
+ cluster_id: str
40
+ last_publication: str
41
+
42
+
43
+ @dataclass(frozen=True)
44
+ class Resolved:
45
+ host: str
46
+ map_id: str
47
+ content_id: str | None
48
+
49
+
50
+ def _meta(raw: dict) -> dict[str, list[str]]:
51
+ return {m["key"]: m.get("values") or [] for m in raw.get("metadata", [])}
52
+
53
+
54
+ def _first(meta: dict[str, list[str]], key: str) -> str:
55
+ values = meta.get(key) or []
56
+ return values[0] if values else ""
57
+
58
+
59
+ def _map_info(raw: dict) -> MapInfo:
60
+ meta = _meta(raw)
61
+ version_key = next((k for k in _VERSION_KEYS if _first(meta, k)), "")
62
+ return MapInfo(
63
+ id=raw["id"],
64
+ title=raw.get("title", ""),
65
+ locale=_first(meta, "ft:locale"),
66
+ product=_first(meta, "Product"),
67
+ version=_first(meta, version_key) if version_key else "",
68
+ version_key=version_key,
69
+ pretty_url=_first(meta, "ft:prettyUrl").removeprefix("/r/"),
70
+ cluster_id=_first(meta, "ft:clusterId"),
71
+ last_publication=_first(meta, "ft:lastPublication"),
72
+ )
73
+
74
+
75
+ class Catalog:
76
+ def __init__(
77
+ self,
78
+ hosts: Sequence[str],
79
+ client_factory: Callable[[str], FluidtopicsClient],
80
+ ttl_seconds: float = 86400,
81
+ clock: Callable[[], float] = time.monotonic,
82
+ ) -> None:
83
+ self.hosts: tuple[str, ...] = tuple(h.lower() for h in hosts)
84
+ self._factory = client_factory
85
+ self._ttl = ttl_seconds
86
+ self._clock = clock
87
+ # ponytail: global lock serializes cold catalog loads across hosts; per-host locks if that hurts
88
+ self._lock = threading.RLock() # re-entrant: maps()/toc() call client() while holding it
89
+ self._clients: dict[str, FluidtopicsClient] = {}
90
+ self._maps: dict[str, tuple[float, dict[str, MapInfo]]] = {}
91
+ # Raw distinct version values per host per key (a map can carry both keys).
92
+ self._versions: dict[str, dict[str, list[str]]] = {}
93
+ # Raw version values per host per map id per key, for filter().
94
+ self._map_versions: dict[str, dict[str, dict[str, str]]] = {}
95
+ self._tocs: dict[tuple[str, str], tuple[float, TocPage]] = {}
96
+ # Lower-cased pretty URLs carried by more than one map per host (ambiguous as reader URLs).
97
+ self._shared: dict[str, set[str]] = {}
98
+
99
+ def _check_host(self, host: str) -> str:
100
+ host = host.lower()
101
+ if host not in self.hosts:
102
+ raise CatalogError(f"Host {host!r} is not allowed. Allowed hosts: {list(self.hosts)}")
103
+ return host
104
+
105
+ def client(self, host: str) -> FluidtopicsClient:
106
+ host = self._check_host(host)
107
+ with self._lock:
108
+ if host not in self._clients:
109
+ self._clients[host] = self._factory(host)
110
+ return self._clients[host]
111
+
112
+ def _fresh(self, stamp: float) -> bool:
113
+ return self._clock() - stamp <= self._ttl
114
+
115
+ def maps(self, host: str) -> dict[str, MapInfo]:
116
+ host = self._check_host(host)
117
+ with self._lock:
118
+ cached = self._maps.get(host)
119
+ if cached and self._fresh(cached[0]):
120
+ return cached[1]
121
+ raws = self.client(host).list_maps()
122
+ infos = {m.id: m for m in map(_map_info, raws)}
123
+ versions: dict[str, list[str]] = {k: [] for k in _VERSION_KEYS}
124
+ per_map: dict[str, dict[str, str]] = {}
125
+ for raw in raws:
126
+ meta = _meta(raw)
127
+ per_map[raw["id"]] = {}
128
+ for key in _VERSION_KEYS:
129
+ v = _first(meta, key)
130
+ if v:
131
+ per_map[raw["id"]][key] = v
132
+ if v not in versions[key]:
133
+ versions[key].append(v)
134
+ self._versions[host] = versions
135
+ self._map_versions[host] = per_map
136
+ counts = Counter(m.pretty_url.lower() for m in infos.values() if m.pretty_url)
137
+ self._shared[host] = {p for p, n in counts.items() if n > 1}
138
+ self._maps[host] = (self._clock(), infos)
139
+ return infos
140
+
141
+ @staticmethod
142
+ def _unknown(host: str, label: str, value: str, known: list[str]) -> CatalogError:
143
+ close = difflib.get_close_matches(value, known, n=5, cutoff=0.5)
144
+ return CatalogError(f"Unknown {label} {value!r} on {host}. Close matches: {close or known[:10]}")
145
+
146
+ def canonical(self, host: str, field: Literal["product", "locale"], value: str) -> str:
147
+ values = sorted({getattr(m, field) for m in self.maps(host).values()} - {""})
148
+ for v in values:
149
+ if v.lower() == value.lower():
150
+ return v
151
+ raise self._unknown(host, field, value, values)
152
+
153
+ def canonical_version(self, host: str, value: str) -> tuple[str, str]:
154
+ self.maps(host)
155
+ versions = self._versions[host.lower()]
156
+ for key in _VERSION_KEYS:
157
+ for v in versions[key]:
158
+ if v.lower() == value.lower():
159
+ return key, v
160
+ known = sorted({v for vs in versions.values() for v in vs})
161
+ raise self._unknown(host, "version", value, known)
162
+
163
+ def filter(
164
+ self,
165
+ host: str,
166
+ product: str | None = None,
167
+ version: str | None = None,
168
+ locale: str | None = None,
169
+ title_contains: str | None = None,
170
+ ) -> list[MapInfo]:
171
+ result = list(self.maps(host).values())
172
+ if product is not None:
173
+ p = self.canonical(host, "product", product)
174
+ result = [m for m in result if m.product == p]
175
+ if locale is not None:
176
+ loc = self.canonical(host, "locale", locale)
177
+ result = [m for m in result if m.locale == loc]
178
+ if version is not None:
179
+ key, v = self.canonical_version(host, version)
180
+ per_map = self._map_versions[host.lower()]
181
+ result = [m for m in result if per_map[m.id].get(key) == v]
182
+ if title_contains:
183
+ needle = title_contains.lower()
184
+ result = [m for m in result if needle in m.title.lower()]
185
+ return sorted(result, key=lambda m: m.title)
186
+
187
+ def toc(self, host: str, map_id: str) -> TocPage:
188
+ key = (self._check_host(host), map_id)
189
+ with self._lock:
190
+ cached = self._tocs.get(key)
191
+ if cached and self._fresh(cached[0]):
192
+ return cached[1]
193
+ root = parse_toc(self.client(host).get_pages(map_id))
194
+ self._tocs[key] = (self._clock(), root)
195
+ return root
196
+
197
+ def resolve(self, url: str) -> Resolved:
198
+ failure = CatalogError(
199
+ f"Could not resolve {url}. Use search_docs or list_publications to find a valid reader URL."
200
+ )
201
+ parts = urlsplit(url.strip())
202
+ host = self._check_host(parts.hostname or "")
203
+ path = unquote(parts.path).rstrip("/")
204
+ if not path.startswith("/r/"):
205
+ raise failure
206
+ rel = path[3:]
207
+ segs = rel.split("/")
208
+ maps = self.maps(host)
209
+ if segs[0] in maps:
210
+ if len(segs) == 1:
211
+ return Resolved(host, segs[0], None)
212
+ if not _CONTENT_ID.fullmatch(segs[1]) or segs[1] in (".", ".."):
213
+ raise failure
214
+ return Resolved(host, segs[0], segs[1])
215
+ low = rel.lower()
216
+ # Several maps can share a pretty URL, and a longer prefix may lack the topic: try every
217
+ # matching map, longest prefix first (stable on ties), until one's TOC holds the topic.
218
+ matches = sorted(
219
+ (
220
+ m for m in maps.values()
221
+ if m.pretty_url
222
+ and (low == m.pretty_url.lower() or (low + "/").startswith(m.pretty_url.lower() + "/"))
223
+ ),
224
+ key=lambda m: len(m.pretty_url),
225
+ reverse=True,
226
+ )
227
+ target = "/r/" + low
228
+ for m in matches:
229
+ if low == m.pretty_url.lower():
230
+ return Resolved(host, m.id, None)
231
+ for page in iter_pages(self.toc(host, m.id)):
232
+ if page.pretty_url.lower() == target:
233
+ return Resolved(host, m.id, page.content_id)
234
+ raise failure
235
+
236
+ def _pretty_ok(self, host: str, map_id: str) -> str:
237
+ """The map's pretty URL if it identifies the map unambiguously, else ""."""
238
+ info = self.maps(host).get(map_id)
239
+ if not info or not info.pretty_url or info.pretty_url.lower() in self._shared[host.lower()]:
240
+ return ""
241
+ return info.pretty_url
242
+
243
+ def map_url(self, host: str, map_id: str) -> str:
244
+ """Reader URL of a publication: its pretty URL, or the id form when missing or shared."""
245
+ host = self._check_host(host)
246
+ return f"https://{host}/r/{self._pretty_ok(host, map_id) or map_id}"
247
+
248
+ def reader_url(self, host: str, map_id: str, page: TocPage) -> str:
249
+ host = self._check_host(host)
250
+ if not page.content_id: # synthetic root of a multi-section TOC = the publication itself
251
+ return self.map_url(host, map_id)
252
+ if page.pretty_url and self._pretty_ok(host, map_id):
253
+ return f"https://{host}{page.pretty_url}"
254
+ return f"https://{host}/r/{map_id}/{page.content_id}"