portolan-python 0.1.2__tar.gz → 0.1.4__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: portolan-python
3
- Version: 0.1.2
3
+ Version: 0.1.4
4
4
  Summary: A lightweight Python implementation of the Portolan specification.
5
5
  Author: Portolan contributors
6
6
  License: Apache-2.0
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "portolan-python"
3
- version = "0.1.2"
3
+ version = "0.1.4"
4
4
  description = "A lightweight Python implementation of the Portolan specification."
5
5
  readme = "README.md"
6
6
  requires-python = ">=3.10"
@@ -0,0 +1,297 @@
1
+ """Read and fetch Portolan registry exports."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import json
6
+ import os
7
+ import shutil
8
+ import tempfile
9
+ from collections.abc import Callable
10
+ from contextlib import contextmanager
11
+ from dataclasses import dataclass
12
+ from pathlib import Path
13
+ from typing import Any
14
+ from urllib.error import HTTPError
15
+ from urllib.parse import urljoin, urlparse
16
+ from urllib.request import HTTPRedirectHandler, Request, build_opener
17
+
18
+ JsonObject = dict[str, Any]
19
+
20
+ DEFAULT_REGISTRY_URL = (
21
+ "https://raw.githubusercontent.com/portolan-sdi/portolan-registry/"
22
+ "refs/heads/main/exports/catalogs.json"
23
+ )
24
+
25
+
26
+ @dataclass(frozen=True)
27
+ class RegistryCatalogEntry:
28
+ """One catalog entry from a Portolan registry export."""
29
+
30
+ id: str
31
+ url: str
32
+ title: str | None = None
33
+ status: str | None = None
34
+
35
+
36
+ def load_registry_entries(
37
+ registry_url: str = DEFAULT_REGISTRY_URL,
38
+ *,
39
+ fetch_json: Callable[[str], JsonObject] | None = None,
40
+ catalog_ids: set[str] | None = None,
41
+ include_stale: bool = False,
42
+ limit: int | None = None,
43
+ ) -> list[RegistryCatalogEntry]:
44
+ """Load catalog entries from a Portolan registry export."""
45
+ fetch = fetch_json or _fetch_json
46
+ registry = fetch(registry_url)
47
+ entries: list[RegistryCatalogEntry] = []
48
+ for link in registry.get("links", []):
49
+ if not isinstance(link, dict) or link.get("rel") != "child":
50
+ continue
51
+ href = link.get("href")
52
+ registry_id = link.get("portolan_registry:id")
53
+ if not isinstance(href, str) or not isinstance(registry_id, str):
54
+ continue
55
+ status = link.get("portolan_registry:status")
56
+ if status != "valid" and not include_stale:
57
+ continue
58
+ if catalog_ids is not None and registry_id not in catalog_ids:
59
+ continue
60
+ title = link.get("title")
61
+ entries.append(
62
+ RegistryCatalogEntry(
63
+ id=registry_id,
64
+ url=urljoin(registry_url, href),
65
+ title=title if isinstance(title, str) else None,
66
+ status=status if isinstance(status, str) else None,
67
+ )
68
+ )
69
+ if limit is not None and len(entries) >= limit:
70
+ break
71
+ return entries
72
+
73
+
74
+ def download_registry_catalog(
75
+ catalog_url: str,
76
+ output_dir: Path,
77
+ *,
78
+ expected_catalog_id: str | None = None,
79
+ fetch_json: Callable[[str], JsonObject] | None = None,
80
+ ) -> Path:
81
+ """Download a published catalog snapshot for local workflows."""
82
+ _validate_remote_url(catalog_url)
83
+ fetch = fetch_json or _fetch_json
84
+ catalog = fetch(catalog_url)
85
+ catalog_id = str(catalog.get("id") or _fallback_catalog_id(catalog_url))
86
+ _validate_catalog_id(catalog_id)
87
+ if expected_catalog_id is not None:
88
+ _validate_catalog_id(expected_catalog_id)
89
+ if catalog_id != expected_catalog_id:
90
+ raise ValueError(
91
+ f"Catalog id '{catalog_id}' does not match registry id '{expected_catalog_id}'"
92
+ )
93
+ catalog_root = output_dir / catalog_id
94
+ output_dir.mkdir(parents=True, exist_ok=True)
95
+ with _catalog_lock(output_dir, catalog_id):
96
+ _validate_catalog_root(output_dir, catalog_root)
97
+ staging_root = Path(
98
+ tempfile.mkdtemp(prefix=f".{catalog_id}.staging-", dir=output_dir.resolve())
99
+ )
100
+ try:
101
+ _write_catalog_tree(catalog_url, catalog, catalog_url, staging_root, fetch)
102
+ _publish_snapshot(staging_root, catalog_root, output_dir)
103
+ except BaseException:
104
+ shutil.rmtree(staging_root, ignore_errors=True)
105
+ raise
106
+ return catalog_root
107
+
108
+
109
+ def _fetch_json(url: str) -> JsonObject:
110
+ _validate_remote_url(url)
111
+ request = Request(url, headers={"User-Agent": "portolan-python"})
112
+ opener = build_opener(_SameOriginRedirectHandler(url))
113
+ with opener.open(request, timeout=30) as response:
114
+ data = json.loads(response.read().decode("utf-8"))
115
+ if not isinstance(data, dict):
116
+ raise TypeError(f"Expected JSON object from {url}")
117
+ return data
118
+
119
+
120
+ def _validate_remote_url(url: str) -> None:
121
+ parsed = urlparse(url)
122
+ if parsed.scheme not in {"http", "https"} or not parsed.netloc:
123
+ raise ValueError(f"Registry URL must use HTTP or HTTPS: {url}")
124
+
125
+
126
+ class _SameOriginRedirectHandler(HTTPRedirectHandler):
127
+ def __init__(self, original_url: str) -> None:
128
+ self._origin = _url_origin(original_url)
129
+ super().__init__()
130
+
131
+ def redirect_request(
132
+ self,
133
+ req: Request,
134
+ fp: Any,
135
+ code: int,
136
+ msg: str,
137
+ headers: Any,
138
+ newurl: str,
139
+ ) -> Request | None:
140
+ if _url_origin(newurl) != self._origin:
141
+ raise HTTPError(newurl, code, f"Redirect changed origin: {newurl}", headers, fp)
142
+ return super().redirect_request(req, fp, code, msg, headers, newurl)
143
+
144
+
145
+ def _url_origin(url: str) -> tuple[str, str]:
146
+ parsed = urlparse(url)
147
+ return parsed.scheme.lower(), parsed.netloc.lower()
148
+
149
+
150
+ def _validate_catalog_id(catalog_id: str) -> None:
151
+ if catalog_id in {"", ".", ".."} or Path(catalog_id).name != catalog_id or "\\" in catalog_id:
152
+ raise ValueError(f"Catalog id must be a safe directory name: {catalog_id}")
153
+
154
+
155
+ @contextmanager
156
+ def _catalog_lock(output_dir: Path, catalog_id: str) -> Any:
157
+ lock_path = output_dir.resolve() / f".{catalog_id}.lock"
158
+ try:
159
+ descriptor = os.open(lock_path, os.O_CREAT | os.O_EXCL | os.O_WRONLY, 0o600)
160
+ except FileExistsError as err:
161
+ raise RuntimeError(f"Catalog download is already in progress: {catalog_id}") from err
162
+ os.close(descriptor)
163
+ try:
164
+ yield
165
+ finally:
166
+ lock_path.unlink(missing_ok=True)
167
+
168
+
169
+ def _validate_catalog_root(output_dir: Path, catalog_root: Path) -> None:
170
+ if catalog_root.is_symlink():
171
+ raise ValueError(f"Catalog directory must not be a symlink: {catalog_root}")
172
+ if not catalog_root.resolve().is_relative_to(output_dir.resolve()):
173
+ raise ValueError(f"Catalog directory escapes output directory: {catalog_root}")
174
+
175
+
176
+ def _publish_snapshot(staging_root: Path, catalog_root: Path, output_dir: Path) -> None:
177
+ _validate_catalog_root(output_dir, catalog_root)
178
+ backup_root = Path(
179
+ tempfile.mkdtemp(prefix=f".{catalog_root.name}.backup-", dir=output_dir.resolve())
180
+ )
181
+ backup_root.rmdir()
182
+ had_previous = catalog_root.exists()
183
+ if had_previous:
184
+ catalog_root.rename(backup_root)
185
+ try:
186
+ staging_root.rename(catalog_root)
187
+ except BaseException:
188
+ if had_previous:
189
+ backup_root.rename(catalog_root)
190
+ raise
191
+ if had_previous:
192
+ shutil.rmtree(backup_root)
193
+
194
+
195
+ def _write_catalog_tree(
196
+ document_url: str,
197
+ document: JsonObject,
198
+ root_url: str,
199
+ output_root: Path,
200
+ fetch_json: Callable[[str], JsonObject],
201
+ visited: set[str] | None = None,
202
+ targets: dict[Path, str] | None = None,
203
+ ) -> None:
204
+ if visited is None:
205
+ visited = set()
206
+ if targets is None:
207
+ targets = {}
208
+ visited.add(document_url)
209
+
210
+ target = _target_document_path(root_url, document_url, output_root)
211
+ owner = targets.get(target)
212
+ if owner is not None and owner != document_url:
213
+ raise ValueError(f"Registry documents map to the same local path: {owner}, {document_url}")
214
+ targets[target] = document_url
215
+ target.parent.mkdir(parents=True, exist_ok=True)
216
+ if document.get("type") == "Collection":
217
+ document = _with_absolute_asset_hrefs(document_url, document)
218
+ target.write_text(json.dumps(document, indent=2) + "\n", encoding="utf-8")
219
+
220
+ for link in document.get("links", []):
221
+ if not isinstance(link, dict) or link.get("rel") != "child":
222
+ continue
223
+ href = link.get("href")
224
+ if not isinstance(href, str):
225
+ continue
226
+ child_url = urljoin(document_url, href)
227
+ if child_url in visited:
228
+ continue
229
+ child_target = _target_document_path(root_url, child_url, output_root)
230
+ owner = targets.get(child_target)
231
+ if owner is not None and owner != child_url:
232
+ raise ValueError(f"Registry documents map to the same local path: {owner}, {child_url}")
233
+ child = fetch_json(child_url)
234
+ if child.get("type") in {"Catalog", "Collection"}:
235
+ _write_catalog_tree(
236
+ child_url,
237
+ child,
238
+ root_url,
239
+ output_root,
240
+ fetch_json,
241
+ visited,
242
+ targets,
243
+ )
244
+
245
+
246
+ def _with_absolute_asset_hrefs(document_url: str, collection: JsonObject) -> JsonObject:
247
+ assets = collection.get("assets")
248
+ if not isinstance(assets, dict):
249
+ return collection
250
+ updated = dict(collection)
251
+ updated_assets: dict[str, Any] = {}
252
+ for key, asset in assets.items():
253
+ if not isinstance(asset, dict):
254
+ updated_assets[key] = asset
255
+ continue
256
+ href = asset.get("href")
257
+ if isinstance(href, str):
258
+ rewritten = dict(asset)
259
+ rewritten["href"] = urljoin(document_url, href)
260
+ updated_assets[key] = rewritten
261
+ else:
262
+ updated_assets[key] = asset
263
+ updated["assets"] = updated_assets
264
+ return updated
265
+
266
+
267
+ def _relative_document_path(root_url: str, document_url: str) -> Path:
268
+ root = urlparse(root_url)
269
+ document = urlparse(document_url)
270
+ if (document.scheme.lower(), document.netloc.lower()) != (
271
+ root.scheme.lower(),
272
+ root.netloc.lower(),
273
+ ):
274
+ raise ValueError(f"Child document has a different origin: {document_url}")
275
+ root_path = Path(root.path).parent
276
+ document_path = Path(document.path)
277
+ try:
278
+ relative_path = document_path.relative_to(root_path)
279
+ except ValueError as err:
280
+ raise ValueError(f"Child document is outside catalog root: {document_url}") from err
281
+ if ".." in relative_path.parts:
282
+ raise ValueError(f"Child document is outside catalog root: {document_url}")
283
+ return relative_path
284
+
285
+
286
+ def _target_document_path(root_url: str, document_url: str, output_root: Path) -> Path:
287
+ relative_path = _relative_document_path(root_url, document_url)
288
+ resolved_root = output_root.resolve()
289
+ target = (output_root / relative_path).resolve()
290
+ if not target.is_relative_to(resolved_root):
291
+ raise ValueError(f"Child document escapes catalog root: {document_url}")
292
+ return target
293
+
294
+
295
+ def _fallback_catalog_id(catalog_url: str) -> str:
296
+ parent = Path(urlparse(catalog_url).path).parent.name
297
+ return parent or "catalog"
@@ -5,10 +5,13 @@ from __future__ import annotations
5
5
  import json
6
6
  from pathlib import Path
7
7
  from typing import Any
8
+ from urllib.error import HTTPError
9
+ from urllib.request import Request
8
10
 
9
11
  import pytest
10
12
 
11
13
  from portolan import download_registry_catalog, load_registry_entries
14
+ from portolan.registry import _SameOriginRedirectHandler
12
15
 
13
16
  pytestmark = pytest.mark.unit
14
17
 
@@ -57,6 +60,26 @@ def test_load_registry_entries_filters_to_valid_children() -> None:
57
60
  ]
58
61
 
59
62
 
63
+ def test_load_registry_entries_resolves_relative_child_urls() -> None:
64
+ registry = {
65
+ "links": [
66
+ {
67
+ "rel": "child",
68
+ "href": "../demo/catalog.json",
69
+ "portolan_registry:id": "demo",
70
+ "portolan_registry:status": "valid",
71
+ }
72
+ ]
73
+ }
74
+
75
+ entries = load_registry_entries(
76
+ "https://registry.test/exports/catalogs.json",
77
+ fetch_json=lambda url: registry,
78
+ )
79
+
80
+ assert entries[0].url == "https://registry.test/demo/catalog.json"
81
+
82
+
60
83
  def test_load_registry_entries_supports_catalog_ids_include_stale_and_limit() -> None:
61
84
  registry = {
62
85
  "links": [
@@ -196,6 +219,162 @@ def test_download_registry_catalog_skips_child_cycles(tmp_path: Path) -> None:
196
219
  assert fetched_urls == [root_url, child_url]
197
220
 
198
221
 
222
+ @pytest.mark.parametrize(
223
+ ("child_url", "message"),
224
+ [
225
+ ("https://other.test/demo/collection.json", "different origin"),
226
+ ("https://example.test/demo/../../outside.json", "outside catalog root"),
227
+ ],
228
+ )
229
+ def test_download_registry_catalog_rejects_unsafe_child_urls_before_fetch(
230
+ child_url: str,
231
+ message: str,
232
+ tmp_path: Path,
233
+ ) -> None:
234
+ root_url = "https://example.test/demo/catalog.json"
235
+ catalog = {
236
+ "type": "Catalog",
237
+ "id": "demo",
238
+ "links": [{"rel": "child", "href": child_url}],
239
+ }
240
+ fetched_urls: list[str] = []
241
+
242
+ def fetch_json(url: str) -> dict[str, Any]:
243
+ fetched_urls.append(url)
244
+ if url != root_url:
245
+ raise AssertionError(f"Fetched unsafe child URL: {url}")
246
+ return catalog
247
+
248
+ with pytest.raises(ValueError, match=message):
249
+ download_registry_catalog(root_url, tmp_path, fetch_json=fetch_json)
250
+
251
+ assert fetched_urls == [root_url]
252
+
253
+
254
+ def test_download_registry_catalog_rejects_symlink_escape(tmp_path: Path) -> None:
255
+ root_url = "https://example.test/demo/catalog.json"
256
+ child_url = "https://example.test/demo/linked/collection.json"
257
+ output_dir = tmp_path / "output"
258
+ outside_dir = tmp_path / "outside"
259
+ catalog_root = output_dir / "demo"
260
+ outside_dir.mkdir()
261
+ catalog_root.mkdir(parents=True)
262
+ (catalog_root / "linked").symlink_to(outside_dir, target_is_directory=True)
263
+ responses = {
264
+ root_url: {
265
+ "type": "Catalog",
266
+ "id": "demo",
267
+ "links": [{"rel": "child", "href": child_url}],
268
+ },
269
+ child_url: _collection("linked", {}),
270
+ }
271
+
272
+ download_registry_catalog(
273
+ root_url,
274
+ output_dir,
275
+ fetch_json=lambda url: responses[url],
276
+ )
277
+
278
+ assert not (outside_dir / "collection.json").exists()
279
+ assert (catalog_root / "linked" / "collection.json").exists()
280
+
281
+
282
+ def test_download_registry_catalog_rejects_catalog_root_symlink(tmp_path: Path) -> None:
283
+ output_dir = tmp_path / "output"
284
+ outside_dir = tmp_path / "outside"
285
+ output_dir.mkdir()
286
+ outside_dir.mkdir()
287
+ (output_dir / "demo").symlink_to(outside_dir, target_is_directory=True)
288
+
289
+ with pytest.raises(ValueError, match="Catalog directory must not be a symlink"):
290
+ download_registry_catalog(
291
+ "https://example.test/demo/catalog.json",
292
+ output_dir,
293
+ fetch_json=lambda url: {"type": "Catalog", "id": "demo", "links": []},
294
+ )
295
+
296
+ assert list(outside_dir.iterdir()) == []
297
+
298
+
299
+ def test_download_registry_catalog_rejects_registry_id_mismatch(tmp_path: Path) -> None:
300
+ with pytest.raises(ValueError, match="does not match registry id"):
301
+ download_registry_catalog(
302
+ "https://example.test/demo/catalog.json",
303
+ tmp_path,
304
+ expected_catalog_id="selected-catalog",
305
+ fetch_json=lambda url: {"type": "Catalog", "id": "other-catalog", "links": []},
306
+ )
307
+
308
+ assert not (tmp_path / "selected-catalog").exists()
309
+
310
+
311
+ def test_download_registry_catalog_preserves_snapshot_when_child_fetch_fails(
312
+ tmp_path: Path,
313
+ ) -> None:
314
+ catalog_root = tmp_path / "demo"
315
+ catalog_root.mkdir()
316
+ previous = catalog_root / "catalog.json"
317
+ previous.write_text('{"id": "previous"}\n', encoding="utf-8")
318
+ root_url = "https://example.test/demo/catalog.json"
319
+
320
+ def fetch_json(url: str) -> dict[str, Any]:
321
+ if url == root_url:
322
+ return {
323
+ "type": "Catalog",
324
+ "id": "demo",
325
+ "links": [{"rel": "child", "href": "./missing.json"}],
326
+ }
327
+ raise HTTPError(url, 503, "Unavailable", hdrs=None, fp=None)
328
+
329
+ with pytest.raises(HTTPError):
330
+ download_registry_catalog(root_url, tmp_path, fetch_json=fetch_json)
331
+
332
+ assert previous.read_text(encoding="utf-8") == '{"id": "previous"}\n'
333
+ assert not list(tmp_path.glob(".demo.staging-*"))
334
+
335
+
336
+ def test_same_origin_redirect_handler_rejects_cross_origin_redirect() -> None:
337
+ handler = _SameOriginRedirectHandler("https://example.test/demo/catalog.json")
338
+
339
+ with pytest.raises(HTTPError, match="Redirect changed origin"):
340
+ handler.redirect_request(
341
+ Request("https://example.test/demo/child.json"),
342
+ fp=None,
343
+ code=302,
344
+ msg="Found",
345
+ headers={},
346
+ newurl="http://127.0.0.1/private",
347
+ )
348
+
349
+
350
+ def test_download_registry_catalog_rejects_local_path_collisions(tmp_path: Path) -> None:
351
+ root_url = "https://example.test/demo/catalog.json"
352
+ first_url = "https://example.test/demo/collection.json?version=1"
353
+ second_url = "https://example.test/demo/collection.json?version=2"
354
+ responses = {
355
+ root_url: {
356
+ "type": "Catalog",
357
+ "id": "demo",
358
+ "links": [
359
+ {"rel": "child", "href": first_url},
360
+ {"rel": "child", "href": second_url},
361
+ ],
362
+ },
363
+ first_url: _collection("first", {}),
364
+ second_url: _collection("second", {}),
365
+ }
366
+ fetched_urls: list[str] = []
367
+
368
+ def fetch_json(url: str) -> dict[str, Any]:
369
+ fetched_urls.append(url)
370
+ return responses[url]
371
+
372
+ with pytest.raises(ValueError, match="same local path"):
373
+ download_registry_catalog(root_url, tmp_path, fetch_json=fetch_json)
374
+
375
+ assert fetched_urls == [root_url, first_url]
376
+
377
+
199
378
  def test_download_registry_catalog_recurses_nested_catalogs_and_uses_fallback_id(
200
379
  tmp_path: Path,
201
380
  ) -> None:
@@ -468,7 +468,7 @@ wheels = [
468
468
 
469
469
  [[package]]
470
470
  name = "portolan-python"
471
- version = "0.1.2"
471
+ version = "0.1.4"
472
472
  source = { editable = "." }
473
473
 
474
474
  [package.optional-dependencies]
@@ -1,169 +0,0 @@
1
- """Read and fetch Portolan registry exports."""
2
-
3
- from __future__ import annotations
4
-
5
- import json
6
- from collections.abc import Callable
7
- from dataclasses import dataclass
8
- from pathlib import Path
9
- from typing import Any
10
- from urllib.parse import urljoin, urlparse
11
- from urllib.request import Request, urlopen
12
-
13
- JsonObject = dict[str, Any]
14
-
15
- DEFAULT_REGISTRY_URL = (
16
- "https://raw.githubusercontent.com/portolan-sdi/portolan-registry/"
17
- "refs/heads/main/exports/catalogs.json"
18
- )
19
-
20
-
21
- @dataclass(frozen=True)
22
- class RegistryCatalogEntry:
23
- """One catalog entry from a Portolan registry export."""
24
-
25
- id: str
26
- url: str
27
- title: str | None = None
28
- status: str | None = None
29
-
30
-
31
- def load_registry_entries(
32
- registry_url: str = DEFAULT_REGISTRY_URL,
33
- *,
34
- fetch_json: Callable[[str], JsonObject] | None = None,
35
- catalog_ids: set[str] | None = None,
36
- include_stale: bool = False,
37
- limit: int | None = None,
38
- ) -> list[RegistryCatalogEntry]:
39
- """Load catalog entries from a Portolan registry export."""
40
- fetch = fetch_json or _fetch_json
41
- registry = fetch(registry_url)
42
- entries: list[RegistryCatalogEntry] = []
43
- for link in registry.get("links", []):
44
- if not isinstance(link, dict) or link.get("rel") != "child":
45
- continue
46
- href = link.get("href")
47
- registry_id = link.get("portolan_registry:id")
48
- if not isinstance(href, str) or not isinstance(registry_id, str):
49
- continue
50
- status = link.get("portolan_registry:status")
51
- if status != "valid" and not include_stale:
52
- continue
53
- if catalog_ids is not None and registry_id not in catalog_ids:
54
- continue
55
- title = link.get("title")
56
- entries.append(
57
- RegistryCatalogEntry(
58
- id=registry_id,
59
- url=href,
60
- title=title if isinstance(title, str) else None,
61
- status=status if isinstance(status, str) else None,
62
- )
63
- )
64
- if limit is not None and len(entries) >= limit:
65
- break
66
- return entries
67
-
68
-
69
- def download_registry_catalog(
70
- catalog_url: str,
71
- output_dir: Path,
72
- *,
73
- fetch_json: Callable[[str], JsonObject] | None = None,
74
- ) -> Path:
75
- """Download a published catalog snapshot for local workflows."""
76
- _validate_remote_url(catalog_url)
77
- fetch = fetch_json or _fetch_json
78
- catalog = fetch(catalog_url)
79
- catalog_id = str(catalog.get("id") or _fallback_catalog_id(catalog_url))
80
- if catalog_id in {"", ".", ".."} or Path(catalog_id).name != catalog_id or "\\" in catalog_id:
81
- raise ValueError(f"Catalog id must be a safe directory name: {catalog_id}")
82
- catalog_root = output_dir / catalog_id
83
- _write_catalog_tree(catalog_url, catalog, catalog_url, catalog_root, fetch)
84
- return catalog_root
85
-
86
-
87
- def _fetch_json(url: str) -> JsonObject:
88
- _validate_remote_url(url)
89
- request = Request(url, headers={"User-Agent": "portolan-python"})
90
- with urlopen(request, timeout=30) as response:
91
- data = json.loads(response.read().decode("utf-8"))
92
- if not isinstance(data, dict):
93
- raise TypeError(f"Expected JSON object from {url}")
94
- return data
95
-
96
-
97
- def _validate_remote_url(url: str) -> None:
98
- parsed = urlparse(url)
99
- if parsed.scheme not in {"http", "https"} or not parsed.netloc:
100
- raise ValueError(f"Registry URL must use HTTP or HTTPS: {url}")
101
-
102
-
103
- def _write_catalog_tree(
104
- document_url: str,
105
- document: JsonObject,
106
- root_url: str,
107
- output_root: Path,
108
- fetch_json: Callable[[str], JsonObject],
109
- visited: set[str] | None = None,
110
- ) -> None:
111
- if visited is None:
112
- visited = set()
113
- visited.add(document_url)
114
-
115
- relative_path = _relative_document_path(root_url, document_url)
116
- target = output_root / relative_path
117
- target.parent.mkdir(parents=True, exist_ok=True)
118
- if document.get("type") == "Collection":
119
- document = _with_absolute_asset_hrefs(document_url, document)
120
- target.write_text(json.dumps(document, indent=2) + "\n", encoding="utf-8")
121
-
122
- for link in document.get("links", []):
123
- if not isinstance(link, dict) or link.get("rel") != "child":
124
- continue
125
- href = link.get("href")
126
- if not isinstance(href, str):
127
- continue
128
- child_url = urljoin(document_url, href)
129
- if child_url in visited:
130
- continue
131
- child = fetch_json(child_url)
132
- if child.get("type") in {"Catalog", "Collection"}:
133
- _write_catalog_tree(child_url, child, root_url, output_root, fetch_json, visited)
134
-
135
-
136
- def _with_absolute_asset_hrefs(document_url: str, collection: JsonObject) -> JsonObject:
137
- assets = collection.get("assets")
138
- if not isinstance(assets, dict):
139
- return collection
140
- updated = dict(collection)
141
- updated_assets: dict[str, Any] = {}
142
- for key, asset in assets.items():
143
- if not isinstance(asset, dict):
144
- updated_assets[key] = asset
145
- continue
146
- href = asset.get("href")
147
- if isinstance(href, str):
148
- rewritten = dict(asset)
149
- rewritten["href"] = urljoin(document_url, href)
150
- updated_assets[key] = rewritten
151
- else:
152
- updated_assets[key] = asset
153
- updated["assets"] = updated_assets
154
- return updated
155
-
156
-
157
- def _relative_document_path(root_url: str, document_url: str) -> Path:
158
- root_path = Path(urlparse(root_url).path).parent
159
- document_path = Path(urlparse(document_url).path)
160
- try:
161
- relative = document_path.relative_to(root_path)
162
- except ValueError:
163
- return Path(document_path.name or "catalog.json")
164
- return relative
165
-
166
-
167
- def _fallback_catalog_id(catalog_url: str) -> str:
168
- parent = Path(urlparse(catalog_url).path).parent.name
169
- return parent or "catalog"