portolan-python 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
portolan/__init__.py ADDED
@@ -0,0 +1,29 @@
1
+ """Public API for Portolan catalog access."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from portolan.catalog import Asset, AssetFormat, Catalog, Collection, Item, Link, is_stac_metadata
6
+ from portolan.registry import (
7
+ DEFAULT_REGISTRY_URL,
8
+ RegistryCatalogEntry,
9
+ download_registry_catalog,
10
+ load_registry_entries,
11
+ )
12
+ from portolan.validation import ValidationError, ValidationResult, Validator
13
+
14
+ __all__ = [
15
+ "Asset",
16
+ "AssetFormat",
17
+ "Catalog",
18
+ "Collection",
19
+ "DEFAULT_REGISTRY_URL",
20
+ "Item",
21
+ "Link",
22
+ "RegistryCatalogEntry",
23
+ "ValidationError",
24
+ "ValidationResult",
25
+ "Validator",
26
+ "download_registry_catalog",
27
+ "is_stac_metadata",
28
+ "load_registry_entries",
29
+ ]
portolan/catalog.py ADDED
@@ -0,0 +1,307 @@
1
+ """Domain model and catalog traversal for Portolan/STAC catalogs."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import json
6
+ from collections.abc import Iterator
7
+ from dataclasses import dataclass
8
+ from enum import Enum
9
+ from pathlib import Path
10
+ from typing import Any
11
+ from urllib.parse import unquote, urljoin, urlparse
12
+ from urllib.request import Request, urlopen
13
+
14
+ JsonObject = dict[str, Any]
15
+ STAC_DOCUMENT_TYPES = frozenset({"Catalog", "Collection", "Feature"})
16
+
17
+
18
+ class AssetFormat(Enum):
19
+ """Known cloud-native asset formats represented by Portolan catalogs."""
20
+
21
+ GEOPARQUET = "GeoParquet"
22
+ COG = "COG"
23
+ PMTILES = "PMTiles"
24
+ UNKNOWN = "Unknown"
25
+
26
+
27
+ @dataclass(frozen=True)
28
+ class Link:
29
+ """A STAC link with its HREF resolved against the containing document."""
30
+
31
+ rel: str
32
+ href: str
33
+ media_type: str | None
34
+ title: str | None
35
+ raw: JsonObject
36
+
37
+
38
+ @dataclass(frozen=True)
39
+ class Asset:
40
+ """A STAC asset with Portolan-friendly accessors."""
41
+
42
+ key: str
43
+ href: str
44
+ media_type: str | None
45
+ roles: tuple[str, ...]
46
+ title: str | None
47
+ description: str | None
48
+ raw: JsonObject
49
+
50
+ @property
51
+ def format(self) -> AssetFormat:
52
+ """Return the known Portolan asset format, without reading the asset bytes."""
53
+ media_type = self.media_type or ""
54
+ href_path = urlparse(self.href).path.lower()
55
+ if media_type == "application/vnd.apache.parquet" or href_path.endswith(".parquet"):
56
+ return AssetFormat.GEOPARQUET
57
+ if (
58
+ media_type == "image/tiff; application=geotiff; profile=cloud-optimized"
59
+ or href_path.endswith((".tif", ".tiff"))
60
+ ):
61
+ return AssetFormat.COG
62
+ if media_type == "application/vnd.pmtiles" or href_path.endswith(".pmtiles"):
63
+ return AssetFormat.PMTILES
64
+ return AssetFormat.UNKNOWN
65
+
66
+
67
+ class Catalog:
68
+ """A loaded Portolan catalog root."""
69
+
70
+ def __init__(self, data: JsonObject, href: str) -> None:
71
+ self._data = data
72
+ self._href = href
73
+
74
+ @classmethod
75
+ def open(cls, source: str | Path) -> Catalog:
76
+ """Open a local or remote Portolan catalog document."""
77
+ href = _source_to_href(source)
78
+ data = _read_json_href(href)
79
+ return cls(data, href)
80
+
81
+ @property
82
+ def id(self) -> str:
83
+ return str(self._data.get("id", ""))
84
+
85
+ @property
86
+ def href(self) -> str:
87
+ return self._href
88
+
89
+ @property
90
+ def data(self) -> JsonObject:
91
+ return dict(self._data)
92
+
93
+ def links(self) -> Iterator[Link]:
94
+ yield from _links(self._data, self._href)
95
+
96
+ def collections(self) -> Iterator[Collection]:
97
+ """Yield child collections, including collections below child catalogs."""
98
+ yield from _collections_from_catalog(self._data, self._href, set())
99
+
100
+ def item_links(self) -> Iterator[Link]:
101
+ """Yield item links owned by this catalog subtree."""
102
+ yield from _item_links_from_document(self._data, self._href, set())
103
+
104
+
105
+ class Collection:
106
+ """A loaded STAC Collection within a Portolan catalog."""
107
+
108
+ def __init__(self, data: JsonObject, href: str) -> None:
109
+ self._data = data
110
+ self._href = href
111
+
112
+ @classmethod
113
+ def open(cls, source: str | Path) -> Collection:
114
+ """Open a local or remote STAC Collection document."""
115
+ href = _source_to_href(source, default_document="collection.json")
116
+ data = _read_json_href(href)
117
+ return cls(data, href)
118
+
119
+ @property
120
+ def id(self) -> str:
121
+ return str(self._data.get("id", ""))
122
+
123
+ @property
124
+ def href(self) -> str:
125
+ return self._href
126
+
127
+ @property
128
+ def data(self) -> JsonObject:
129
+ return dict(self._data)
130
+
131
+ def links(self) -> Iterator[Link]:
132
+ yield from _links(self._data, self._href)
133
+
134
+ def assets(self) -> Iterator[Asset]:
135
+ yield from _assets(self._data, self._href)
136
+
137
+ def items(self) -> Iterator[Item]:
138
+ """Yield linked items from this collection."""
139
+ for link in self.item_links():
140
+ data = _read_json_href(link.href)
141
+ if data.get("type") == "Feature":
142
+ yield Item(data, link.href)
143
+
144
+ def item_links(self) -> Iterator[Link]:
145
+ """Yield item links owned by this collection.
146
+
147
+ A collection can group items behind child catalogs. Those items still
148
+ belong to the collection, so traversal follows child catalogs and stops
149
+ at child collections.
150
+ """
151
+ yield from _item_links_from_document(self._data, self._href, set())
152
+
153
+
154
+ class Item:
155
+ """A loaded STAC Item."""
156
+
157
+ def __init__(self, data: JsonObject, href: str) -> None:
158
+ self._data = data
159
+ self._href = href
160
+
161
+ @classmethod
162
+ def open(cls, source: str | Path) -> Item:
163
+ """Open a local or remote STAC Item document."""
164
+ href = _source_to_href(source, default_document="item.json")
165
+ data = _read_json_href(href)
166
+ return cls(data, href)
167
+
168
+ @property
169
+ def id(self) -> str:
170
+ return str(self._data.get("id", ""))
171
+
172
+ @property
173
+ def href(self) -> str:
174
+ return self._href
175
+
176
+ @property
177
+ def data(self) -> JsonObject:
178
+ return dict(self._data)
179
+
180
+ def links(self) -> Iterator[Link]:
181
+ yield from _links(self._data, self._href)
182
+
183
+ def assets(self) -> Iterator[Asset]:
184
+ yield from _assets(self._data, self._href)
185
+
186
+
187
+ def is_stac_metadata(source: str | Path) -> bool:
188
+ """Return whether ``source`` is a STAC Catalog, Collection, or Item document."""
189
+ if isinstance(source, Path) and source.suffix.lower() != ".json":
190
+ return False
191
+ try:
192
+ data = _read_json_href(_source_to_href(source, default_document="catalog.json"))
193
+ except (OSError, TypeError, ValueError, json.JSONDecodeError):
194
+ return False
195
+ return data.get("type") in STAC_DOCUMENT_TYPES
196
+
197
+
198
+ def _collections_from_catalog(
199
+ data: JsonObject, href: str, visited: set[str]
200
+ ) -> Iterator[Collection]:
201
+ if href in visited:
202
+ return
203
+ visited.add(href)
204
+ for link in _links(data, href):
205
+ if link.rel != "child":
206
+ continue
207
+ child = _read_json_href(link.href)
208
+ child_type = child.get("type")
209
+ if child_type == "Collection":
210
+ yield Collection(child, link.href)
211
+ elif child_type == "Catalog":
212
+ yield from _collections_from_catalog(child, link.href, visited)
213
+
214
+
215
+ def _item_links_from_document(data: JsonObject, href: str, visited: set[str]) -> Iterator[Link]:
216
+ if href in visited:
217
+ return
218
+ visited.add(href)
219
+ for link in _links(data, href):
220
+ if link.rel == "item":
221
+ yield link
222
+ elif link.rel == "child":
223
+ child = _read_json_href(link.href)
224
+ if child.get("type") == "Catalog":
225
+ yield from _item_links_from_document(child, link.href, visited)
226
+
227
+
228
+ def _links(data: JsonObject, document_href: str) -> Iterator[Link]:
229
+ links = data.get("links")
230
+ if not isinstance(links, list):
231
+ return
232
+ for raw_link in links:
233
+ if not isinstance(raw_link, dict):
234
+ continue
235
+ rel = raw_link.get("rel")
236
+ href = raw_link.get("href")
237
+ if not isinstance(rel, str) or not isinstance(href, str):
238
+ continue
239
+ media_type = raw_link.get("type")
240
+ title = raw_link.get("title")
241
+ yield Link(
242
+ rel=rel,
243
+ href=_resolve_href(document_href, href),
244
+ media_type=media_type if isinstance(media_type, str) else None,
245
+ title=title if isinstance(title, str) else None,
246
+ raw=dict(raw_link),
247
+ )
248
+
249
+
250
+ def _assets(data: JsonObject, document_href: str) -> Iterator[Asset]:
251
+ assets = data.get("assets")
252
+ if not isinstance(assets, dict):
253
+ return
254
+ for key, raw_asset in assets.items():
255
+ if not isinstance(key, str) or not isinstance(raw_asset, dict):
256
+ continue
257
+ href = raw_asset.get("href")
258
+ if not isinstance(href, str):
259
+ continue
260
+ media_type = raw_asset.get("type") or raw_asset.get("media_type")
261
+ roles = raw_asset.get("roles")
262
+ title = raw_asset.get("title")
263
+ description = raw_asset.get("description")
264
+ yield Asset(
265
+ key=key,
266
+ href=_resolve_href(document_href, href),
267
+ media_type=media_type if isinstance(media_type, str) else None,
268
+ roles=tuple(role for role in roles if isinstance(role, str))
269
+ if isinstance(roles, list)
270
+ else (),
271
+ title=title if isinstance(title, str) else None,
272
+ description=description if isinstance(description, str) else None,
273
+ raw=dict(raw_asset),
274
+ )
275
+
276
+
277
+ def _source_to_href(source: str | Path, *, default_document: str = "catalog.json") -> str:
278
+ if isinstance(source, Path):
279
+ path = source
280
+ else:
281
+ parsed = urlparse(source)
282
+ if parsed.scheme in {"http", "https", "file"}:
283
+ return source
284
+ path = Path(source)
285
+ if path.is_dir():
286
+ path = path / default_document
287
+ return path.resolve().as_uri()
288
+
289
+
290
+ def _read_json_href(href: str) -> JsonObject:
291
+ parsed = urlparse(href)
292
+ if parsed.scheme in {"http", "https"}:
293
+ request = Request(href, headers={"User-Agent": "portolan-python"})
294
+ with urlopen(request, timeout=30) as response:
295
+ data = json.loads(response.read().decode("utf-8"))
296
+ elif parsed.scheme == "file":
297
+ path = Path(unquote(parsed.path))
298
+ data = json.loads(path.read_text(encoding="utf-8"))
299
+ else:
300
+ data = json.loads(Path(href).read_text(encoding="utf-8"))
301
+ if not isinstance(data, dict):
302
+ raise TypeError(f"Expected JSON object from {href}")
303
+ return data
304
+
305
+
306
+ def _resolve_href(document_href: str, href: str) -> str:
307
+ return urljoin(document_href, href)
portolan/py.typed ADDED
@@ -0,0 +1 @@
1
+
portolan/registry.py ADDED
@@ -0,0 +1,152 @@
1
+ """Read and fetch Portolan registry exports."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import json
6
+ from collections.abc import Callable
7
+ from dataclasses import dataclass
8
+ from pathlib import Path
9
+ from typing import Any
10
+ from urllib.parse import urljoin, urlparse
11
+ from urllib.request import Request, urlopen
12
+
13
+ JsonObject = dict[str, Any]
14
+
15
+ DEFAULT_REGISTRY_URL = (
16
+ "https://raw.githubusercontent.com/portolan-sdi/portolan-registry/"
17
+ "refs/heads/main/exports/catalogs.json"
18
+ )
19
+
20
+
21
+ @dataclass(frozen=True)
22
+ class RegistryCatalogEntry:
23
+ """One catalog entry from a Portolan registry export."""
24
+
25
+ id: str
26
+ url: str
27
+ title: str | None = None
28
+ status: str | None = None
29
+
30
+
31
+ def load_registry_entries(
32
+ registry_url: str = DEFAULT_REGISTRY_URL,
33
+ *,
34
+ fetch_json: Callable[[str], JsonObject] | None = None,
35
+ catalog_ids: set[str] | None = None,
36
+ include_stale: bool = False,
37
+ limit: int | None = None,
38
+ ) -> list[RegistryCatalogEntry]:
39
+ """Load catalog entries from a Portolan registry export."""
40
+ fetch = fetch_json or _fetch_json
41
+ registry = fetch(registry_url)
42
+ entries: list[RegistryCatalogEntry] = []
43
+ for link in registry.get("links", []):
44
+ if not isinstance(link, dict) or link.get("rel") != "child":
45
+ continue
46
+ href = link.get("href")
47
+ registry_id = link.get("portolan_registry:id")
48
+ if not isinstance(href, str) or not isinstance(registry_id, str):
49
+ continue
50
+ status = link.get("portolan_registry:status")
51
+ if status != "valid" and not include_stale:
52
+ continue
53
+ if catalog_ids is not None and registry_id not in catalog_ids:
54
+ continue
55
+ title = link.get("title")
56
+ entries.append(
57
+ RegistryCatalogEntry(
58
+ id=registry_id,
59
+ url=href,
60
+ title=title if isinstance(title, str) else None,
61
+ status=status if isinstance(status, str) else None,
62
+ )
63
+ )
64
+ if limit is not None and len(entries) >= limit:
65
+ break
66
+ return entries
67
+
68
+
69
+ def download_registry_catalog(
70
+ catalog_url: str,
71
+ output_dir: Path,
72
+ *,
73
+ fetch_json: Callable[[str], JsonObject] | None = None,
74
+ ) -> Path:
75
+ """Download a published catalog snapshot for local workflows."""
76
+ fetch = fetch_json or _fetch_json
77
+ catalog = fetch(catalog_url)
78
+ catalog_id = str(catalog.get("id") or _fallback_catalog_id(catalog_url))
79
+ catalog_root = output_dir / catalog_id
80
+ _write_catalog_tree(catalog_url, catalog, catalog_url, catalog_root, fetch)
81
+ return catalog_root
82
+
83
+
84
+ def _fetch_json(url: str) -> JsonObject:
85
+ request = Request(url, headers={"User-Agent": "portolan-python"})
86
+ with urlopen(request, timeout=30) as response:
87
+ data = json.loads(response.read().decode("utf-8"))
88
+ if not isinstance(data, dict):
89
+ raise TypeError(f"Expected JSON object from {url}")
90
+ return data
91
+
92
+
93
+ def _write_catalog_tree(
94
+ document_url: str,
95
+ document: JsonObject,
96
+ root_url: str,
97
+ output_root: Path,
98
+ fetch_json: Callable[[str], JsonObject],
99
+ ) -> None:
100
+ relative_path = _relative_document_path(root_url, document_url)
101
+ target = output_root / relative_path
102
+ target.parent.mkdir(parents=True, exist_ok=True)
103
+ if document.get("type") == "Collection":
104
+ document = _with_absolute_asset_hrefs(document_url, document)
105
+ target.write_text(json.dumps(document, indent=2) + "\n", encoding="utf-8")
106
+
107
+ for link in document.get("links", []):
108
+ if not isinstance(link, dict) or link.get("rel") != "child":
109
+ continue
110
+ href = link.get("href")
111
+ if not isinstance(href, str):
112
+ continue
113
+ child_url = urljoin(document_url, href)
114
+ child = fetch_json(child_url)
115
+ if child.get("type") in {"Catalog", "Collection"}:
116
+ _write_catalog_tree(child_url, child, root_url, output_root, fetch_json)
117
+
118
+
119
+ def _with_absolute_asset_hrefs(document_url: str, collection: JsonObject) -> JsonObject:
120
+ assets = collection.get("assets")
121
+ if not isinstance(assets, dict):
122
+ return collection
123
+ updated = dict(collection)
124
+ updated_assets: dict[str, Any] = {}
125
+ for key, asset in assets.items():
126
+ if not isinstance(asset, dict):
127
+ updated_assets[key] = asset
128
+ continue
129
+ href = asset.get("href")
130
+ if isinstance(href, str):
131
+ rewritten = dict(asset)
132
+ rewritten["href"] = urljoin(document_url, href)
133
+ updated_assets[key] = rewritten
134
+ else:
135
+ updated_assets[key] = asset
136
+ updated["assets"] = updated_assets
137
+ return updated
138
+
139
+
140
+ def _relative_document_path(root_url: str, document_url: str) -> Path:
141
+ root_path = Path(urlparse(root_url).path).parent
142
+ document_path = Path(urlparse(document_url).path)
143
+ try:
144
+ relative = document_path.relative_to(root_path)
145
+ except ValueError:
146
+ return Path(document_path.name or "catalog.json")
147
+ return relative
148
+
149
+
150
+ def _fallback_catalog_id(catalog_url: str) -> str:
151
+ parent = Path(urlparse(catalog_url).path).parent.name
152
+ return parent or "catalog"
portolan/validation.py ADDED
@@ -0,0 +1,165 @@
1
+ """Small programmatic validation API for Portolan/STAC structures."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from dataclasses import dataclass
6
+ from typing import Any
7
+
8
+ from portolan.catalog import Catalog
9
+
10
+ JsonObject = dict[str, Any]
11
+
12
+
13
+ @dataclass(frozen=True)
14
+ class ValidationError:
15
+ """One validation finding."""
16
+
17
+ code: str
18
+ path: str
19
+ message: str
20
+
21
+
22
+ @dataclass(frozen=True)
23
+ class ValidationResult:
24
+ """Validation outcome."""
25
+
26
+ errors: tuple[ValidationError, ...]
27
+
28
+ @property
29
+ def valid(self) -> bool:
30
+ return not self.errors
31
+
32
+
33
+ class Validator:
34
+ """Validate loaded Portolan catalog structures."""
35
+
36
+ @staticmethod
37
+ def validate(catalog: Catalog) -> ValidationResult:
38
+ """Validate a catalog and return structured findings."""
39
+ errors: list[ValidationError] = []
40
+ _validate_catalog_document(catalog.data, "catalog", errors)
41
+ for collection in catalog.collections():
42
+ collection_path = f"collection.{collection.id}"
43
+ _validate_collection_document(collection.data, collection_path, errors)
44
+ for item in collection.items():
45
+ _validate_item_document(item.data, f"item.{item.id}", errors)
46
+ return ValidationResult(errors=tuple(errors))
47
+
48
+
49
+ def _validate_catalog_document(data: JsonObject, path: str, errors: list[ValidationError]) -> None:
50
+ _require_string(data, "type", path, errors)
51
+ _require_string(data, "stac_version", path, errors)
52
+ _require_string(data, "id", path, errors)
53
+ _require_string(data, "description", path, errors)
54
+ _validate_links(data, path, errors)
55
+
56
+
57
+ def _validate_collection_document(
58
+ data: JsonObject, path: str, errors: list[ValidationError]
59
+ ) -> None:
60
+ _require_string(data, "type", path, errors)
61
+ _require_string(data, "stac_version", path, errors)
62
+ _require_string(data, "id", path, errors)
63
+ _require_string(data, "description", path, errors)
64
+ _require_string(data, "license", path, errors)
65
+ if not isinstance(data.get("extent"), dict):
66
+ errors.append(
67
+ ValidationError(
68
+ code="PTL-STAC-001",
69
+ path=f"{path}.extent",
70
+ message="extent is required",
71
+ )
72
+ )
73
+ _validate_links(data, path, errors)
74
+ _validate_assets(data, path, errors)
75
+
76
+
77
+ def _validate_item_document(data: JsonObject, path: str, errors: list[ValidationError]) -> None:
78
+ _require_string(data, "type", path, errors)
79
+ _require_string(data, "stac_version", path, errors)
80
+ _require_string(data, "id", path, errors)
81
+ _require_string(data, "collection", path, errors)
82
+ if not isinstance(data.get("properties"), dict):
83
+ errors.append(
84
+ ValidationError(
85
+ code="PTL-STAC-001",
86
+ path=f"{path}.properties",
87
+ message="properties is required",
88
+ )
89
+ )
90
+ _validate_links(data, path, errors)
91
+ _validate_assets(data, path, errors)
92
+
93
+
94
+ def _validate_assets(data: JsonObject, path: str, errors: list[ValidationError]) -> None:
95
+ assets = data.get("assets")
96
+ if assets is None:
97
+ return
98
+ if not isinstance(assets, dict):
99
+ errors.append(
100
+ ValidationError(
101
+ code="PTL-STAC-003",
102
+ path=f"{path}.assets",
103
+ message="assets must be an object",
104
+ )
105
+ )
106
+ return
107
+ for key, asset in assets.items():
108
+ asset_path = f"{path}.assets.{key}"
109
+ if not isinstance(asset, dict):
110
+ errors.append(
111
+ ValidationError(
112
+ code="PTL-STAC-003",
113
+ path=asset_path,
114
+ message="asset must be an object",
115
+ )
116
+ )
117
+ continue
118
+ _require_string(asset, "href", asset_path, errors, code="PTL-STAC-003")
119
+
120
+
121
+ def _validate_links(data: JsonObject, path: str, errors: list[ValidationError]) -> None:
122
+ links = data.get("links")
123
+ if links is None:
124
+ return
125
+ if not isinstance(links, list):
126
+ errors.append(
127
+ ValidationError(
128
+ code="PTL-STAC-002",
129
+ path=f"{path}.links",
130
+ message="links must be an array",
131
+ )
132
+ )
133
+ return
134
+ for index, link in enumerate(links):
135
+ link_path = f"{path}.links[{index}]"
136
+ if not isinstance(link, dict):
137
+ errors.append(
138
+ ValidationError(
139
+ code="PTL-STAC-002",
140
+ path=link_path,
141
+ message="link must be an object",
142
+ )
143
+ )
144
+ continue
145
+ _require_string(link, "rel", link_path, errors, code="PTL-STAC-002")
146
+ _require_string(link, "href", link_path, errors, code="PTL-STAC-002")
147
+
148
+
149
+ def _require_string(
150
+ data: JsonObject,
151
+ field: str,
152
+ path: str,
153
+ errors: list[ValidationError],
154
+ *,
155
+ code: str = "PTL-STAC-001",
156
+ ) -> None:
157
+ value = data.get(field)
158
+ if not isinstance(value, str) or not value:
159
+ errors.append(
160
+ ValidationError(
161
+ code=code,
162
+ path=f"{path}.{field}",
163
+ message=f"{field} is required",
164
+ )
165
+ )
@@ -0,0 +1,224 @@
1
+ Metadata-Version: 2.5
2
+ Name: portolan-python
3
+ Version: 0.1.0
4
+ Summary: A lightweight Python implementation of the Portolan specification.
5
+ Author: Portolan contributors
6
+ License: Apache-2.0
7
+ Requires-Python: >=3.10
8
+ Provides-Extra: dev
9
+ Requires-Dist: mypy>=1.18.0; extra == 'dev'
10
+ Requires-Dist: pytest-cov>=6.0.0; extra == 'dev'
11
+ Requires-Dist: pytest>=8.0.0; extra == 'dev'
12
+ Requires-Dist: ruff>=0.13.0; extra == 'dev'
13
+ Description-Content-Type: text/markdown
14
+
15
+ # Portolan Python
16
+
17
+ A lightweight Python implementation of the Portolan specification.
18
+
19
+ `portolan-python` provides the core domain model, catalog access, validation, and conformance APIs required to work with Portolan catalogs from Python applications.
20
+
21
+ The project intentionally focuses on **Portolan itself**, rather than on geospatial data processing.
22
+
23
+ It does not aim to replace GDAL, Rasterio, GeoPandas, GeoParquet tooling, tile generators, GIS servers, or other specialized geospatial software.
24
+
25
+ ## Motivation
26
+
27
+ Portolan defines an opinionated way of organizing and describing cloud-native geospatial data using existing standards and formats such as STAC, GeoParquet, COG, PMTiles, COPC, and Zarr.
28
+
29
+ Applications integrating with Portolan need a reliable implementation of that contract.
30
+
31
+ Without a reusable core library, every integration would need to independently implement:
32
+
33
+ * Portolan catalog semantics;
34
+ * STAC traversal;
35
+ * collection and asset handling;
36
+ * Portolan-specific requirements;
37
+ * link and HREF resolution;
38
+ * metadata validation;
39
+ * conformance rules;
40
+ * serialization and deserialization.
41
+
42
+ `portolan-python` provides that reusable implementation.
43
+
44
+ The core design principle is:
45
+
46
+ > **Portolan defines the contract. Specialized tools implement data processing.**
47
+
48
+ For example, this library may determine that an asset is a GeoParquet asset and expose its URI and metadata. It does not need to read the GeoParquet rows itself.
49
+
50
+ Likewise, it may identify a COG asset without becoming a raster processing library.
51
+
52
+ ## Scope
53
+
54
+ `portolan-python` is responsible for the Portolan domain model and specification semantics.
55
+
56
+ Expected responsibilities include:
57
+
58
+ * opening Portolan catalogs;
59
+ * creating Portolan catalogs;
60
+ * reading and writing Portolan/STAC metadata;
61
+ * navigating catalogs, collections, items, assets, and links;
62
+ * resolving relative and absolute HREFs;
63
+ * exposing Portolan extensions and metadata;
64
+ * validating Portolan structures;
65
+ * checking Portolan conformance;
66
+ * exposing asset type and media-type information;
67
+ * handling specification versions;
68
+ * providing stable Python APIs for applications built on Portolan.
69
+
70
+ A conceptual API may look like:
71
+
72
+ ```python
73
+ from portolan import Catalog
74
+
75
+ catalog = Catalog.open("https://example.com/catalog.json")
76
+
77
+ for collection in catalog.collections():
78
+ print(collection.id)
79
+
80
+ for asset in collection.assets():
81
+ print(asset.href)
82
+ print(asset.media_type)
83
+ print(asset.roles)
84
+ ```
85
+
86
+ Validation should similarly be available programmatically:
87
+
88
+ ```python
89
+ from portolan import Catalog, Validator
90
+
91
+ catalog = Catalog.open("./catalog.json")
92
+
93
+ result = Validator.validate(catalog)
94
+
95
+ if not result.valid:
96
+ for error in result.errors:
97
+ print(error)
98
+ ```
99
+
100
+ The exact API will evolve during implementation, but it should remain small, explicit, typed, and independent from any CLI.
101
+
102
+ ## Non-goals
103
+
104
+ This project should **not** become a general-purpose geospatial processing library.
105
+
106
+ In particular, the core should not be responsible for:
107
+
108
+ * reading GeoParquet feature data;
109
+ * writing GeoParquet datasets;
110
+ * converting Shapefile to GeoParquet;
111
+ * creating COGs;
112
+ * reading raster pixels;
113
+ * creating PMTiles;
114
+ * processing COPC;
115
+ * converting MrSID or ECW;
116
+ * extracting data from WFS;
117
+ * extracting data from ArcGIS;
118
+ * extracting data from CARTO;
119
+ * publishing data to GeoServer;
120
+ * running pygeoapi;
121
+ * managing QGIS projects.
122
+
123
+ Those capabilities belong to specialized libraries, applications, or optional integrations.
124
+
125
+ Portolan should be opinionated about **what constitutes a conformant Portolan catalog and asset**, not unnecessarily opinionated about **which software must produce or consume those assets**.
126
+
127
+ ## Architecture
128
+
129
+ ```text
130
+ Portolan Specification
131
+ │
132
+ ▼
133
+ portolan-python
134
+ ┌────────────────────┐
135
+ │ Domain model │
136
+ │ Catalog access │
137
+ │ STAC semantics │
138
+ │ Portolan semantics │
139
+ │ HREF resolution │
140
+ │ Validation │
141
+ │ Conformance │
142
+ └─────────┬──────────┘
143
+ │
144
+ ┌──────────────┼──────────────┐
145
+ ▼ ▼ ▼
146
+ portolan-cli portolan-geoserver other apps
147
+ ```
148
+
149
+ ## Relationship with `portolan-cli`
150
+
151
+ `portolan-cli` should consume this library rather than implement Portolan domain logic itself.
152
+
153
+ For example:
154
+
155
+ ```text
156
+ portolan validate
157
+ portolan inspect
158
+ portolan registry list
159
+ ```
160
+
161
+ should be CLI representations of APIs provided by the Portolan Python ecosystem.
162
+
163
+ The CLI should remain an interface layer.
164
+
165
+ ## Relationship with `portolan-geoserver`
166
+
167
+ `portolan-geoserver` uses this library to understand Portolan catalogs and uses `python-geoservercloud` to interact with GeoServer.
168
+
169
+ The responsibilities remain separated:
170
+
171
+ ```text
172
+ portolan-python
173
+ │
174
+ │ understands Portolan
175
+ ▼
176
+ portolan-geoserver
177
+ │
178
+ │ maps Portolan resources to GeoServer
179
+ ▼
180
+ python-geoservercloud
181
+ │
182
+ │ manages GeoServer
183
+ ▼
184
+ GeoServer
185
+ ```
186
+
187
+ `portolan-python` therefore contains no GeoServer-specific logic.
188
+
189
+ ## Relationship with `portolan-java`
190
+
191
+ `portolan-java` is the Java counterpart of this project.
192
+
193
+ The two libraries should implement the same **conceptual Portolan contract**, while following the conventions of their respective languages.
194
+
195
+ They should not necessarily expose identical classes or method signatures.
196
+
197
+ The Portolan specification remains the source of truth.
198
+
199
+ ```text
200
+ Portolan Specification
201
+ / \
202
+ / \
203
+ portolan-python portolan-java
204
+ ```
205
+
206
+ ## Design principles
207
+
208
+ 1. Specification first.
209
+ 2. Small and stable public API.
210
+ 3. No CLI dependency.
211
+ 4. No GIS server dependency.
212
+ 5. No mandatory geospatial processing stack.
213
+ 6. Specialized formats are represented, not reimplemented.
214
+ 7. External applications should not need to reimplement Portolan semantics.
215
+ 8. The library should be suitable as a dependency of long-lived applications.
216
+
217
+ ## Local documentation
218
+
219
+ - [Scope](docs/scope.md) defines what belongs in this core library.
220
+ - [Examples](docs/examples.md) shows basic API usage.
221
+ - [Development](docs/development.md) lists setup, test, and build commands.
222
+ - [Distribution](docs/distribution.md) explains releases, `pip`, and PyPI publishing.
223
+
224
+ ---
@@ -0,0 +1,8 @@
1
+ portolan/__init__.py,sha256=F2FcjOpJKeDI17ckinCrb80wWh7Wg8Glv3xZYVlJr54,713
2
+ portolan/catalog.py,sha256=zE5aVm3xXY6WZeSx2zaxWJgAloqqEFhgdHkT6oTt1aw,9768
3
+ portolan/py.typed,sha256=AbpHGcgLb-kRsJGnwFEktk7uzpZOCcBY74-YBdrKVGs,1
4
+ portolan/registry.py,sha256=9ASCT6vI6JmU92W9K8MOU6mLTCkW73lCmgKqjbvaRLI,5116
5
+ portolan/validation.py,sha256=ZRjshZaErO-By0K653ruFL7EgIKBrOMvMqapLUEVfvo,5163
6
+ portolan_python-0.1.0.dist-info/METADATA,sha256=HiK2CYsK29RnyCAYyIKUoYNJl_gyZTicNGu6rADUsCU,7154
7
+ portolan_python-0.1.0.dist-info/WHEEL,sha256=W3fkpkm7-wf9vBI5Z-7s0eWkeM-spu78I8Neb98DeEg,87
8
+ portolan_python-0.1.0.dist-info/RECORD,,
@@ -0,0 +1,4 @@
1
+ Wheel-Version: 1.0
2
+ Generator: hatchling 1.32.4
3
+ Root-Is-Purelib: true
4
+ Tag: py3-none-any