kpubdata 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- kpubdata/__init__.py +52 -0
- kpubdata/catalog.py +101 -0
- kpubdata/client.py +179 -0
- kpubdata/config.py +119 -0
- kpubdata/core/__init__.py +25 -0
- kpubdata/core/capability.py +56 -0
- kpubdata/core/dataset.py +153 -0
- kpubdata/core/models.py +156 -0
- kpubdata/core/protocol.py +60 -0
- kpubdata/core/representation.py +19 -0
- kpubdata/exceptions.py +116 -0
- kpubdata/providers/__init__.py +5 -0
- kpubdata/providers/_common.py +204 -0
- kpubdata/providers/bok/__init__.py +7 -0
- kpubdata/providers/bok/adapter.py +333 -0
- kpubdata/providers/bok/catalogue.json +25 -0
- kpubdata/providers/datago/__init__.py +7 -0
- kpubdata/providers/datago/adapter.py +313 -0
- kpubdata/providers/datago/catalogue.json +92 -0
- kpubdata/providers/kosis/__init__.py +7 -0
- kpubdata/providers/kosis/adapter.py +254 -0
- kpubdata/providers/kosis/catalogue.json +26 -0
- kpubdata/providers/lofin/__init__.py +7 -0
- kpubdata/providers/lofin/adapter.py +329 -0
- kpubdata/providers/lofin/catalogue.json +127 -0
- kpubdata/py.typed +0 -0
- kpubdata/registry.py +135 -0
- kpubdata/transport/__init__.py +16 -0
- kpubdata/transport/decode.py +74 -0
- kpubdata/transport/http.py +380 -0
- kpubdata/transport/retry.py +68 -0
- kpubdata-0.1.0.dist-info/METADATA +380 -0
- kpubdata-0.1.0.dist-info/RECORD +35 -0
- kpubdata-0.1.0.dist-info/WHEEL +4 -0
- kpubdata-0.1.0.dist-info/licenses/LICENSE +21 -0
kpubdata/core/models.py
ADDED
|
@@ -0,0 +1,156 @@
|
|
|
1
|
+
"""Canonical domain models shared across providers and adapters."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from collections.abc import Callable, Iterator
|
|
6
|
+
from dataclasses import dataclass as _stdlib_dataclass
|
|
7
|
+
from dataclasses import field
|
|
8
|
+
from types import MappingProxyType
|
|
9
|
+
from typing import TypeVar
|
|
10
|
+
|
|
11
|
+
from typing_extensions import dataclass_transform
|
|
12
|
+
|
|
13
|
+
from kpubdata.core.capability import Operation, QuerySupport
|
|
14
|
+
from kpubdata.core.representation import Representation
|
|
15
|
+
|
|
16
|
+
_T = TypeVar("_T")
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
@dataclass_transform()
|
|
20
|
+
def _dataclass(
|
|
21
|
+
*,
|
|
22
|
+
slots: bool = False,
|
|
23
|
+
frozen: bool = False,
|
|
24
|
+
) -> Callable[[type[_T]], type[_T]]:
|
|
25
|
+
"""Return a decorator that applies ``dataclasses.dataclass`` with fixed options."""
|
|
26
|
+
|
|
27
|
+
def _decorate(cls: type[_T]) -> type[_T]:
|
|
28
|
+
return _stdlib_dataclass(slots=slots, frozen=frozen)(cls) # pyright: ignore[reportCallIssue]
|
|
29
|
+
|
|
30
|
+
return _decorate
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def _empty_proxy() -> MappingProxyType[str, object]:
|
|
34
|
+
"""Return an empty immutable string-keyed mapping proxy."""
|
|
35
|
+
return MappingProxyType({})
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def _empty_object_proxy() -> MappingProxyType[str, object]:
|
|
39
|
+
"""Return an empty immutable object-valued mapping proxy."""
|
|
40
|
+
return MappingProxyType({})
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
@_dataclass(slots=True, frozen=True)
|
|
44
|
+
class DatasetRef:
|
|
45
|
+
"""Canonical immutable reference to a provider dataset.
|
|
46
|
+
|
|
47
|
+
Attributes:
|
|
48
|
+
query_support: Structured list-query feature metadata, if known.
|
|
49
|
+
raw_metadata: Provider-native discovery metadata for debugging.
|
|
50
|
+
"""
|
|
51
|
+
|
|
52
|
+
id: str
|
|
53
|
+
provider: str
|
|
54
|
+
dataset_key: str
|
|
55
|
+
name: str
|
|
56
|
+
representation: Representation
|
|
57
|
+
operations: frozenset[Operation] = frozenset()
|
|
58
|
+
query_support: QuerySupport | None = None
|
|
59
|
+
raw_metadata: MappingProxyType[str, object] = field(default_factory=_empty_proxy)
|
|
60
|
+
|
|
61
|
+
def supports(self, op: Operation) -> bool:
|
|
62
|
+
"""Return whether this dataset supports the requested operation."""
|
|
63
|
+
|
|
64
|
+
return op in self.operations
|
|
65
|
+
|
|
66
|
+
def __repr__(self) -> str:
|
|
67
|
+
"""Return a concise developer-friendly representation."""
|
|
68
|
+
ops = ", ".join(sorted(operation.value for operation in self.operations))
|
|
69
|
+
return f"DatasetRef(id={self.id!r}, provider={self.provider!r}, ops=[{ops}])"
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
@_dataclass(slots=True)
|
|
73
|
+
class Query:
|
|
74
|
+
"""Provider-agnostic query object for listing records.
|
|
75
|
+
|
|
76
|
+
Attributes:
|
|
77
|
+
filters: Provider-specific filter payload merged into query translation.
|
|
78
|
+
extra: Additional provider-native parameters not covered canonically.
|
|
79
|
+
"""
|
|
80
|
+
|
|
81
|
+
filters: dict[str, object] = field(default_factory=dict)
|
|
82
|
+
page: int | None = None
|
|
83
|
+
page_size: int | None = None
|
|
84
|
+
cursor: str | None = None
|
|
85
|
+
start_date: str | None = None
|
|
86
|
+
end_date: str | None = None
|
|
87
|
+
fields: list[str] | None = None
|
|
88
|
+
sort: list[str] | None = None
|
|
89
|
+
extra: dict[str, object] = field(default_factory=dict)
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
@_dataclass(slots=True)
|
|
93
|
+
class RecordBatch:
|
|
94
|
+
"""Batch of normalized records returned from a dataset query.
|
|
95
|
+
|
|
96
|
+
Attributes:
|
|
97
|
+
next_page: Next offset page number for offset pagination.
|
|
98
|
+
next_cursor: Opaque cursor token for cursor pagination.
|
|
99
|
+
raw: Provider-native response payload used to derive this batch.
|
|
100
|
+
meta: Additional adapter metadata that does not fit canonical fields.
|
|
101
|
+
"""
|
|
102
|
+
|
|
103
|
+
items: list[dict[str, object]]
|
|
104
|
+
dataset: DatasetRef
|
|
105
|
+
total_count: int | None = None
|
|
106
|
+
next_page: int | None = None
|
|
107
|
+
next_cursor: str | None = None
|
|
108
|
+
raw: object | None = None
|
|
109
|
+
meta: dict[str, object] = field(default_factory=dict)
|
|
110
|
+
|
|
111
|
+
def __len__(self) -> int:
|
|
112
|
+
return len(self.items)
|
|
113
|
+
|
|
114
|
+
def __iter__(self) -> Iterator[dict[str, object]]:
|
|
115
|
+
return iter(self.items)
|
|
116
|
+
|
|
117
|
+
def __bool__(self) -> bool:
|
|
118
|
+
return bool(self.items)
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
@_dataclass(slots=True)
|
|
122
|
+
class FieldDescriptor:
|
|
123
|
+
"""Describe a single field in a dataset schema.
|
|
124
|
+
|
|
125
|
+
Attributes:
|
|
126
|
+
raw: Provider-native field metadata retained for advanced use.
|
|
127
|
+
"""
|
|
128
|
+
|
|
129
|
+
name: str
|
|
130
|
+
title: str | None = None
|
|
131
|
+
type: str | None = None
|
|
132
|
+
description: str | None = None
|
|
133
|
+
nullable: bool | None = None
|
|
134
|
+
raw: MappingProxyType[str, object] = field(default_factory=_empty_object_proxy)
|
|
135
|
+
|
|
136
|
+
|
|
137
|
+
@_dataclass(slots=True)
|
|
138
|
+
class SchemaDescriptor:
|
|
139
|
+
"""Describe schema metadata exposed for a dataset.
|
|
140
|
+
|
|
141
|
+
Attributes:
|
|
142
|
+
raw: Provider-native schema metadata retained without normalization.
|
|
143
|
+
"""
|
|
144
|
+
|
|
145
|
+
dataset: DatasetRef
|
|
146
|
+
fields: list[FieldDescriptor]
|
|
147
|
+
raw: MappingProxyType[str, object] = field(default_factory=_empty_object_proxy)
|
|
148
|
+
|
|
149
|
+
|
|
150
|
+
__all__ = [
|
|
151
|
+
"DatasetRef",
|
|
152
|
+
"FieldDescriptor",
|
|
153
|
+
"Query",
|
|
154
|
+
"RecordBatch",
|
|
155
|
+
"SchemaDescriptor",
|
|
156
|
+
]
|
|
@@ -0,0 +1,60 @@
|
|
|
1
|
+
"""Provider adapter protocol — the extension point for KPubData."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from typing import Protocol, runtime_checkable
|
|
6
|
+
|
|
7
|
+
from kpubdata.core.models import DatasetRef, Query, RecordBatch, SchemaDescriptor
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
@runtime_checkable
|
|
11
|
+
class ProviderAdapter(Protocol):
|
|
12
|
+
"""Protocol that every provider adapter must satisfy.
|
|
13
|
+
|
|
14
|
+
Adapters are responsible for auth, discovery, translation, error mapping,
|
|
15
|
+
raw access, and truthful capability declaration.
|
|
16
|
+
"""
|
|
17
|
+
|
|
18
|
+
@property
|
|
19
|
+
def name(self) -> str:
|
|
20
|
+
"""Return provider identifier (for example ``datago`` or ``seoul``)."""
|
|
21
|
+
|
|
22
|
+
...
|
|
23
|
+
|
|
24
|
+
def list_datasets(self) -> list[DatasetRef]:
|
|
25
|
+
"""Return discoverable datasets for this provider."""
|
|
26
|
+
|
|
27
|
+
...
|
|
28
|
+
|
|
29
|
+
def search_datasets(self, text: str) -> list[DatasetRef]:
|
|
30
|
+
"""Return datasets matching free-text search for this provider."""
|
|
31
|
+
|
|
32
|
+
...
|
|
33
|
+
|
|
34
|
+
def get_dataset(self, dataset_key: str) -> DatasetRef:
|
|
35
|
+
"""Resolve a provider-local dataset key to a canonical dataset ref."""
|
|
36
|
+
|
|
37
|
+
...
|
|
38
|
+
|
|
39
|
+
def query_records(self, dataset: DatasetRef, query: Query) -> RecordBatch:
|
|
40
|
+
"""Execute a canonical list/query request for a dataset."""
|
|
41
|
+
|
|
42
|
+
...
|
|
43
|
+
|
|
44
|
+
def get_record(self, dataset: DatasetRef, key: dict[str, object]) -> dict[str, object] | None:
|
|
45
|
+
"""Return a single normalized record by key or ``None`` if missing."""
|
|
46
|
+
|
|
47
|
+
...
|
|
48
|
+
|
|
49
|
+
def get_schema(self, dataset: DatasetRef) -> SchemaDescriptor | None:
|
|
50
|
+
"""Return canonical schema metadata if supported."""
|
|
51
|
+
|
|
52
|
+
...
|
|
53
|
+
|
|
54
|
+
def call_raw(self, dataset: DatasetRef, operation: str, params: dict[str, object]) -> object:
|
|
55
|
+
"""Execute provider-native operation and return unnormalized response."""
|
|
56
|
+
|
|
57
|
+
...
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
__all__ = ["ProviderAdapter"]
|
|
@@ -0,0 +1,19 @@
|
|
|
1
|
+
"""Dataset representation types for source/shape modeling."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from enum import Enum
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
class Representation(str, Enum):
|
|
9
|
+
"""How a dataset is provisioned: source shape rather than access mode."""
|
|
10
|
+
|
|
11
|
+
API_JSON = "api_json"
|
|
12
|
+
API_XML = "api_xml"
|
|
13
|
+
FILE_CSV = "file_csv"
|
|
14
|
+
FILE_EXCEL = "file_excel"
|
|
15
|
+
SHEET = "sheet"
|
|
16
|
+
OTHER = "other"
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
__all__ = ["Representation"]
|
kpubdata/exceptions.py
ADDED
|
@@ -0,0 +1,116 @@
|
|
|
1
|
+
"""KPubData exception hierarchy with structured error context."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from typing import Any
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
class PublicDataError(Exception):
|
|
9
|
+
"""Base for all KPubData errors with structured context attributes."""
|
|
10
|
+
|
|
11
|
+
def __init__(
|
|
12
|
+
self,
|
|
13
|
+
message: str,
|
|
14
|
+
*,
|
|
15
|
+
provider: str | None = None,
|
|
16
|
+
dataset_id: str | None = None,
|
|
17
|
+
operation: str | None = None,
|
|
18
|
+
status_code: int | None = None,
|
|
19
|
+
provider_code: str | None = None,
|
|
20
|
+
retryable: bool = False,
|
|
21
|
+
detail: object = None,
|
|
22
|
+
) -> None:
|
|
23
|
+
"""Initialize an error with optional provider and transport metadata."""
|
|
24
|
+
|
|
25
|
+
super().__init__(message)
|
|
26
|
+
self.provider = provider
|
|
27
|
+
self.dataset_id = dataset_id
|
|
28
|
+
self.operation = operation
|
|
29
|
+
self.status_code = status_code
|
|
30
|
+
self.provider_code = provider_code
|
|
31
|
+
self.retryable = retryable
|
|
32
|
+
self.detail = detail
|
|
33
|
+
|
|
34
|
+
def __repr__(self) -> str:
|
|
35
|
+
"""Return a structured repr including provider and transport metadata."""
|
|
36
|
+
parts = [f"{type(self).__name__}({self.args[0]!r}"]
|
|
37
|
+
if self.provider:
|
|
38
|
+
parts.append(f"provider={self.provider!r}")
|
|
39
|
+
if self.dataset_id:
|
|
40
|
+
parts.append(f"dataset={self.dataset_id!r}")
|
|
41
|
+
if self.status_code is not None:
|
|
42
|
+
parts.append(f"status={self.status_code}")
|
|
43
|
+
if self.retryable:
|
|
44
|
+
parts.append("retryable=True")
|
|
45
|
+
return ", ".join(parts) + ")"
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
class ConfigError(PublicDataError):
|
|
49
|
+
"""Raised when KPubData configuration is invalid or incomplete."""
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
class AuthError(PublicDataError):
|
|
53
|
+
"""Raised for authentication or authorization failures."""
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
class TransportError(PublicDataError):
|
|
57
|
+
"""Raised for network and transport-layer failures."""
|
|
58
|
+
|
|
59
|
+
def __init__(self, message: str, **kwargs: Any) -> None:
|
|
60
|
+
"""Initialize a retryable transport error by default."""
|
|
61
|
+
|
|
62
|
+
kwargs.setdefault("retryable", True)
|
|
63
|
+
super().__init__(message, **kwargs)
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
class TransportTimeoutError(TransportError):
|
|
67
|
+
"""Raised when a provider request exceeds timeout limits."""
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
class RateLimitError(TransportError):
|
|
71
|
+
"""Raised when the provider rejects requests due to throttling."""
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
class ServiceUnavailableError(TransportError):
|
|
75
|
+
"""Raised when the upstream provider service is temporarily unavailable."""
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
class ParseError(PublicDataError):
|
|
79
|
+
"""Raised when provider payloads cannot be parsed safely."""
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
class InvalidRequestError(PublicDataError):
|
|
83
|
+
"""Raised when query or operation inputs are semantically invalid."""
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
class ProviderResponseError(PublicDataError):
|
|
87
|
+
"""Raised when provider responses violate contract expectations."""
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
class UnsupportedCapabilityError(PublicDataError):
|
|
91
|
+
"""Raised when a requested operation is unsupported for a dataset."""
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
class DatasetNotFoundError(PublicDataError):
|
|
95
|
+
"""Raised when a requested dataset identifier cannot be resolved."""
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
class ProviderNotRegisteredError(PublicDataError):
|
|
99
|
+
"""Raised when a provider key is not present in the registry."""
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
__all__ = [
|
|
103
|
+
"AuthError",
|
|
104
|
+
"ConfigError",
|
|
105
|
+
"DatasetNotFoundError",
|
|
106
|
+
"InvalidRequestError",
|
|
107
|
+
"ParseError",
|
|
108
|
+
"ProviderNotRegisteredError",
|
|
109
|
+
"ProviderResponseError",
|
|
110
|
+
"PublicDataError",
|
|
111
|
+
"RateLimitError",
|
|
112
|
+
"ServiceUnavailableError",
|
|
113
|
+
"TransportError",
|
|
114
|
+
"TransportTimeoutError",
|
|
115
|
+
"UnsupportedCapabilityError",
|
|
116
|
+
]
|
|
@@ -0,0 +1,204 @@
|
|
|
1
|
+
"""Shared catalogue and schema utilities for provider adapters."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import json
|
|
6
|
+
from collections.abc import Mapping
|
|
7
|
+
from importlib.resources import files
|
|
8
|
+
from types import MappingProxyType
|
|
9
|
+
from typing import cast
|
|
10
|
+
|
|
11
|
+
from kpubdata.core.capability import Operation, PaginationMode, QuerySupport
|
|
12
|
+
from kpubdata.core.models import (
|
|
13
|
+
DatasetRef,
|
|
14
|
+
FieldDescriptor,
|
|
15
|
+
SchemaDescriptor,
|
|
16
|
+
)
|
|
17
|
+
from kpubdata.core.representation import Representation
|
|
18
|
+
from kpubdata.exceptions import ConfigError
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def load_catalogue(package_name: str, provider: str) -> tuple[DatasetRef, ...]:
|
|
22
|
+
"""Load and parse a catalogue.json from a provider package."""
|
|
23
|
+
package_files = files(package_name)
|
|
24
|
+
catalogue_text = package_files.joinpath("catalogue.json").read_text(encoding="utf-8")
|
|
25
|
+
parsed_catalogue = cast(object, json.loads(catalogue_text))
|
|
26
|
+
if not isinstance(parsed_catalogue, list):
|
|
27
|
+
msg = f"{provider} catalogue.json must contain a top-level JSON array"
|
|
28
|
+
raise ConfigError(msg, provider=provider)
|
|
29
|
+
|
|
30
|
+
catalogue_entries = cast(list[object], parsed_catalogue)
|
|
31
|
+
datasets: list[DatasetRef] = []
|
|
32
|
+
for entry_object in catalogue_entries:
|
|
33
|
+
if not isinstance(entry_object, dict):
|
|
34
|
+
msg = f"{provider} catalogue entries must be JSON objects"
|
|
35
|
+
raise ConfigError(msg, provider=provider)
|
|
36
|
+
typed_entry_object = cast(dict[object, object], entry_object)
|
|
37
|
+
entry: dict[str, object] = {}
|
|
38
|
+
for key, value in typed_entry_object.items():
|
|
39
|
+
if not isinstance(key, str):
|
|
40
|
+
msg = f"{provider} catalogue entry keys must be strings"
|
|
41
|
+
raise ConfigError(msg, provider=provider)
|
|
42
|
+
entry[key] = value
|
|
43
|
+
datasets.append(build_dataset_ref(provider, entry))
|
|
44
|
+
|
|
45
|
+
dataset_ids = [dataset.id for dataset in datasets]
|
|
46
|
+
duplicate_ids = sorted(
|
|
47
|
+
{dataset_id for dataset_id in dataset_ids if dataset_ids.count(dataset_id) > 1}
|
|
48
|
+
)
|
|
49
|
+
if duplicate_ids:
|
|
50
|
+
duplicates = ", ".join(duplicate_ids)
|
|
51
|
+
msg = f"{provider} catalogue contains duplicate dataset ids: {duplicates}"
|
|
52
|
+
raise ConfigError(msg, provider=provider)
|
|
53
|
+
|
|
54
|
+
return tuple(datasets)
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
def build_dataset_ref(provider: str, entry: dict[str, object]) -> DatasetRef:
|
|
58
|
+
"""Build a DatasetRef from a raw catalogue entry dict."""
|
|
59
|
+
dataset_key = require_string_field(entry, "dataset_key", provider)
|
|
60
|
+
name = require_string_field(entry, "name", provider)
|
|
61
|
+
representation_value = require_string_field(entry, "representation", provider)
|
|
62
|
+
dataset_id = f"{provider}.{dataset_key}"
|
|
63
|
+
try:
|
|
64
|
+
representation = Representation(representation_value)
|
|
65
|
+
except ValueError as exc:
|
|
66
|
+
msg = f"{provider} catalogue entry has invalid representation: {representation_value}"
|
|
67
|
+
raise ConfigError(msg, provider=provider, dataset_id=dataset_id) from exc
|
|
68
|
+
|
|
69
|
+
ops_raw_obj = entry.get("operations", [])
|
|
70
|
+
ops_raw = cast(list[object], ops_raw_obj) if isinstance(ops_raw_obj, list) else []
|
|
71
|
+
operations: set[Operation] = set()
|
|
72
|
+
for op_raw in ops_raw:
|
|
73
|
+
if not isinstance(op_raw, str):
|
|
74
|
+
msg = f"{provider} catalogue entry has non-string operation value: {op_raw!r}"
|
|
75
|
+
raise ConfigError(msg, provider=provider, dataset_id=dataset_id)
|
|
76
|
+
try:
|
|
77
|
+
operations.add(Operation(op_raw))
|
|
78
|
+
except ValueError as exc:
|
|
79
|
+
msg = f"{provider} catalogue entry has invalid operation: {op_raw}"
|
|
80
|
+
raise ConfigError(msg, provider=provider, dataset_id=dataset_id) from exc
|
|
81
|
+
|
|
82
|
+
query_support = _parse_query_support(entry, provider)
|
|
83
|
+
|
|
84
|
+
raw_metadata = MappingProxyType(
|
|
85
|
+
{
|
|
86
|
+
key: value
|
|
87
|
+
for key, value in entry.items()
|
|
88
|
+
if key not in ("dataset_key", "name", "representation", "operations", "query_support")
|
|
89
|
+
}
|
|
90
|
+
)
|
|
91
|
+
|
|
92
|
+
return DatasetRef(
|
|
93
|
+
id=dataset_id,
|
|
94
|
+
provider=provider,
|
|
95
|
+
dataset_key=dataset_key,
|
|
96
|
+
name=name,
|
|
97
|
+
representation=representation,
|
|
98
|
+
operations=frozenset(operations),
|
|
99
|
+
query_support=query_support,
|
|
100
|
+
raw_metadata=raw_metadata,
|
|
101
|
+
)
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
def _parse_query_support(entry: dict[str, object], provider: str) -> QuerySupport | None:
|
|
105
|
+
"""Parse query_support from a catalogue entry."""
|
|
106
|
+
qs_raw_obj = entry.get("query_support")
|
|
107
|
+
if not isinstance(qs_raw_obj, dict):
|
|
108
|
+
return None
|
|
109
|
+
|
|
110
|
+
qs_raw = cast(dict[str, object], qs_raw_obj)
|
|
111
|
+
pagination_raw = qs_raw.get("pagination", "none")
|
|
112
|
+
valid_pagination = {member.value for member in PaginationMode}
|
|
113
|
+
if isinstance(pagination_raw, str):
|
|
114
|
+
if pagination_raw not in valid_pagination:
|
|
115
|
+
msg = f"{provider} query_support.pagination has invalid value: {pagination_raw}"
|
|
116
|
+
raise ConfigError(msg, provider=provider)
|
|
117
|
+
pagination = PaginationMode(pagination_raw)
|
|
118
|
+
else:
|
|
119
|
+
pagination = PaginationMode.NONE
|
|
120
|
+
|
|
121
|
+
max_page_size = None
|
|
122
|
+
if "max_page_size" in qs_raw:
|
|
123
|
+
max_page_size_raw = qs_raw["max_page_size"]
|
|
124
|
+
if isinstance(max_page_size_raw, int):
|
|
125
|
+
max_page_size = max_page_size_raw
|
|
126
|
+
elif isinstance(max_page_size_raw, str):
|
|
127
|
+
max_page_size = int(max_page_size_raw)
|
|
128
|
+
else:
|
|
129
|
+
msg = f"{provider} query_support.max_page_size must be int-like"
|
|
130
|
+
raise ConfigError(msg, provider=provider)
|
|
131
|
+
|
|
132
|
+
return QuerySupport(pagination=pagination, max_page_size=max_page_size)
|
|
133
|
+
|
|
134
|
+
|
|
135
|
+
def build_schema_from_metadata(dataset: DatasetRef) -> SchemaDescriptor | None:
|
|
136
|
+
"""Build SchemaDescriptor from catalogue metadata fields."""
|
|
137
|
+
fields_raw = dataset.raw_metadata.get("fields")
|
|
138
|
+
if not isinstance(fields_raw, list) or not fields_raw:
|
|
139
|
+
return None
|
|
140
|
+
|
|
141
|
+
entries = cast(list[object], fields_raw)
|
|
142
|
+
field_descriptors: list[FieldDescriptor] = []
|
|
143
|
+
for entry_obj in entries:
|
|
144
|
+
if not isinstance(entry_obj, dict):
|
|
145
|
+
continue
|
|
146
|
+
entry = cast(dict[str, object], entry_obj)
|
|
147
|
+
name_raw = entry.get("name")
|
|
148
|
+
if not isinstance(name_raw, str) or not name_raw:
|
|
149
|
+
continue
|
|
150
|
+
title_raw = entry.get("title")
|
|
151
|
+
type_raw = entry.get("type")
|
|
152
|
+
desc_raw = entry.get("description")
|
|
153
|
+
nullable_raw = entry.get("nullable")
|
|
154
|
+
field_descriptors.append(
|
|
155
|
+
FieldDescriptor(
|
|
156
|
+
name=name_raw,
|
|
157
|
+
title=title_raw if isinstance(title_raw, str) else None,
|
|
158
|
+
type=type_raw if isinstance(type_raw, str) else None,
|
|
159
|
+
description=desc_raw if isinstance(desc_raw, str) else None,
|
|
160
|
+
nullable=nullable_raw if isinstance(nullable_raw, bool) else None,
|
|
161
|
+
raw=MappingProxyType({k: v for k, v in entry.items() if k != "name"}),
|
|
162
|
+
)
|
|
163
|
+
)
|
|
164
|
+
|
|
165
|
+
if not field_descriptors:
|
|
166
|
+
return None
|
|
167
|
+
|
|
168
|
+
return SchemaDescriptor(
|
|
169
|
+
dataset=dataset,
|
|
170
|
+
fields=field_descriptors,
|
|
171
|
+
raw=MappingProxyType({"source": "catalogue"}),
|
|
172
|
+
)
|
|
173
|
+
|
|
174
|
+
|
|
175
|
+
def coerce_int(value: object, default: int) -> int:
|
|
176
|
+
"""Coerce a value to int, returning default on failure."""
|
|
177
|
+
if isinstance(value, int):
|
|
178
|
+
return value
|
|
179
|
+
if isinstance(value, str):
|
|
180
|
+
try:
|
|
181
|
+
return int(value)
|
|
182
|
+
except ValueError:
|
|
183
|
+
return default
|
|
184
|
+
return default
|
|
185
|
+
|
|
186
|
+
|
|
187
|
+
def require_string_field(entry: Mapping[str, object], field_name: str, provider: str) -> str:
|
|
188
|
+
"""Extract a required non-empty string field from a catalogue entry."""
|
|
189
|
+
value = entry.get(field_name)
|
|
190
|
+
if isinstance(value, str) and value:
|
|
191
|
+
return value
|
|
192
|
+
raise ConfigError(
|
|
193
|
+
f"{provider} catalogue entry missing non-empty string field: {field_name}",
|
|
194
|
+
provider=provider,
|
|
195
|
+
)
|
|
196
|
+
|
|
197
|
+
|
|
198
|
+
__all__ = [
|
|
199
|
+
"build_dataset_ref",
|
|
200
|
+
"build_schema_from_metadata",
|
|
201
|
+
"coerce_int",
|
|
202
|
+
"load_catalogue",
|
|
203
|
+
"require_string_field",
|
|
204
|
+
]
|