sdmxlib 0.62.2__tar.gz → 0.63.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/PKG-INFO +1 -1
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/pyproject.toml +1 -1
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/pyproject.toml.orig +1 -1
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/__init__.py +1 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/api/federated.py +3 -3
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/api/registry.py +3 -3
- sdmxlib-0.63.0/src/sdmxlib/formats/sdmx_csv/reader.py +321 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/formats/sdmx_csv/writer.py +35 -6
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/local/registry.py +9 -6
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/model/__init__.py +1 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/model/_items.py +92 -2
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/model/category.py +17 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/model/codelist.py +17 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/model/concept.py +8 -11
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/model/dataset.py +23 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/model/datastructure.py +25 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/model/in_memory_registry.py +2 -2
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/model/metadatastructure.py +17 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/model/organisation.py +50 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/model/ref.py +19 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/model/urn.py +5 -3
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/model/validation.py +36 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/polars.py +65 -9
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/storage/lazy.py +23 -4
- sdmxlib-0.62.2/src/sdmxlib/formats/sdmx_csv/reader.py +0 -162
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/README.md +0 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/_duckdb.py +0 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/_singleflight.py +0 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/api/__init__.py +0 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/api/_freshness.py +0 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/api/client.py +0 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/api/filters.py +0 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/api/policy.py +0 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/api/providers.py +0 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/api/query.py +0 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/api/session.py +0 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/builders/__init__.py +0 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/builders/contentconstraint.py +0 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/builders/metadata_annotation.py +0 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/data_store.py +0 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/formats/__init__.py +0 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/formats/_coerce.py +0 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/formats/_index.py +0 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/formats/data_format.py +0 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/formats/gaps.py +0 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/formats/sdmx_csv/__init__.py +0 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/formats/sdmx_json/__init__.py +0 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/formats/sdmx_json/_data_common.py +0 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/formats/sdmx_json/_dto.py +0 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/formats/sdmx_json/_resolve.py +0 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/formats/sdmx_json/_structure_dto.py +0 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/formats/sdmx_json/_structure_ingest.py +0 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/formats/sdmx_json/data_v1.py +0 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/formats/sdmx_json/data_v2.py +0 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/formats/sdmx_json/metadata.py +0 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/formats/sdmx_json/reader.py +0 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/formats/sdmx_json/structure_values.py +0 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/formats/sdmx_json/writer.py +0 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/formats/sdmx_json_1_0/__init__.py +0 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/formats/sdmx_json_1_0/writer.py +0 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/formats/sdmx_ml/__init__.py +0 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/formats/sdmx_ml/_common.py +0 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/formats/sdmx_ml/_v3_reader.py +0 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/formats/sdmx_ml/_v3_writer.py +0 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/formats/sdmx_ml21/__init__.py +0 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/formats/sdmx_ml21/_synthetic.py +0 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/formats/sdmx_ml21/namespaces.py +0 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/formats/sdmx_ml21/reader.py +0 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/formats/sdmx_ml21/writer.py +0 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/formats/sdmx_ml30/__init__.py +0 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/formats/sdmx_ml30/metadata.py +0 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/formats/sdmx_ml30/namespaces.py +0 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/formats/sdmx_ml30/reader.py +0 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/formats/sdmx_ml30/writer.py +0 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/formats/sdmx_ml31/__init__.py +0 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/formats/sdmx_ml31/namespaces.py +0 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/formats/sdmx_ml31/reader.py +0 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/formats/sdmx_ml31/writer.py +0 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/local/__init__.py +0 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/local/data_store.py +0 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/local/sql.py +0 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/model/_interning.py +0 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/model/_label_match.py +0 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/model/_refshape.py +0 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/model/annotations.py +0 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/model/base.py +0 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/model/binding.py +0 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/model/collections.py +0 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/model/constraint.py +0 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/model/convert.py +0 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/model/dataflow.py +0 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/model/errors.py +0 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/model/expr.py +0 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/model/hierarchy.py +0 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/model/istring.py +0 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/model/mapping.py +0 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/model/message.py +0 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/model/metadata_provision.py +0 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/model/metadataflow.py +0 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/model/metadataset.py +0 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/model/provision.py +0 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/model/registry.py +0 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/model/representation.py +0 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/model/rest_api.py +0 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/model/rollup.py +0 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/model/version.py +0 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/py.typed +0 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/query/__init__.py +0 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/query/_closure.py +0 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/query/_common.py +0 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/query/_compile.py +0 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/query/_edges.py +0 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/query/_execute.py +0 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/query/_expr.py +0 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/query/_joins.py +0 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/query/_plan.py +0 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/query/_pushdown.py +0 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/query/_rebase.py +0 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/query/_rest.py +0 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/query/accessors.py +0 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/query/observations.py +0 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/query/query.py +0 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/query/raw.py +0 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/query/schema.py +0 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/rest.py +0 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/session.py +0 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/sql.py +0 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/storage/__init__.py +0 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/storage/_kinds.py +0 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/storage/_projections.py +0 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/storage/_time.py +0 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/storage/readers.py +0 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/storage/resolve.py +0 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/storage/schema.py +0 -0
- {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/storage/writers.py +0 -0
|
@@ -11,6 +11,7 @@ from sdmxlib.api.client import ResponseTooLargeError
|
|
|
11
11
|
from sdmxlib.api.registry import BearerToken, FmrRegistry, RestRegistry
|
|
12
12
|
from sdmxlib.data_store import DataStore, WriteSource
|
|
13
13
|
from sdmxlib.model.datastructure import ComponentRole
|
|
14
|
+
from sdmxlib.model.ref import short_urn as short_urn
|
|
14
15
|
from sdmxlib.model.ref import SupportsUrn
|
|
15
16
|
from sdmxlib.model.rest_api import ApiVersion
|
|
16
17
|
from sdmxlib.model.rollup import AncestorMap
|
|
@@ -32,7 +32,7 @@ Example:
|
|
|
32
32
|
|
|
33
33
|
import functools
|
|
34
34
|
import inspect
|
|
35
|
-
from typing import TYPE_CHECKING,
|
|
35
|
+
from typing import TYPE_CHECKING, overload
|
|
36
36
|
|
|
37
37
|
import httpx
|
|
38
38
|
|
|
@@ -124,7 +124,7 @@ class FederatedRegistry:
|
|
|
124
124
|
@overload
|
|
125
125
|
def get[T](
|
|
126
126
|
self,
|
|
127
|
-
target: "SdmxUrn[object] | Ref[
|
|
127
|
+
target: "SdmxUrn[object] | Ref[object]",
|
|
128
128
|
/,
|
|
129
129
|
*,
|
|
130
130
|
expect: "type[T]",
|
|
@@ -144,7 +144,7 @@ class FederatedRegistry:
|
|
|
144
144
|
|
|
145
145
|
def get[T](
|
|
146
146
|
self,
|
|
147
|
-
target: "SdmxUrn[object] | Ref[
|
|
147
|
+
target: "SdmxUrn[object] | Ref[object]",
|
|
148
148
|
/,
|
|
149
149
|
*,
|
|
150
150
|
expect: "type[T] | None" = None,
|
|
@@ -4,7 +4,7 @@ import tempfile
|
|
|
4
4
|
import threading
|
|
5
5
|
from collections.abc import Callable, Sequence
|
|
6
6
|
from pathlib import Path
|
|
7
|
-
from typing import TYPE_CHECKING,
|
|
7
|
+
from typing import TYPE_CHECKING, Self, Unpack, cast, overload
|
|
8
8
|
|
|
9
9
|
import attrs
|
|
10
10
|
from attrs import frozen
|
|
@@ -779,7 +779,7 @@ class RestRegistry:
|
|
|
779
779
|
@overload
|
|
780
780
|
def get[T](
|
|
781
781
|
self,
|
|
782
|
-
target: "SdmxUrn[object] | Ref[
|
|
782
|
+
target: "SdmxUrn[object] | Ref[object]",
|
|
783
783
|
/,
|
|
784
784
|
*,
|
|
785
785
|
expect: "type[T]",
|
|
@@ -799,7 +799,7 @@ class RestRegistry:
|
|
|
799
799
|
|
|
800
800
|
def get[T](
|
|
801
801
|
self,
|
|
802
|
-
target: "SdmxUrn[object] | Ref[
|
|
802
|
+
target: "SdmxUrn[object] | Ref[object]",
|
|
803
803
|
/,
|
|
804
804
|
*,
|
|
805
805
|
expect: "type[T] | None" = None,
|
|
@@ -0,0 +1,321 @@
|
|
|
1
|
+
"""SDMX-CSV data reader.
|
|
2
|
+
|
|
3
|
+
Parses SDMX-CSV messages into `Dataset` objects.
|
|
4
|
+
|
|
5
|
+
SDMX-CSV format::
|
|
6
|
+
|
|
7
|
+
DATAFLOW,FREQ,REF_AREA,INDICATOR,TIME_PERIOD,OBS_VALUE,OBS_STATUS
|
|
8
|
+
ESTAT:STS_INPR_M(1.0),M,AT,INDPRO,2024-01,98.5,A
|
|
9
|
+
|
|
10
|
+
The ``DATAFLOW`` column identifies the dataflow but is not a DSD component.
|
|
11
|
+
It is extracted and then dropped; the remaining columns are parsed using the
|
|
12
|
+
DSD schema.
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
import csv
|
|
16
|
+
import io
|
|
17
|
+
import re
|
|
18
|
+
from pathlib import Path
|
|
19
|
+
from typing import TYPE_CHECKING, Final
|
|
20
|
+
|
|
21
|
+
import polars as pl
|
|
22
|
+
|
|
23
|
+
import sdmxlib.polars as slpl
|
|
24
|
+
from sdmxlib.model.dataflow import Dataflow
|
|
25
|
+
from sdmxlib.model.dataset import Dataset
|
|
26
|
+
from sdmxlib.model.datastructure import DataStructure
|
|
27
|
+
from sdmxlib.model.provision import ProvisionAgreement
|
|
28
|
+
from sdmxlib.model.ref import Ref, short_urn
|
|
29
|
+
from sdmxlib.model.urn import SdmxUrn
|
|
30
|
+
|
|
31
|
+
if TYPE_CHECKING:
|
|
32
|
+
from sdmxlib.model.registry import SupportsGet
|
|
33
|
+
from sdmxlib.polars import CodedAs
|
|
34
|
+
|
|
35
|
+
_DATAFLOW_COL = "DATAFLOW"
|
|
36
|
+
_KEY_COL = "KEY"
|
|
37
|
+
# SDMX-CSV 2.0.0 envelope columns — not DSD components, dropped on read.
|
|
38
|
+
_STRUCTURE_COL = "STRUCTURE"
|
|
39
|
+
_STRUCTURE_ID_COL = "STRUCTURE_ID"
|
|
40
|
+
_ACTION_COL = "ACTION"
|
|
41
|
+
_DATAFLOW_RE = re.compile(r"^([^:]+):([^(]+)\(([^)]+)\)$")
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def parse_dataflow_id(dataflow_id: str) -> tuple[str, str, str]:
|
|
45
|
+
"""Parse an SDMX-CSV DATAFLOW value into ``(agency, id, version)``.
|
|
46
|
+
|
|
47
|
+
Example:
|
|
48
|
+
parse_dataflow_id("ESTAT:STS_INPR_M(1.0)") # ("ESTAT", "STS_INPR_M", "1.0")
|
|
49
|
+
"""
|
|
50
|
+
m = _DATAFLOW_RE.match(dataflow_id)
|
|
51
|
+
if m is None:
|
|
52
|
+
msg = f"Cannot parse DATAFLOW value {dataflow_id!r} — expected 'AGENCY:ID(VERSION)'"
|
|
53
|
+
raise ValueError(msg)
|
|
54
|
+
return m.group(1), m.group(2), m.group(3)
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
#: The values SDMX-CSV 2.0's ``STRUCTURE`` column may carry.
|
|
58
|
+
#:
|
|
59
|
+
#: Verbatim from the field guide: "The first column contains: ``dataflow``,
|
|
60
|
+
#: ``datastructure`` or ``dataprovision``". Worth stating because two of our own
|
|
61
|
+
#: docstrings said ``provisionagreement`` — which is the RFC 5988 *link
|
|
62
|
+
#: relation* SDMX-JSON uses, a different vocabulary — and a caller following
|
|
63
|
+
#: them wrote a non-conformant file. sdmx1's CSV reader enumerates the same
|
|
64
|
+
#: three (#409).
|
|
65
|
+
STRUCTURE_KINDS: Final[tuple[str, ...]] = ("dataflow", "datastructure", "dataprovision")
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def _peek_dataflow_id(source: "str | Path | bytes") -> str | None:
|
|
69
|
+
"""Extract the DATAFLOW value from the first data row without reading the whole file."""
|
|
70
|
+
if isinstance(source, (str, Path)):
|
|
71
|
+
with Path(source).open(newline="", encoding="utf-8") as f:
|
|
72
|
+
head = f.read(4096)
|
|
73
|
+
else:
|
|
74
|
+
head = source[:4096].decode("utf-8", errors="replace")
|
|
75
|
+
|
|
76
|
+
reader = csv.reader(io.StringIO(head))
|
|
77
|
+
try:
|
|
78
|
+
header = next(reader)
|
|
79
|
+
except StopIteration:
|
|
80
|
+
return None
|
|
81
|
+
df_idx: int | None = None
|
|
82
|
+
for col in (_DATAFLOW_COL, _STRUCTURE_ID_COL):
|
|
83
|
+
if col in header:
|
|
84
|
+
df_idx = header.index(col)
|
|
85
|
+
break
|
|
86
|
+
if df_idx is None:
|
|
87
|
+
return None
|
|
88
|
+
try:
|
|
89
|
+
first_row = next(reader)
|
|
90
|
+
except StopIteration:
|
|
91
|
+
return None
|
|
92
|
+
return first_row[df_idx] if df_idx < len(first_row) else None
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
def _peek_structure(source: "str | Path | bytes") -> "tuple[str | None, str | None]":
|
|
96
|
+
"""``(STRUCTURE, STRUCTURE_ID)`` from the first data row, without reading on.
|
|
97
|
+
|
|
98
|
+
The kind matters: an SDMX-CSV file may name a dataflow, a data structure or
|
|
99
|
+
a provision agreement, and resolving all three is the difference between
|
|
100
|
+
honouring the column and assuming what it says (#409). SDMX-CSV 1.0 has no
|
|
101
|
+
``STRUCTURE`` column at all — its ``DATAFLOW`` implies the kind — so the
|
|
102
|
+
first element is None there and the caller treats it as a dataflow.
|
|
103
|
+
"""
|
|
104
|
+
if isinstance(source, (str, Path)):
|
|
105
|
+
with Path(source).open(newline="", encoding="utf-8") as f:
|
|
106
|
+
head = f.read(4096)
|
|
107
|
+
else:
|
|
108
|
+
head = source[:4096].decode("utf-8", errors="replace")
|
|
109
|
+
reader = csv.reader(io.StringIO(head))
|
|
110
|
+
try:
|
|
111
|
+
header = next(reader)
|
|
112
|
+
first_row = next(reader)
|
|
113
|
+
except StopIteration:
|
|
114
|
+
return None, None
|
|
115
|
+
|
|
116
|
+
def value(column: str) -> str | None:
|
|
117
|
+
if column not in header:
|
|
118
|
+
return None
|
|
119
|
+
index = header.index(column)
|
|
120
|
+
return first_row[index] if index < len(first_row) else None
|
|
121
|
+
|
|
122
|
+
kind = value(_STRUCTURE_COL)
|
|
123
|
+
identifier = value(_STRUCTURE_ID_COL) or value(_DATAFLOW_COL)
|
|
124
|
+
return kind, identifier
|
|
125
|
+
|
|
126
|
+
|
|
127
|
+
def _dataflow_ref(dataflow_str: str) -> "Ref[Dataflow]":
|
|
128
|
+
"""Build an unresolved Ref[Dataflow] from an SDMX-CSV DATAFLOW column value."""
|
|
129
|
+
agency, id_, version = parse_dataflow_id(dataflow_str)
|
|
130
|
+
urn = SdmxUrn.make(
|
|
131
|
+
Dataflow,
|
|
132
|
+
package="datastructure",
|
|
133
|
+
artefact_class="Dataflow",
|
|
134
|
+
agency=agency,
|
|
135
|
+
id=id_,
|
|
136
|
+
version=version,
|
|
137
|
+
)
|
|
138
|
+
return Ref[Dataflow].unresolved(urn)
|
|
139
|
+
|
|
140
|
+
|
|
141
|
+
def _resolve_structure(
|
|
142
|
+
source: "str | Path | bytes",
|
|
143
|
+
registry: "SupportsGet",
|
|
144
|
+
) -> "tuple[DataStructure, Ref[Dataflow] | None]":
|
|
145
|
+
"""The `DataStructure` this file names, looked up in *registry*.
|
|
146
|
+
|
|
147
|
+
Every SDMX-CSV 2.0 row names its own structure, and this reader has always
|
|
148
|
+
parsed that column and then thrown it away — `scan_csv` required the DSD as
|
|
149
|
+
a positional argument, so the file's answer was only available after you had
|
|
150
|
+
supplied it. #409 is that asymmetry.
|
|
151
|
+
|
|
152
|
+
All three kinds are honoured rather than assuming `dataflow`, which is what
|
|
153
|
+
"only `dataflow` appears to be considered" described. A provision agreement
|
|
154
|
+
resolves through its dataflow, which is the same second hop a dataflow makes
|
|
155
|
+
to its structure.
|
|
156
|
+
|
|
157
|
+
Raises:
|
|
158
|
+
ValueError: If the file names no structure (no ``STRUCTURE_ID`` and no
|
|
159
|
+
``DATAFLOW`` column), if its ``STRUCTURE`` value is not one of
|
|
160
|
+
`STRUCTURE_KINDS`, if the registry cannot resolve what it names, or
|
|
161
|
+
if what it resolves to has no structure to read. Four distinct
|
|
162
|
+
messages, because a caller's next move differs for each.
|
|
163
|
+
TypeError: If the URN resolves to something that is not a dataflow,
|
|
164
|
+
provision agreement or data structure at all.
|
|
165
|
+
"""
|
|
166
|
+
kind, identifier = _peek_structure(source)
|
|
167
|
+
if identifier is None:
|
|
168
|
+
msg = (
|
|
169
|
+
"this file names no structure: no STRUCTURE_ID column (SDMX-CSV 2.0) "
|
|
170
|
+
"and no DATAFLOW column (1.0). Pass structure=<DataStructure> instead"
|
|
171
|
+
)
|
|
172
|
+
raise ValueError(msg)
|
|
173
|
+
if kind is not None and kind not in STRUCTURE_KINDS:
|
|
174
|
+
msg = f"unknown STRUCTURE value {kind!r} — SDMX-CSV allows {', '.join(STRUCTURE_KINDS)}"
|
|
175
|
+
raise ValueError(msg)
|
|
176
|
+
|
|
177
|
+
agency, id_, version = parse_dataflow_id(identifier)
|
|
178
|
+
artefact_class = {"datastructure": DataStructure, "dataprovision": ProvisionAgreement}.get(
|
|
179
|
+
kind or "dataflow", Dataflow
|
|
180
|
+
)
|
|
181
|
+
urn = artefact_class.urn_for(agency=agency, id=id_, version=version)
|
|
182
|
+
found: object = registry.get(urn)
|
|
183
|
+
if found is None:
|
|
184
|
+
msg = f"{kind or 'dataflow'} {identifier} is not in this registry — add it, or pass structure=<DataStructure>"
|
|
185
|
+
raise ValueError(msg)
|
|
186
|
+
|
|
187
|
+
flow_ref: Ref[Dataflow] | None = None
|
|
188
|
+
if isinstance(found, ProvisionAgreement):
|
|
189
|
+
if found.dataflow is None:
|
|
190
|
+
msg = f"provision agreement {identifier} has no dataflow, so its structure cannot be reached"
|
|
191
|
+
raise ValueError(msg)
|
|
192
|
+
flow_ref = found.dataflow
|
|
193
|
+
found = found.dataflow()
|
|
194
|
+
if isinstance(found, Dataflow):
|
|
195
|
+
flow_ref = flow_ref or _dataflow_ref(identifier)
|
|
196
|
+
if found.structure is None:
|
|
197
|
+
msg = f"dataflow {identifier} has no structure — it cannot say how to read this file"
|
|
198
|
+
raise ValueError(msg)
|
|
199
|
+
found = found.structure()
|
|
200
|
+
if not isinstance(found, DataStructure):
|
|
201
|
+
# A `TypeError`, per ruff's TRY004 and correctly: the registry answered
|
|
202
|
+
# with the wrong *kind* of thing, which is a different fault from the
|
|
203
|
+
# file naming something absent.
|
|
204
|
+
msg = f"{identifier} resolved to {type(found).__name__}, which is not a DataStructure"
|
|
205
|
+
raise TypeError(msg)
|
|
206
|
+
return found, flow_ref
|
|
207
|
+
|
|
208
|
+
|
|
209
|
+
def scan_csv(
|
|
210
|
+
source: "str | Path | bytes",
|
|
211
|
+
structure: "DataStructure | Dataflow | None" = None,
|
|
212
|
+
*,
|
|
213
|
+
registry: "SupportsGet | None" = None,
|
|
214
|
+
infer_schema_length: "int | None" = 0,
|
|
215
|
+
dataflow: "Ref[Dataflow] | None" = None,
|
|
216
|
+
coded_as: "CodedAs" = "enum",
|
|
217
|
+
) -> Dataset:
|
|
218
|
+
"""Parse an SDMX-CSV message into a `Dataset`.
|
|
219
|
+
|
|
220
|
+
File paths are scanned lazily via ``pl.scan_csv``; bytes (e.g. from an
|
|
221
|
+
HTTP response) are read eagerly and wrapped in a ``LazyFrame``. Either
|
|
222
|
+
way the returned ``Dataset.data`` is always a ``pl.LazyFrame``.
|
|
223
|
+
|
|
224
|
+
Args:
|
|
225
|
+
source: File path or raw bytes from an SDMX-CSV response.
|
|
226
|
+
structure: Fully resolved `DataStructure`, or the `Dataflow` that names
|
|
227
|
+
one — the flow is unwrapped for you, as sdmx1's reader does.
|
|
228
|
+
Codelists should be resolved for ``pl.Enum`` typing.
|
|
229
|
+
Omit it and pass ``registry`` to use the file's own
|
|
230
|
+
``STRUCTURE_ID`` instead (#409).
|
|
231
|
+
registry: Any `Registry` to resolve the file's ``STRUCTURE_ID``
|
|
232
|
+
against, when ``structure`` is omitted. All three
|
|
233
|
+
``STRUCTURE`` kinds are honoured — ``dataflow``,
|
|
234
|
+
``datastructure`` and ``dataprovision``.
|
|
235
|
+
infer_schema_length: Passed to Polars. ``0`` (default) means
|
|
236
|
+
schema is fully DSD-driven with no row sampling,
|
|
237
|
+
which is required for true streaming on paths.
|
|
238
|
+
dataflow: Optional resolved `Dataflow` ref.
|
|
239
|
+
When ``None`` and a ``DATAFLOW`` column is present, an
|
|
240
|
+
unresolved ref is built from the column value.
|
|
241
|
+
coded_as: How coded columns are typed — see `sdmxlib.polars.schema`.
|
|
242
|
+
``"enum"`` (default) fails fast on a value outside its
|
|
243
|
+
codelist; ``"string"`` lets it through so
|
|
244
|
+
`Dataset.validate` can report every one (#408).
|
|
245
|
+
|
|
246
|
+
Returns:
|
|
247
|
+
A `Dataset` with a lazy ``data`` frame.
|
|
248
|
+
|
|
249
|
+
Example:
|
|
250
|
+
import sdmxlib.polars as slpl
|
|
251
|
+
|
|
252
|
+
ds = slpl.scan_csv("data.csv", dsd)
|
|
253
|
+
df = ds.collect(with_labels=True, lang="en")
|
|
254
|
+
|
|
255
|
+
# The file says which structure it needs; the registry has it:
|
|
256
|
+
ds = slpl.scan_csv("data.csv", registry=reg)
|
|
257
|
+
|
|
258
|
+
# Or hand over the flow and skip the `.structure()` unwrap:
|
|
259
|
+
ds = slpl.scan_csv("data.csv", message.dataflows["DF"])
|
|
260
|
+
|
|
261
|
+
# A report rather than a raise, for a file you do not trust yet:
|
|
262
|
+
report = slpl.scan_csv("suspect.csv", dsd, coded_as="string").validate()
|
|
263
|
+
"""
|
|
264
|
+
resolved_ref: Ref[Dataflow] | None = None
|
|
265
|
+
if isinstance(structure, Dataflow):
|
|
266
|
+
# sdmx1's CSV reader takes either, and for the same reason: a caller
|
|
267
|
+
# holding the flow has already done the lookup, and making them add
|
|
268
|
+
# `.structure()` is the hand-written join #409 is about.
|
|
269
|
+
if structure.structure is None:
|
|
270
|
+
msg = f"dataflow {short_urn(structure)} has no structure — it cannot say how to read this file"
|
|
271
|
+
raise ValueError(msg)
|
|
272
|
+
resolved_ref = _dataflow_ref(short_urn(structure))
|
|
273
|
+
structure = structure.structure()
|
|
274
|
+
elif structure is None:
|
|
275
|
+
if registry is None:
|
|
276
|
+
msg = (
|
|
277
|
+
"scan_csv needs structure=<DataStructure | Dataflow>, or "
|
|
278
|
+
"registry=<Registry> to resolve the file's own STRUCTURE_ID"
|
|
279
|
+
)
|
|
280
|
+
raise ValueError(msg)
|
|
281
|
+
structure, resolved_ref = _resolve_structure(source, registry)
|
|
282
|
+
|
|
283
|
+
# Only a `dataflow` file gets a `Dataset.dataflow`. This used to build one
|
|
284
|
+
# from `STRUCTURE_ID` whatever the `STRUCTURE` column said, so a
|
|
285
|
+
# `datastructure` file came back claiming a dataflow whose id was really a
|
|
286
|
+
# DSD's. Reading the kind (#409) is what makes the distinction available.
|
|
287
|
+
kind, identifier = _peek_structure(source)
|
|
288
|
+
dataflow_ref = dataflow or resolved_ref
|
|
289
|
+
if dataflow_ref is None and identifier is not None and kind in (None, "dataflow"):
|
|
290
|
+
dataflow_ref = _dataflow_ref(identifier)
|
|
291
|
+
|
|
292
|
+
schema_overrides = slpl.schema(structure, coded_as=coded_as)
|
|
293
|
+
|
|
294
|
+
if isinstance(source, (str, Path)):
|
|
295
|
+
lf = pl.scan_csv(
|
|
296
|
+
source,
|
|
297
|
+
schema_overrides=schema_overrides,
|
|
298
|
+
infer_schema_length=infer_schema_length,
|
|
299
|
+
)
|
|
300
|
+
else:
|
|
301
|
+
lf = pl.read_csv(
|
|
302
|
+
io.BytesIO(source),
|
|
303
|
+
schema_overrides=schema_overrides,
|
|
304
|
+
infer_schema_length=infer_schema_length,
|
|
305
|
+
).lazy()
|
|
306
|
+
|
|
307
|
+
# Drop metadata columns that are not DSD components. Covers both
|
|
308
|
+
# SDMX-CSV 1.0 (DATAFLOW, KEY) and 2.0 (STRUCTURE, STRUCTURE_ID, ACTION).
|
|
309
|
+
col_names = lf.collect_schema().names()
|
|
310
|
+
envelope_cols = (
|
|
311
|
+
_DATAFLOW_COL,
|
|
312
|
+
_KEY_COL,
|
|
313
|
+
_STRUCTURE_COL,
|
|
314
|
+
_STRUCTURE_ID_COL,
|
|
315
|
+
_ACTION_COL,
|
|
316
|
+
)
|
|
317
|
+
to_drop = [c for c in envelope_cols if c in col_names]
|
|
318
|
+
if to_drop:
|
|
319
|
+
lf = lf.drop(to_drop)
|
|
320
|
+
|
|
321
|
+
return Dataset(structure=structure, data=lf, dataflow=dataflow_ref)
|
|
@@ -47,6 +47,21 @@ from typing import TYPE_CHECKING, cast
|
|
|
47
47
|
|
|
48
48
|
import polars as pl
|
|
49
49
|
|
|
50
|
+
from sdmxlib.formats.sdmx_csv.reader import STRUCTURE_KINDS
|
|
51
|
+
from sdmxlib.model.dataflow import Dataflow
|
|
52
|
+
from sdmxlib.model.datastructure import DataStructure
|
|
53
|
+
from sdmxlib.model.provision import ProvisionAgreement
|
|
54
|
+
from sdmxlib.model.ref import short_urn
|
|
55
|
+
|
|
56
|
+
#: Artefact class -> the ``STRUCTURE`` value SDMX-CSV gives it, for the form of
|
|
57
|
+
#: `write_sdmx_csv` that takes the artefact instead of a pre-formatted string.
|
|
58
|
+
_KIND_OF: dict[type, str] = {
|
|
59
|
+
Dataflow: "dataflow",
|
|
60
|
+
DataStructure: "datastructure",
|
|
61
|
+
ProvisionAgreement: "dataprovision",
|
|
62
|
+
}
|
|
63
|
+
|
|
64
|
+
|
|
50
65
|
if TYPE_CHECKING:
|
|
51
66
|
from collections.abc import Iterable, Iterator
|
|
52
67
|
|
|
@@ -59,7 +74,7 @@ def write_sdmx_csv(
|
|
|
59
74
|
batches: Iterable[pa.RecordBatch],
|
|
60
75
|
dsd: DataStructure,
|
|
61
76
|
*,
|
|
62
|
-
structure_id: str,
|
|
77
|
+
structure_id: str | Dataflow | DataStructure | ProvisionAgreement,
|
|
63
78
|
structure: str = "dataflow",
|
|
64
79
|
action: str = "I",
|
|
65
80
|
) -> Iterator[bytes]:
|
|
@@ -73,11 +88,18 @@ def write_sdmx_csv(
|
|
|
73
88
|
Determines column order in the output: dimensions, then
|
|
74
89
|
measures, then attributes, each in the order the structure
|
|
75
90
|
declares them.
|
|
76
|
-
structure_id: The ``AGENCY:ID(VERSION)`` identifier emitted in
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
|
|
80
|
-
|
|
91
|
+
structure_id: The ``AGENCY:ID(VERSION)`` identifier emitted in the
|
|
92
|
+
``STRUCTURE_ID`` column on every row — or the artefact itself, in
|
|
93
|
+
which case both this and ``structure`` are read off it. Passing the
|
|
94
|
+
artefact is the point of #409: every caller was formatting the short
|
|
95
|
+
URN by hand while `parse_dataflow_id` owned reading it.
|
|
96
|
+
structure: Value emitted in the ``STRUCTURE`` column. SDMX-CSV allows
|
|
97
|
+
``"dataflow"``, ``"datastructure"`` or ``"dataprovision"``, and the
|
|
98
|
+
field guide's wording is exactly those three. This said
|
|
99
|
+
``"provisionagreement"`` until #409 — that is SDMX-JSON's RFC 5988
|
|
100
|
+
link relation, a different vocabulary, and a caller who followed it
|
|
101
|
+
wrote a file no conformant reader accepts. Ignored when
|
|
102
|
+
``structure_id`` is an artefact.
|
|
81
103
|
action: Value emitted in the ``ACTION`` column. One of ``I``
|
|
82
104
|
(Insert), ``R`` (Replace), ``A`` (Append), ``D`` (Delete).
|
|
83
105
|
Defaults to ``"I"``.
|
|
@@ -89,7 +111,14 @@ def write_sdmx_csv(
|
|
|
89
111
|
|
|
90
112
|
Raises:
|
|
91
113
|
KeyError: If a batch is missing a component the structure declares.
|
|
114
|
+
ValueError: If ``structure`` is not one of SDMX-CSV's three kinds.
|
|
92
115
|
"""
|
|
116
|
+
if not isinstance(structure_id, str):
|
|
117
|
+
structure = _KIND_OF.get(type(structure_id), "dataflow")
|
|
118
|
+
structure_id = short_urn(structure_id)
|
|
119
|
+
if structure not in STRUCTURE_KINDS:
|
|
120
|
+
msg = f"STRUCTURE must be one of {', '.join(STRUCTURE_KINDS)}, not {structure!r}"
|
|
121
|
+
raise ValueError(msg)
|
|
93
122
|
# Component ids, in SDMX order, from the structure. This read the
|
|
94
123
|
# *binding's* physical column names until an observation query began
|
|
95
124
|
# projecting by component id — after which it raised a `KeyError` naming
|
|
@@ -37,7 +37,7 @@ import time
|
|
|
37
37
|
from collections.abc import Generator, Iterable
|
|
38
38
|
from contextlib import contextmanager
|
|
39
39
|
from pathlib import Path
|
|
40
|
-
from typing import TYPE_CHECKING,
|
|
40
|
+
from typing import TYPE_CHECKING, Final, Literal, Self, cast, overload, override
|
|
41
41
|
|
|
42
42
|
import duckdb
|
|
43
43
|
|
|
@@ -921,7 +921,7 @@ class LocalRegistry(SessionFactory):
|
|
|
921
921
|
# phantom happens to be. See the docstring for why it exists.
|
|
922
922
|
@overload
|
|
923
923
|
def get[T](
|
|
924
|
-
self, target: "SdmxUrn[object] | Ref[
|
|
924
|
+
self, target: "SdmxUrn[object] | Ref[object]", /, *, expect: "type[T]", eager: bool = False
|
|
925
925
|
) -> "T | None": ...
|
|
926
926
|
|
|
927
927
|
@overload
|
|
@@ -938,7 +938,7 @@ class LocalRegistry(SessionFactory):
|
|
|
938
938
|
def get[T](self, target: "SdmxUrn[T] | Ref[T]", /, *, eager: bool = False) -> "T | None": ...
|
|
939
939
|
|
|
940
940
|
def get[T](
|
|
941
|
-
self, target: "SdmxUrn[object] | Ref[
|
|
941
|
+
self, target: "SdmxUrn[object] | Ref[object]", /, *, expect: "type[T] | None" = None, eager: bool = False
|
|
942
942
|
) -> "object":
|
|
943
943
|
"""Look up an artefact by URN, or by a `Ref` naming one.
|
|
944
944
|
|
|
@@ -959,9 +959,12 @@ class LocalRegistry(SessionFactory):
|
|
|
959
959
|
claiming `Codelist` for a `LazyCodelist`, the silent lie #397 removed,
|
|
960
960
|
and the explicit spelling is the common one. **Narrowing
|
|
961
961
|
`SdmxUrn.parse` to `SdmxUrn[object]`** — #399's option 2 — measures at
|
|
962
|
-
52 basedpyright + 27 pyrefly findings, every one the same fact:
|
|
963
|
-
|
|
964
|
-
`Ref
|
|
962
|
+
52 basedpyright + 27 pyrefly findings, every one the same fact: a
|
|
963
|
+
`Ref[object]` does not go where a `Ref[Codelist]` is wanted. `Ref` is
|
|
964
|
+
*covariant* (see `Ref`'s own class comment and #289), so the assignment
|
|
965
|
+
that works is the other direction; narrowing `parse` asks for this one,
|
|
966
|
+
which covariance does not give. They are sites of correct code, not
|
|
967
|
+
defects.
|
|
965
968
|
|
|
966
969
|
Pass a typed URN where you have one; narrow with `isinstance` where you
|
|
967
970
|
do not. `tests/typing/probe_urn_any.py` pins the declared types and
|
|
@@ -95,6 +95,7 @@ from sdmxlib.model.organisation import (
|
|
|
95
95
|
MetadataProviderScheme,
|
|
96
96
|
)
|
|
97
97
|
from sdmxlib.model.provision import ProvisionAgreement
|
|
98
|
+
from sdmxlib.model.ref import short_urn as short_urn
|
|
98
99
|
from sdmxlib.model.ref import Ref, SdmxRef, UnresolvedRef
|
|
99
100
|
from sdmxlib.model.registry import InMemoryRegistry, RegistryReader, artefact_urn
|
|
100
101
|
from sdmxlib.model.representation import DataType, Facet, Representation
|
|
@@ -38,12 +38,12 @@ import attrs
|
|
|
38
38
|
from sdmxlib.model._refshape import item_fields
|
|
39
39
|
from sdmxlib.model.base import MaintainableArtefact
|
|
40
40
|
from sdmxlib.model.collections import ArtefactList, ItemList, SdmxItem
|
|
41
|
+
from sdmxlib.model.ref import Ref, SupportsUrn
|
|
42
|
+
from sdmxlib.model.urn import SdmxUrn
|
|
41
43
|
|
|
42
44
|
if TYPE_CHECKING:
|
|
43
45
|
from collections.abc import Iterator, Mapping
|
|
44
46
|
|
|
45
|
-
from sdmxlib.model.urn import SdmxUrn
|
|
46
|
-
|
|
47
47
|
#: The component classes SDMX spells differently across versions. A URN carries
|
|
48
48
|
#: the spelling of whatever produced it — 2.1 names an attribute ``Attribute``
|
|
49
49
|
#: and the single measure ``PrimaryMeasure``, 3.0 names them ``DataAttribute``
|
|
@@ -297,3 +297,93 @@ def _search(owner: object, owner_cls: type, item_id: str, wanted: tuple[str, ...
|
|
|
297
297
|
if found is not None:
|
|
298
298
|
return found
|
|
299
299
|
return None
|
|
300
|
+
|
|
301
|
+
|
|
302
|
+
def item_urn_of[T: SdmxItem](scheme: SupportsUrn, item_cls: type[T], item_id: str) -> SdmxUrn[T]:
|
|
303
|
+
"""The URN of *item_id* inside *scheme*, as SDMX spells it.
|
|
304
|
+
|
|
305
|
+
The inverse of `scheme_urn_of`, and the constructive half of what #404 and
|
|
306
|
+
#405 made the library able to *read*. Agency, id and version come off the
|
|
307
|
+
scheme rather than from the caller, which is the whole point: a URN built by
|
|
308
|
+
hand can name a parent that does not exist — six keyword arguments, three of
|
|
309
|
+
them restating the scheme — and `validate_references` will now correctly
|
|
310
|
+
report that as broken (#407).
|
|
311
|
+
"""
|
|
312
|
+
return SdmxUrn.make(
|
|
313
|
+
item_cls,
|
|
314
|
+
package=item_cls.sdmx_package,
|
|
315
|
+
artefact_class=item_cls.sdmx_class,
|
|
316
|
+
agency=scheme.agency_id,
|
|
317
|
+
id=scheme.id,
|
|
318
|
+
version=scheme.version,
|
|
319
|
+
item_id=item_id,
|
|
320
|
+
)
|
|
321
|
+
|
|
322
|
+
|
|
323
|
+
def locate_item(scheme: object) -> Iterator[tuple[str, type[SdmxItem]]]:
|
|
324
|
+
"""The (field name, item class) pairs `ref_to` searches, in order."""
|
|
325
|
+
yield from item_fields(type(scheme))
|
|
326
|
+
|
|
327
|
+
|
|
328
|
+
def item_ref[T: SdmxItem](scheme: SupportsUrn, item_id: str, item_cls: type[T]) -> Ref[T]:
|
|
329
|
+
"""A resolved `Ref` to one item of *scheme*, found by id.
|
|
330
|
+
|
|
331
|
+
Raises:
|
|
332
|
+
KeyError: If no item of that class in *scheme* has this id. The message
|
|
333
|
+
names the scheme, so a caller who passed the right id to the wrong
|
|
334
|
+
artefact is told which artefact answered.
|
|
335
|
+
"""
|
|
336
|
+
found = _search(scheme, type(scheme), item_id, spellings(item_cls.sdmx_class))
|
|
337
|
+
if found is None and isinstance(scheme, SupportsItemLookup):
|
|
338
|
+
# A lazy facade keeps its items behind a query, so the declared-field
|
|
339
|
+
# walk finds nothing; its subscript is the targeted lookup. Same order
|
|
340
|
+
# as `find_item`, and the reason `LazyCodelist.ref_to` can exist at all
|
|
341
|
+
# rather than being refused until you materialise (#407).
|
|
342
|
+
found = _by_subscript(scheme, item_id, spellings(item_cls.sdmx_class))
|
|
343
|
+
# `isinstance` rather than a cast: it narrows `_search`'s `object | None` to
|
|
344
|
+
# `T` with a runtime check, and covers "not found" in the same branch.
|
|
345
|
+
if not isinstance(found, item_cls):
|
|
346
|
+
msg = (
|
|
347
|
+
f"no {item_cls.sdmx_class} with id {item_id!r} in "
|
|
348
|
+
f"{type(scheme).__name__} {scheme.agency_id}:{scheme.id}({scheme.version})"
|
|
349
|
+
)
|
|
350
|
+
raise KeyError(msg)
|
|
351
|
+
return Ref[T].of(item_urn_of(scheme, item_cls, item_id), found)
|
|
352
|
+
|
|
353
|
+
|
|
354
|
+
def any_item_ref(scheme: SupportsUrn, item_id: str) -> Ref[object]:
|
|
355
|
+
"""A resolved `Ref` to one item of *scheme*, whichever kind holds the id.
|
|
356
|
+
|
|
357
|
+
For a `DataStructure`, whose `dimensions`, `time_dimension`,
|
|
358
|
+
`measure_dimension`, `measures`, `attributes` and `groups` are six item
|
|
359
|
+
lists under one maintainable. Dispatching is unambiguous rather than
|
|
360
|
+
convenient: `storage.schema` declares
|
|
361
|
+
``PRIMARY KEY (dsd_urn, component_id)`` on `dsd_component`, so this library
|
|
362
|
+
already treats a component id as unique across the whole DSD, whichever
|
|
363
|
+
list it sits in (#407).
|
|
364
|
+
|
|
365
|
+
`Ref[object]` because the class varies and every field that consumes one of
|
|
366
|
+
these — `HierarchyAssociation.linked_object`, `Categorisation.target` — is
|
|
367
|
+
declared `Ref[object]`. The URN carries the real class, so nothing is lost
|
|
368
|
+
at runtime; a caller who needs it statically can `isinstance` the unwrapped
|
|
369
|
+
item.
|
|
370
|
+
|
|
371
|
+
Raises:
|
|
372
|
+
KeyError: If no item of *scheme* has this id, naming the fields
|
|
373
|
+
searched so the message says where it looked.
|
|
374
|
+
"""
|
|
375
|
+
for name, item_cls in locate_item(scheme):
|
|
376
|
+
value = getattr(scheme, name, None)
|
|
377
|
+
found = _by_id(value, item_id)
|
|
378
|
+
if found is None:
|
|
379
|
+
found = _search(value, item_cls, item_id, spellings(item_cls.sdmx_class))
|
|
380
|
+
# `isinstance` rather than `is not None`: it narrows to the item class,
|
|
381
|
+
# which is what `Ref.of` needs to type the pair it is handed.
|
|
382
|
+
if isinstance(found, item_cls):
|
|
383
|
+
return Ref[object].of(item_urn_of(scheme, item_cls, item_id), found)
|
|
384
|
+
searched = ", ".join(name for name, _ in locate_item(scheme)) or "no item fields"
|
|
385
|
+
msg = (
|
|
386
|
+
f"no item with id {item_id!r} in {type(scheme).__name__} "
|
|
387
|
+
f"{scheme.agency_id}:{scheme.id}({scheme.version}) — searched {searched}"
|
|
388
|
+
)
|
|
389
|
+
raise KeyError(msg)
|
|
@@ -5,6 +5,7 @@ from typing import ClassVar
|
|
|
5
5
|
|
|
6
6
|
from attrs import define, evolve, field
|
|
7
7
|
|
|
8
|
+
from sdmxlib.model._items import item_ref
|
|
8
9
|
from sdmxlib.model._refshape import derived_view
|
|
9
10
|
from sdmxlib.model.annotations import Annotations
|
|
10
11
|
from sdmxlib.model.base import MaintainableArtefact
|
|
@@ -73,6 +74,22 @@ class CategoryScheme(MaintainableArtefact):
|
|
|
73
74
|
_flat: ItemList[Category] = field(init=False, repr=False, factory=ItemList, metadata=derived_view())
|
|
74
75
|
_parent_of: dict[str, str | None] = field(init=False, repr=False, eq=False, factory=dict)
|
|
75
76
|
|
|
77
|
+
def ref_to(self, category_id: str) -> "Ref[Category]":
|
|
78
|
+
"""Build a resolved `Ref` to a category within this scheme.
|
|
79
|
+
|
|
80
|
+
The URN is built from *this* scheme's agency, id and version, so it
|
|
81
|
+
cannot name a parent that does not exist — which is what six hand-written
|
|
82
|
+
keyword arguments to `SdmxUrn.make` invite, and what
|
|
83
|
+
`validate_references` now correctly reports as broken (#407).
|
|
84
|
+
|
|
85
|
+
Raises:
|
|
86
|
+
KeyError: If no category with this id is in this scheme.
|
|
87
|
+
|
|
88
|
+
Example:
|
|
89
|
+
Categorisation(..., target=cs.ref_to("ECON"))
|
|
90
|
+
"""
|
|
91
|
+
return item_ref(self, category_id, Category)
|
|
92
|
+
|
|
76
93
|
def __attrs_post_init__(self) -> None:
|
|
77
94
|
self._flat = _flatten_categories(self.categories)
|
|
78
95
|
parent: dict[str, str | None] = {n.id: None for n in self._flat}
|