sdmxlib 0.62.2__tar.gz → 0.63.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (135) hide show
  1. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/PKG-INFO +1 -1
  2. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/pyproject.toml +1 -1
  3. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/pyproject.toml.orig +1 -1
  4. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/__init__.py +1 -0
  5. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/api/federated.py +3 -3
  6. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/api/registry.py +3 -3
  7. sdmxlib-0.63.0/src/sdmxlib/formats/sdmx_csv/reader.py +321 -0
  8. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/formats/sdmx_csv/writer.py +35 -6
  9. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/local/registry.py +9 -6
  10. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/model/__init__.py +1 -0
  11. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/model/_items.py +92 -2
  12. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/model/category.py +17 -0
  13. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/model/codelist.py +17 -0
  14. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/model/concept.py +8 -11
  15. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/model/dataset.py +23 -0
  16. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/model/datastructure.py +25 -0
  17. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/model/in_memory_registry.py +2 -2
  18. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/model/metadatastructure.py +17 -0
  19. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/model/organisation.py +50 -0
  20. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/model/ref.py +19 -0
  21. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/model/urn.py +5 -3
  22. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/model/validation.py +36 -0
  23. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/polars.py +65 -9
  24. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/storage/lazy.py +23 -4
  25. sdmxlib-0.62.2/src/sdmxlib/formats/sdmx_csv/reader.py +0 -162
  26. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/README.md +0 -0
  27. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/_duckdb.py +0 -0
  28. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/_singleflight.py +0 -0
  29. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/api/__init__.py +0 -0
  30. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/api/_freshness.py +0 -0
  31. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/api/client.py +0 -0
  32. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/api/filters.py +0 -0
  33. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/api/policy.py +0 -0
  34. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/api/providers.py +0 -0
  35. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/api/query.py +0 -0
  36. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/api/session.py +0 -0
  37. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/builders/__init__.py +0 -0
  38. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/builders/contentconstraint.py +0 -0
  39. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/builders/metadata_annotation.py +0 -0
  40. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/data_store.py +0 -0
  41. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/formats/__init__.py +0 -0
  42. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/formats/_coerce.py +0 -0
  43. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/formats/_index.py +0 -0
  44. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/formats/data_format.py +0 -0
  45. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/formats/gaps.py +0 -0
  46. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/formats/sdmx_csv/__init__.py +0 -0
  47. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/formats/sdmx_json/__init__.py +0 -0
  48. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/formats/sdmx_json/_data_common.py +0 -0
  49. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/formats/sdmx_json/_dto.py +0 -0
  50. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/formats/sdmx_json/_resolve.py +0 -0
  51. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/formats/sdmx_json/_structure_dto.py +0 -0
  52. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/formats/sdmx_json/_structure_ingest.py +0 -0
  53. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/formats/sdmx_json/data_v1.py +0 -0
  54. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/formats/sdmx_json/data_v2.py +0 -0
  55. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/formats/sdmx_json/metadata.py +0 -0
  56. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/formats/sdmx_json/reader.py +0 -0
  57. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/formats/sdmx_json/structure_values.py +0 -0
  58. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/formats/sdmx_json/writer.py +0 -0
  59. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/formats/sdmx_json_1_0/__init__.py +0 -0
  60. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/formats/sdmx_json_1_0/writer.py +0 -0
  61. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/formats/sdmx_ml/__init__.py +0 -0
  62. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/formats/sdmx_ml/_common.py +0 -0
  63. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/formats/sdmx_ml/_v3_reader.py +0 -0
  64. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/formats/sdmx_ml/_v3_writer.py +0 -0
  65. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/formats/sdmx_ml21/__init__.py +0 -0
  66. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/formats/sdmx_ml21/_synthetic.py +0 -0
  67. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/formats/sdmx_ml21/namespaces.py +0 -0
  68. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/formats/sdmx_ml21/reader.py +0 -0
  69. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/formats/sdmx_ml21/writer.py +0 -0
  70. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/formats/sdmx_ml30/__init__.py +0 -0
  71. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/formats/sdmx_ml30/metadata.py +0 -0
  72. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/formats/sdmx_ml30/namespaces.py +0 -0
  73. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/formats/sdmx_ml30/reader.py +0 -0
  74. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/formats/sdmx_ml30/writer.py +0 -0
  75. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/formats/sdmx_ml31/__init__.py +0 -0
  76. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/formats/sdmx_ml31/namespaces.py +0 -0
  77. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/formats/sdmx_ml31/reader.py +0 -0
  78. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/formats/sdmx_ml31/writer.py +0 -0
  79. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/local/__init__.py +0 -0
  80. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/local/data_store.py +0 -0
  81. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/local/sql.py +0 -0
  82. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/model/_interning.py +0 -0
  83. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/model/_label_match.py +0 -0
  84. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/model/_refshape.py +0 -0
  85. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/model/annotations.py +0 -0
  86. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/model/base.py +0 -0
  87. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/model/binding.py +0 -0
  88. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/model/collections.py +0 -0
  89. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/model/constraint.py +0 -0
  90. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/model/convert.py +0 -0
  91. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/model/dataflow.py +0 -0
  92. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/model/errors.py +0 -0
  93. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/model/expr.py +0 -0
  94. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/model/hierarchy.py +0 -0
  95. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/model/istring.py +0 -0
  96. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/model/mapping.py +0 -0
  97. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/model/message.py +0 -0
  98. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/model/metadata_provision.py +0 -0
  99. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/model/metadataflow.py +0 -0
  100. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/model/metadataset.py +0 -0
  101. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/model/provision.py +0 -0
  102. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/model/registry.py +0 -0
  103. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/model/representation.py +0 -0
  104. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/model/rest_api.py +0 -0
  105. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/model/rollup.py +0 -0
  106. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/model/version.py +0 -0
  107. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/py.typed +0 -0
  108. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/query/__init__.py +0 -0
  109. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/query/_closure.py +0 -0
  110. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/query/_common.py +0 -0
  111. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/query/_compile.py +0 -0
  112. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/query/_edges.py +0 -0
  113. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/query/_execute.py +0 -0
  114. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/query/_expr.py +0 -0
  115. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/query/_joins.py +0 -0
  116. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/query/_plan.py +0 -0
  117. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/query/_pushdown.py +0 -0
  118. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/query/_rebase.py +0 -0
  119. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/query/_rest.py +0 -0
  120. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/query/accessors.py +0 -0
  121. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/query/observations.py +0 -0
  122. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/query/query.py +0 -0
  123. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/query/raw.py +0 -0
  124. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/query/schema.py +0 -0
  125. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/rest.py +0 -0
  126. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/session.py +0 -0
  127. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/sql.py +0 -0
  128. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/storage/__init__.py +0 -0
  129. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/storage/_kinds.py +0 -0
  130. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/storage/_projections.py +0 -0
  131. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/storage/_time.py +0 -0
  132. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/storage/readers.py +0 -0
  133. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/storage/resolve.py +0 -0
  134. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/storage/schema.py +0 -0
  135. {sdmxlib-0.62.2 → sdmxlib-0.63.0}/src/sdmxlib/storage/writers.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.3
2
2
  Name: sdmxlib
3
- Version: 0.62.2
3
+ Version: 0.63.0
4
4
  Summary: SDMX structural metadata library for Python
5
5
  Keywords: sdmx,statistics,metadata,datastructure
6
6
  Author: gabrielgellner
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "sdmxlib"
3
- version = "0.62.2"
3
+ version = "0.63.0"
4
4
  description = "SDMX structural metadata library for Python"
5
5
  readme = "README.md"
6
6
  requires-python = ">=3.13"
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "sdmxlib"
3
- version = "0.62.2"
3
+ version = "0.63.0"
4
4
  description = "SDMX structural metadata library for Python"
5
5
  readme = "README.md"
6
6
  license = { text = "Apache-2.0" }
@@ -11,6 +11,7 @@ from sdmxlib.api.client import ResponseTooLargeError
11
11
  from sdmxlib.api.registry import BearerToken, FmrRegistry, RestRegistry
12
12
  from sdmxlib.data_store import DataStore, WriteSource
13
13
  from sdmxlib.model.datastructure import ComponentRole
14
+ from sdmxlib.model.ref import short_urn as short_urn
14
15
  from sdmxlib.model.ref import SupportsUrn
15
16
  from sdmxlib.model.rest_api import ApiVersion
16
17
  from sdmxlib.model.rollup import AncestorMap
@@ -32,7 +32,7 @@ Example:
32
32
 
33
33
  import functools
34
34
  import inspect
35
- from typing import TYPE_CHECKING, Any, overload
35
+ from typing import TYPE_CHECKING, overload
36
36
 
37
37
  import httpx
38
38
 
@@ -124,7 +124,7 @@ class FederatedRegistry:
124
124
  @overload
125
125
  def get[T](
126
126
  self,
127
- target: "SdmxUrn[object] | Ref[Any]",
127
+ target: "SdmxUrn[object] | Ref[object]",
128
128
  /,
129
129
  *,
130
130
  expect: "type[T]",
@@ -144,7 +144,7 @@ class FederatedRegistry:
144
144
 
145
145
  def get[T](
146
146
  self,
147
- target: "SdmxUrn[object] | Ref[Any]",
147
+ target: "SdmxUrn[object] | Ref[object]",
148
148
  /,
149
149
  *,
150
150
  expect: "type[T] | None" = None,
@@ -4,7 +4,7 @@ import tempfile
4
4
  import threading
5
5
  from collections.abc import Callable, Sequence
6
6
  from pathlib import Path
7
- from typing import TYPE_CHECKING, Any, Self, Unpack, cast, overload
7
+ from typing import TYPE_CHECKING, Self, Unpack, cast, overload
8
8
 
9
9
  import attrs
10
10
  from attrs import frozen
@@ -779,7 +779,7 @@ class RestRegistry:
779
779
  @overload
780
780
  def get[T](
781
781
  self,
782
- target: "SdmxUrn[object] | Ref[Any]",
782
+ target: "SdmxUrn[object] | Ref[object]",
783
783
  /,
784
784
  *,
785
785
  expect: "type[T]",
@@ -799,7 +799,7 @@ class RestRegistry:
799
799
 
800
800
  def get[T](
801
801
  self,
802
- target: "SdmxUrn[object] | Ref[Any]",
802
+ target: "SdmxUrn[object] | Ref[object]",
803
803
  /,
804
804
  *,
805
805
  expect: "type[T] | None" = None,
@@ -0,0 +1,321 @@
1
+ """SDMX-CSV data reader.
2
+
3
+ Parses SDMX-CSV messages into `Dataset` objects.
4
+
5
+ SDMX-CSV format::
6
+
7
+ DATAFLOW,FREQ,REF_AREA,INDICATOR,TIME_PERIOD,OBS_VALUE,OBS_STATUS
8
+ ESTAT:STS_INPR_M(1.0),M,AT,INDPRO,2024-01,98.5,A
9
+
10
+ The ``DATAFLOW`` column identifies the dataflow but is not a DSD component.
11
+ It is extracted and then dropped; the remaining columns are parsed using the
12
+ DSD schema.
13
+ """
14
+
15
+ import csv
16
+ import io
17
+ import re
18
+ from pathlib import Path
19
+ from typing import TYPE_CHECKING, Final
20
+
21
+ import polars as pl
22
+
23
+ import sdmxlib.polars as slpl
24
+ from sdmxlib.model.dataflow import Dataflow
25
+ from sdmxlib.model.dataset import Dataset
26
+ from sdmxlib.model.datastructure import DataStructure
27
+ from sdmxlib.model.provision import ProvisionAgreement
28
+ from sdmxlib.model.ref import Ref, short_urn
29
+ from sdmxlib.model.urn import SdmxUrn
30
+
31
+ if TYPE_CHECKING:
32
+ from sdmxlib.model.registry import SupportsGet
33
+ from sdmxlib.polars import CodedAs
34
+
35
+ _DATAFLOW_COL = "DATAFLOW"
36
+ _KEY_COL = "KEY"
37
+ # SDMX-CSV 2.0.0 envelope columns — not DSD components, dropped on read.
38
+ _STRUCTURE_COL = "STRUCTURE"
39
+ _STRUCTURE_ID_COL = "STRUCTURE_ID"
40
+ _ACTION_COL = "ACTION"
41
+ _DATAFLOW_RE = re.compile(r"^([^:]+):([^(]+)\(([^)]+)\)$")
42
+
43
+
44
+ def parse_dataflow_id(dataflow_id: str) -> tuple[str, str, str]:
45
+ """Parse an SDMX-CSV DATAFLOW value into ``(agency, id, version)``.
46
+
47
+ Example:
48
+ parse_dataflow_id("ESTAT:STS_INPR_M(1.0)") # ("ESTAT", "STS_INPR_M", "1.0")
49
+ """
50
+ m = _DATAFLOW_RE.match(dataflow_id)
51
+ if m is None:
52
+ msg = f"Cannot parse DATAFLOW value {dataflow_id!r} — expected 'AGENCY:ID(VERSION)'"
53
+ raise ValueError(msg)
54
+ return m.group(1), m.group(2), m.group(3)
55
+
56
+
57
+ #: The values SDMX-CSV 2.0's ``STRUCTURE`` column may carry.
58
+ #:
59
+ #: Verbatim from the field guide: "The first column contains: ``dataflow``,
60
+ #: ``datastructure`` or ``dataprovision``". Worth stating because two of our own
61
+ #: docstrings said ``provisionagreement`` — which is the RFC 5988 *link
62
+ #: relation* SDMX-JSON uses, a different vocabulary — and a caller following
63
+ #: them wrote a non-conformant file. sdmx1's CSV reader enumerates the same
64
+ #: three (#409).
65
+ STRUCTURE_KINDS: Final[tuple[str, ...]] = ("dataflow", "datastructure", "dataprovision")
66
+
67
+
68
+ def _peek_dataflow_id(source: "str | Path | bytes") -> str | None:
69
+ """Extract the DATAFLOW value from the first data row without reading the whole file."""
70
+ if isinstance(source, (str, Path)):
71
+ with Path(source).open(newline="", encoding="utf-8") as f:
72
+ head = f.read(4096)
73
+ else:
74
+ head = source[:4096].decode("utf-8", errors="replace")
75
+
76
+ reader = csv.reader(io.StringIO(head))
77
+ try:
78
+ header = next(reader)
79
+ except StopIteration:
80
+ return None
81
+ df_idx: int | None = None
82
+ for col in (_DATAFLOW_COL, _STRUCTURE_ID_COL):
83
+ if col in header:
84
+ df_idx = header.index(col)
85
+ break
86
+ if df_idx is None:
87
+ return None
88
+ try:
89
+ first_row = next(reader)
90
+ except StopIteration:
91
+ return None
92
+ return first_row[df_idx] if df_idx < len(first_row) else None
93
+
94
+
95
+ def _peek_structure(source: "str | Path | bytes") -> "tuple[str | None, str | None]":
96
+ """``(STRUCTURE, STRUCTURE_ID)`` from the first data row, without reading on.
97
+
98
+ The kind matters: an SDMX-CSV file may name a dataflow, a data structure or
99
+ a provision agreement, and resolving all three is the difference between
100
+ honouring the column and assuming what it says (#409). SDMX-CSV 1.0 has no
101
+ ``STRUCTURE`` column at all — its ``DATAFLOW`` implies the kind — so the
102
+ first element is None there and the caller treats it as a dataflow.
103
+ """
104
+ if isinstance(source, (str, Path)):
105
+ with Path(source).open(newline="", encoding="utf-8") as f:
106
+ head = f.read(4096)
107
+ else:
108
+ head = source[:4096].decode("utf-8", errors="replace")
109
+ reader = csv.reader(io.StringIO(head))
110
+ try:
111
+ header = next(reader)
112
+ first_row = next(reader)
113
+ except StopIteration:
114
+ return None, None
115
+
116
+ def value(column: str) -> str | None:
117
+ if column not in header:
118
+ return None
119
+ index = header.index(column)
120
+ return first_row[index] if index < len(first_row) else None
121
+
122
+ kind = value(_STRUCTURE_COL)
123
+ identifier = value(_STRUCTURE_ID_COL) or value(_DATAFLOW_COL)
124
+ return kind, identifier
125
+
126
+
127
+ def _dataflow_ref(dataflow_str: str) -> "Ref[Dataflow]":
128
+ """Build an unresolved Ref[Dataflow] from an SDMX-CSV DATAFLOW column value."""
129
+ agency, id_, version = parse_dataflow_id(dataflow_str)
130
+ urn = SdmxUrn.make(
131
+ Dataflow,
132
+ package="datastructure",
133
+ artefact_class="Dataflow",
134
+ agency=agency,
135
+ id=id_,
136
+ version=version,
137
+ )
138
+ return Ref[Dataflow].unresolved(urn)
139
+
140
+
141
+ def _resolve_structure(
142
+ source: "str | Path | bytes",
143
+ registry: "SupportsGet",
144
+ ) -> "tuple[DataStructure, Ref[Dataflow] | None]":
145
+ """The `DataStructure` this file names, looked up in *registry*.
146
+
147
+ Every SDMX-CSV 2.0 row names its own structure, and this reader has always
148
+ parsed that column and then thrown it away — `scan_csv` required the DSD as
149
+ a positional argument, so the file's answer was only available after you had
150
+ supplied it. #409 is that asymmetry.
151
+
152
+ All three kinds are honoured rather than assuming `dataflow`, which is what
153
+ "only `dataflow` appears to be considered" described. A provision agreement
154
+ resolves through its dataflow, which is the same second hop a dataflow makes
155
+ to its structure.
156
+
157
+ Raises:
158
+ ValueError: If the file names no structure (no ``STRUCTURE_ID`` and no
159
+ ``DATAFLOW`` column), if its ``STRUCTURE`` value is not one of
160
+ `STRUCTURE_KINDS`, if the registry cannot resolve what it names, or
161
+ if what it resolves to has no structure to read. Four distinct
162
+ messages, because a caller's next move differs for each.
163
+ TypeError: If the URN resolves to something that is not a dataflow,
164
+ provision agreement or data structure at all.
165
+ """
166
+ kind, identifier = _peek_structure(source)
167
+ if identifier is None:
168
+ msg = (
169
+ "this file names no structure: no STRUCTURE_ID column (SDMX-CSV 2.0) "
170
+ "and no DATAFLOW column (1.0). Pass structure=<DataStructure> instead"
171
+ )
172
+ raise ValueError(msg)
173
+ if kind is not None and kind not in STRUCTURE_KINDS:
174
+ msg = f"unknown STRUCTURE value {kind!r} — SDMX-CSV allows {', '.join(STRUCTURE_KINDS)}"
175
+ raise ValueError(msg)
176
+
177
+ agency, id_, version = parse_dataflow_id(identifier)
178
+ artefact_class = {"datastructure": DataStructure, "dataprovision": ProvisionAgreement}.get(
179
+ kind or "dataflow", Dataflow
180
+ )
181
+ urn = artefact_class.urn_for(agency=agency, id=id_, version=version)
182
+ found: object = registry.get(urn)
183
+ if found is None:
184
+ msg = f"{kind or 'dataflow'} {identifier} is not in this registry — add it, or pass structure=<DataStructure>"
185
+ raise ValueError(msg)
186
+
187
+ flow_ref: Ref[Dataflow] | None = None
188
+ if isinstance(found, ProvisionAgreement):
189
+ if found.dataflow is None:
190
+ msg = f"provision agreement {identifier} has no dataflow, so its structure cannot be reached"
191
+ raise ValueError(msg)
192
+ flow_ref = found.dataflow
193
+ found = found.dataflow()
194
+ if isinstance(found, Dataflow):
195
+ flow_ref = flow_ref or _dataflow_ref(identifier)
196
+ if found.structure is None:
197
+ msg = f"dataflow {identifier} has no structure — it cannot say how to read this file"
198
+ raise ValueError(msg)
199
+ found = found.structure()
200
+ if not isinstance(found, DataStructure):
201
+ # A `TypeError`, per ruff's TRY004 and correctly: the registry answered
202
+ # with the wrong *kind* of thing, which is a different fault from the
203
+ # file naming something absent.
204
+ msg = f"{identifier} resolved to {type(found).__name__}, which is not a DataStructure"
205
+ raise TypeError(msg)
206
+ return found, flow_ref
207
+
208
+
209
+ def scan_csv(
210
+ source: "str | Path | bytes",
211
+ structure: "DataStructure | Dataflow | None" = None,
212
+ *,
213
+ registry: "SupportsGet | None" = None,
214
+ infer_schema_length: "int | None" = 0,
215
+ dataflow: "Ref[Dataflow] | None" = None,
216
+ coded_as: "CodedAs" = "enum",
217
+ ) -> Dataset:
218
+ """Parse an SDMX-CSV message into a `Dataset`.
219
+
220
+ File paths are scanned lazily via ``pl.scan_csv``; bytes (e.g. from an
221
+ HTTP response) are read eagerly and wrapped in a ``LazyFrame``. Either
222
+ way the returned ``Dataset.data`` is always a ``pl.LazyFrame``.
223
+
224
+ Args:
225
+ source: File path or raw bytes from an SDMX-CSV response.
226
+ structure: Fully resolved `DataStructure`, or the `Dataflow` that names
227
+ one — the flow is unwrapped for you, as sdmx1's reader does.
228
+ Codelists should be resolved for ``pl.Enum`` typing.
229
+ Omit it and pass ``registry`` to use the file's own
230
+ ``STRUCTURE_ID`` instead (#409).
231
+ registry: Any `Registry` to resolve the file's ``STRUCTURE_ID``
232
+ against, when ``structure`` is omitted. All three
233
+ ``STRUCTURE`` kinds are honoured — ``dataflow``,
234
+ ``datastructure`` and ``dataprovision``.
235
+ infer_schema_length: Passed to Polars. ``0`` (default) means
236
+ schema is fully DSD-driven with no row sampling,
237
+ which is required for true streaming on paths.
238
+ dataflow: Optional resolved `Dataflow` ref.
239
+ When ``None`` and a ``DATAFLOW`` column is present, an
240
+ unresolved ref is built from the column value.
241
+ coded_as: How coded columns are typed — see `sdmxlib.polars.schema`.
242
+ ``"enum"`` (default) fails fast on a value outside its
243
+ codelist; ``"string"`` lets it through so
244
+ `Dataset.validate` can report every one (#408).
245
+
246
+ Returns:
247
+ A `Dataset` with a lazy ``data`` frame.
248
+
249
+ Example:
250
+ import sdmxlib.polars as slpl
251
+
252
+ ds = slpl.scan_csv("data.csv", dsd)
253
+ df = ds.collect(with_labels=True, lang="en")
254
+
255
+ # The file says which structure it needs; the registry has it:
256
+ ds = slpl.scan_csv("data.csv", registry=reg)
257
+
258
+ # Or hand over the flow and skip the `.structure()` unwrap:
259
+ ds = slpl.scan_csv("data.csv", message.dataflows["DF"])
260
+
261
+ # A report rather than a raise, for a file you do not trust yet:
262
+ report = slpl.scan_csv("suspect.csv", dsd, coded_as="string").validate()
263
+ """
264
+ resolved_ref: Ref[Dataflow] | None = None
265
+ if isinstance(structure, Dataflow):
266
+ # sdmx1's CSV reader takes either, and for the same reason: a caller
267
+ # holding the flow has already done the lookup, and making them add
268
+ # `.structure()` is the hand-written join #409 is about.
269
+ if structure.structure is None:
270
+ msg = f"dataflow {short_urn(structure)} has no structure — it cannot say how to read this file"
271
+ raise ValueError(msg)
272
+ resolved_ref = _dataflow_ref(short_urn(structure))
273
+ structure = structure.structure()
274
+ elif structure is None:
275
+ if registry is None:
276
+ msg = (
277
+ "scan_csv needs structure=<DataStructure | Dataflow>, or "
278
+ "registry=<Registry> to resolve the file's own STRUCTURE_ID"
279
+ )
280
+ raise ValueError(msg)
281
+ structure, resolved_ref = _resolve_structure(source, registry)
282
+
283
+ # Only a `dataflow` file gets a `Dataset.dataflow`. This used to build one
284
+ # from `STRUCTURE_ID` whatever the `STRUCTURE` column said, so a
285
+ # `datastructure` file came back claiming a dataflow whose id was really a
286
+ # DSD's. Reading the kind (#409) is what makes the distinction available.
287
+ kind, identifier = _peek_structure(source)
288
+ dataflow_ref = dataflow or resolved_ref
289
+ if dataflow_ref is None and identifier is not None and kind in (None, "dataflow"):
290
+ dataflow_ref = _dataflow_ref(identifier)
291
+
292
+ schema_overrides = slpl.schema(structure, coded_as=coded_as)
293
+
294
+ if isinstance(source, (str, Path)):
295
+ lf = pl.scan_csv(
296
+ source,
297
+ schema_overrides=schema_overrides,
298
+ infer_schema_length=infer_schema_length,
299
+ )
300
+ else:
301
+ lf = pl.read_csv(
302
+ io.BytesIO(source),
303
+ schema_overrides=schema_overrides,
304
+ infer_schema_length=infer_schema_length,
305
+ ).lazy()
306
+
307
+ # Drop metadata columns that are not DSD components. Covers both
308
+ # SDMX-CSV 1.0 (DATAFLOW, KEY) and 2.0 (STRUCTURE, STRUCTURE_ID, ACTION).
309
+ col_names = lf.collect_schema().names()
310
+ envelope_cols = (
311
+ _DATAFLOW_COL,
312
+ _KEY_COL,
313
+ _STRUCTURE_COL,
314
+ _STRUCTURE_ID_COL,
315
+ _ACTION_COL,
316
+ )
317
+ to_drop = [c for c in envelope_cols if c in col_names]
318
+ if to_drop:
319
+ lf = lf.drop(to_drop)
320
+
321
+ return Dataset(structure=structure, data=lf, dataflow=dataflow_ref)
@@ -47,6 +47,21 @@ from typing import TYPE_CHECKING, cast
47
47
 
48
48
  import polars as pl
49
49
 
50
+ from sdmxlib.formats.sdmx_csv.reader import STRUCTURE_KINDS
51
+ from sdmxlib.model.dataflow import Dataflow
52
+ from sdmxlib.model.datastructure import DataStructure
53
+ from sdmxlib.model.provision import ProvisionAgreement
54
+ from sdmxlib.model.ref import short_urn
55
+
56
+ #: Artefact class -> the ``STRUCTURE`` value SDMX-CSV gives it, for the form of
57
+ #: `write_sdmx_csv` that takes the artefact instead of a pre-formatted string.
58
+ _KIND_OF: dict[type, str] = {
59
+ Dataflow: "dataflow",
60
+ DataStructure: "datastructure",
61
+ ProvisionAgreement: "dataprovision",
62
+ }
63
+
64
+
50
65
  if TYPE_CHECKING:
51
66
  from collections.abc import Iterable, Iterator
52
67
 
@@ -59,7 +74,7 @@ def write_sdmx_csv(
59
74
  batches: Iterable[pa.RecordBatch],
60
75
  dsd: DataStructure,
61
76
  *,
62
- structure_id: str,
77
+ structure_id: str | Dataflow | DataStructure | ProvisionAgreement,
63
78
  structure: str = "dataflow",
64
79
  action: str = "I",
65
80
  ) -> Iterator[bytes]:
@@ -73,11 +88,18 @@ def write_sdmx_csv(
73
88
  Determines column order in the output: dimensions, then
74
89
  measures, then attributes, each in the order the structure
75
90
  declares them.
76
- structure_id: The ``AGENCY:ID(VERSION)`` identifier emitted in
77
- the ``STRUCTURE_ID`` column on every row.
78
- structure: Value emitted in the ``STRUCTURE`` column. SDMX-CSV
79
- allows ``"dataflow"``, ``"datastructure"``, or
80
- ``"provisionagreement"``. Defaults to ``"dataflow"``.
91
+ structure_id: The ``AGENCY:ID(VERSION)`` identifier emitted in the
92
+ ``STRUCTURE_ID`` column on every row — or the artefact itself, in
93
+ which case both this and ``structure`` are read off it. Passing the
94
+ artefact is the point of #409: every caller was formatting the short
95
+ URN by hand while `parse_dataflow_id` owned reading it.
96
+ structure: Value emitted in the ``STRUCTURE`` column. SDMX-CSV allows
97
+ ``"dataflow"``, ``"datastructure"`` or ``"dataprovision"``, and the
98
+ field guide's wording is exactly those three. This said
99
+ ``"provisionagreement"`` until #409 — that is SDMX-JSON's RFC 5988
100
+ link relation, a different vocabulary, and a caller who followed it
101
+ wrote a file no conformant reader accepts. Ignored when
102
+ ``structure_id`` is an artefact.
81
103
  action: Value emitted in the ``ACTION`` column. One of ``I``
82
104
  (Insert), ``R`` (Replace), ``A`` (Append), ``D`` (Delete).
83
105
  Defaults to ``"I"``.
@@ -89,7 +111,14 @@ def write_sdmx_csv(
89
111
 
90
112
  Raises:
91
113
  KeyError: If a batch is missing a component the structure declares.
114
+ ValueError: If ``structure`` is not one of SDMX-CSV's three kinds.
92
115
  """
116
+ if not isinstance(structure_id, str):
117
+ structure = _KIND_OF.get(type(structure_id), "dataflow")
118
+ structure_id = short_urn(structure_id)
119
+ if structure not in STRUCTURE_KINDS:
120
+ msg = f"STRUCTURE must be one of {', '.join(STRUCTURE_KINDS)}, not {structure!r}"
121
+ raise ValueError(msg)
93
122
  # Component ids, in SDMX order, from the structure. This read the
94
123
  # *binding's* physical column names until an observation query began
95
124
  # projecting by component id — after which it raised a `KeyError` naming
@@ -37,7 +37,7 @@ import time
37
37
  from collections.abc import Generator, Iterable
38
38
  from contextlib import contextmanager
39
39
  from pathlib import Path
40
- from typing import TYPE_CHECKING, Any, Final, Literal, Self, cast, overload, override
40
+ from typing import TYPE_CHECKING, Final, Literal, Self, cast, overload, override
41
41
 
42
42
  import duckdb
43
43
 
@@ -921,7 +921,7 @@ class LocalRegistry(SessionFactory):
921
921
  # phantom happens to be. See the docstring for why it exists.
922
922
  @overload
923
923
  def get[T](
924
- self, target: "SdmxUrn[object] | Ref[Any]", /, *, expect: "type[T]", eager: bool = False
924
+ self, target: "SdmxUrn[object] | Ref[object]", /, *, expect: "type[T]", eager: bool = False
925
925
  ) -> "T | None": ...
926
926
 
927
927
  @overload
@@ -938,7 +938,7 @@ class LocalRegistry(SessionFactory):
938
938
  def get[T](self, target: "SdmxUrn[T] | Ref[T]", /, *, eager: bool = False) -> "T | None": ...
939
939
 
940
940
  def get[T](
941
- self, target: "SdmxUrn[object] | Ref[Any]", /, *, expect: "type[T] | None" = None, eager: bool = False
941
+ self, target: "SdmxUrn[object] | Ref[object]", /, *, expect: "type[T] | None" = None, eager: bool = False
942
942
  ) -> "object":
943
943
  """Look up an artefact by URN, or by a `Ref` naming one.
944
944
 
@@ -959,9 +959,12 @@ class LocalRegistry(SessionFactory):
959
959
  claiming `Codelist` for a `LazyCodelist`, the silent lie #397 removed,
960
960
  and the explicit spelling is the common one. **Narrowing
961
961
  `SdmxUrn.parse` to `SdmxUrn[object]`** — #399's option 2 — measures at
962
- 52 basedpyright + 27 pyrefly findings, every one the same fact: `Ref`
963
- is invariant by design (#289), so `Ref[object]` does not go where
964
- `Ref[Codelist]` is wanted. They are sites of correct code, not defects.
962
+ 52 basedpyright + 27 pyrefly findings, every one the same fact: a
963
+ `Ref[object]` does not go where a `Ref[Codelist]` is wanted. `Ref` is
964
+ *covariant* (see `Ref`'s own class comment and #289), so the assignment
965
+ that works is the other direction; narrowing `parse` asks for this one,
966
+ which covariance does not give. They are sites of correct code, not
967
+ defects.
965
968
 
966
969
  Pass a typed URN where you have one; narrow with `isinstance` where you
967
970
  do not. `tests/typing/probe_urn_any.py` pins the declared types and
@@ -95,6 +95,7 @@ from sdmxlib.model.organisation import (
95
95
  MetadataProviderScheme,
96
96
  )
97
97
  from sdmxlib.model.provision import ProvisionAgreement
98
+ from sdmxlib.model.ref import short_urn as short_urn
98
99
  from sdmxlib.model.ref import Ref, SdmxRef, UnresolvedRef
99
100
  from sdmxlib.model.registry import InMemoryRegistry, RegistryReader, artefact_urn
100
101
  from sdmxlib.model.representation import DataType, Facet, Representation
@@ -38,12 +38,12 @@ import attrs
38
38
  from sdmxlib.model._refshape import item_fields
39
39
  from sdmxlib.model.base import MaintainableArtefact
40
40
  from sdmxlib.model.collections import ArtefactList, ItemList, SdmxItem
41
+ from sdmxlib.model.ref import Ref, SupportsUrn
42
+ from sdmxlib.model.urn import SdmxUrn
41
43
 
42
44
  if TYPE_CHECKING:
43
45
  from collections.abc import Iterator, Mapping
44
46
 
45
- from sdmxlib.model.urn import SdmxUrn
46
-
47
47
  #: The component classes SDMX spells differently across versions. A URN carries
48
48
  #: the spelling of whatever produced it — 2.1 names an attribute ``Attribute``
49
49
  #: and the single measure ``PrimaryMeasure``, 3.0 names them ``DataAttribute``
@@ -297,3 +297,93 @@ def _search(owner: object, owner_cls: type, item_id: str, wanted: tuple[str, ...
297
297
  if found is not None:
298
298
  return found
299
299
  return None
300
+
301
+
302
+ def item_urn_of[T: SdmxItem](scheme: SupportsUrn, item_cls: type[T], item_id: str) -> SdmxUrn[T]:
303
+ """The URN of *item_id* inside *scheme*, as SDMX spells it.
304
+
305
+ The inverse of `scheme_urn_of`, and the constructive half of what #404 and
306
+ #405 made the library able to *read*. Agency, id and version come off the
307
+ scheme rather than from the caller, which is the whole point: a URN built by
308
+ hand can name a parent that does not exist — six keyword arguments, three of
309
+ them restating the scheme — and `validate_references` will now correctly
310
+ report that as broken (#407).
311
+ """
312
+ return SdmxUrn.make(
313
+ item_cls,
314
+ package=item_cls.sdmx_package,
315
+ artefact_class=item_cls.sdmx_class,
316
+ agency=scheme.agency_id,
317
+ id=scheme.id,
318
+ version=scheme.version,
319
+ item_id=item_id,
320
+ )
321
+
322
+
323
+ def locate_item(scheme: object) -> Iterator[tuple[str, type[SdmxItem]]]:
324
+ """The (field name, item class) pairs `ref_to` searches, in order."""
325
+ yield from item_fields(type(scheme))
326
+
327
+
328
+ def item_ref[T: SdmxItem](scheme: SupportsUrn, item_id: str, item_cls: type[T]) -> Ref[T]:
329
+ """A resolved `Ref` to one item of *scheme*, found by id.
330
+
331
+ Raises:
332
+ KeyError: If no item of that class in *scheme* has this id. The message
333
+ names the scheme, so a caller who passed the right id to the wrong
334
+ artefact is told which artefact answered.
335
+ """
336
+ found = _search(scheme, type(scheme), item_id, spellings(item_cls.sdmx_class))
337
+ if found is None and isinstance(scheme, SupportsItemLookup):
338
+ # A lazy facade keeps its items behind a query, so the declared-field
339
+ # walk finds nothing; its subscript is the targeted lookup. Same order
340
+ # as `find_item`, and the reason `LazyCodelist.ref_to` can exist at all
341
+ # rather than being refused until you materialise (#407).
342
+ found = _by_subscript(scheme, item_id, spellings(item_cls.sdmx_class))
343
+ # `isinstance` rather than a cast: it narrows `_search`'s `object | None` to
344
+ # `T` with a runtime check, and covers "not found" in the same branch.
345
+ if not isinstance(found, item_cls):
346
+ msg = (
347
+ f"no {item_cls.sdmx_class} with id {item_id!r} in "
348
+ f"{type(scheme).__name__} {scheme.agency_id}:{scheme.id}({scheme.version})"
349
+ )
350
+ raise KeyError(msg)
351
+ return Ref[T].of(item_urn_of(scheme, item_cls, item_id), found)
352
+
353
+
354
+ def any_item_ref(scheme: SupportsUrn, item_id: str) -> Ref[object]:
355
+ """A resolved `Ref` to one item of *scheme*, whichever kind holds the id.
356
+
357
+ For a `DataStructure`, whose `dimensions`, `time_dimension`,
358
+ `measure_dimension`, `measures`, `attributes` and `groups` are six item
359
+ lists under one maintainable. Dispatching is unambiguous rather than
360
+ convenient: `storage.schema` declares
361
+ ``PRIMARY KEY (dsd_urn, component_id)`` on `dsd_component`, so this library
362
+ already treats a component id as unique across the whole DSD, whichever
363
+ list it sits in (#407).
364
+
365
+ `Ref[object]` because the class varies and every field that consumes one of
366
+ these — `HierarchyAssociation.linked_object`, `Categorisation.target` — is
367
+ declared `Ref[object]`. The URN carries the real class, so nothing is lost
368
+ at runtime; a caller who needs it statically can `isinstance` the unwrapped
369
+ item.
370
+
371
+ Raises:
372
+ KeyError: If no item of *scheme* has this id, naming the fields
373
+ searched so the message says where it looked.
374
+ """
375
+ for name, item_cls in locate_item(scheme):
376
+ value = getattr(scheme, name, None)
377
+ found = _by_id(value, item_id)
378
+ if found is None:
379
+ found = _search(value, item_cls, item_id, spellings(item_cls.sdmx_class))
380
+ # `isinstance` rather than `is not None`: it narrows to the item class,
381
+ # which is what `Ref.of` needs to type the pair it is handed.
382
+ if isinstance(found, item_cls):
383
+ return Ref[object].of(item_urn_of(scheme, item_cls, item_id), found)
384
+ searched = ", ".join(name for name, _ in locate_item(scheme)) or "no item fields"
385
+ msg = (
386
+ f"no item with id {item_id!r} in {type(scheme).__name__} "
387
+ f"{scheme.agency_id}:{scheme.id}({scheme.version}) — searched {searched}"
388
+ )
389
+ raise KeyError(msg)
@@ -5,6 +5,7 @@ from typing import ClassVar
5
5
 
6
6
  from attrs import define, evolve, field
7
7
 
8
+ from sdmxlib.model._items import item_ref
8
9
  from sdmxlib.model._refshape import derived_view
9
10
  from sdmxlib.model.annotations import Annotations
10
11
  from sdmxlib.model.base import MaintainableArtefact
@@ -73,6 +74,22 @@ class CategoryScheme(MaintainableArtefact):
73
74
  _flat: ItemList[Category] = field(init=False, repr=False, factory=ItemList, metadata=derived_view())
74
75
  _parent_of: dict[str, str | None] = field(init=False, repr=False, eq=False, factory=dict)
75
76
 
77
+ def ref_to(self, category_id: str) -> "Ref[Category]":
78
+ """Build a resolved `Ref` to a category within this scheme.
79
+
80
+ The URN is built from *this* scheme's agency, id and version, so it
81
+ cannot name a parent that does not exist — which is what six hand-written
82
+ keyword arguments to `SdmxUrn.make` invite, and what
83
+ `validate_references` now correctly reports as broken (#407).
84
+
85
+ Raises:
86
+ KeyError: If no category with this id is in this scheme.
87
+
88
+ Example:
89
+ Categorisation(..., target=cs.ref_to("ECON"))
90
+ """
91
+ return item_ref(self, category_id, Category)
92
+
76
93
  def __attrs_post_init__(self) -> None:
77
94
  self._flat = _flatten_categories(self.categories)
78
95
  parent: dict[str, str | None] = {n.id: None for n in self._flat}