dirigent-block-parquet 0.17.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,18 @@
1
+ Copyright (c) 2026 Morten Olav Hansen <morten@winterop.com>. All rights reserved.
2
+
3
+ This source code and accompanying documentation are the property of
4
+ Morten Olav Hansen. No license, express or implied, is granted to use, copy,
5
+ modify, merge, publish, distribute, sublicense, or sell copies of this
6
+ software or its derivatives.
7
+
8
+ The source is published for reference only. Any use beyond reading
9
+ requires written permission from the copyright holder.
10
+
11
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS
12
+ OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF
13
+ MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE, AND NONINFRINGEMENT.
14
+ IN NO EVENT SHALL THE COPYRIGHT HOLDER BE LIABLE FOR ANY CLAIM, DAMAGES,
15
+ OR OTHER LIABILITY ARISING FROM THE USE OF THE SOFTWARE.
16
+
17
+ Third-party components redistributed with this software, and the licences they
18
+ carry, are listed in THIRD_PARTY_NOTICES.md.
@@ -0,0 +1,44 @@
1
+ Metadata-Version: 2.4
2
+ Name: dirigent-block-parquet
3
+ Version: 0.17.2
4
+ Summary: The parquet block family for dirigent: the convert.arrow codec, on pyarrow.
5
+ License-Expression: LicenseRef-Proprietary
6
+ License-File: LICENSE
7
+ Classifier: Programming Language :: Python :: 3
8
+ Classifier: Programming Language :: Python :: 3.13
9
+ Requires-Dist: dirigent-common==0.17.2
10
+ Requires-Dist: dirigent-plugin==0.17.2
11
+ Requires-Dist: pyarrow>=21.0.0
12
+ Requires-Python: >=3.13
13
+ Description-Content-Type: text/markdown
14
+
15
+ # dirigent-block-parquet
16
+
17
+ The parquet block family: one codec, `convert.arrow`, trading parquet with the
18
+ three text spellings `convert.std` already trades between.
19
+
20
+ | From | To | What it does |
21
+ | --- | --- | --- |
22
+ | `parquet` | `json`, `ndjson`, `csv` | Each row becomes one record, every value in its JSON spelling. |
23
+ | `json`, `ndjson`, `csv` | `parquet` | Each record becomes one row, under a schema inferred from the records. |
24
+
25
+ The record model is `convert.std`'s: a sequence of flat records. What parquet adds is
26
+ types, and the codec is honest about them in both directions:
27
+
28
+ - **Writing** infers one type per column -- boolean, int64, float64, or string -- from the
29
+ records. Integers and floats unify to a float column, because JSON calls both a number;
30
+ any other mix is refused naming the column. A column every row leaves null is written as
31
+ a nullable string column. A csv source carries no types, so csv to parquet writes string
32
+ columns and nothing else.
33
+ - **Reading** gives every value its JSON spelling: timestamps, dates and times come back as
34
+ ISO strings, decimals as strings, and a float that is NaN or infinite as null. A nested
35
+ column, or one holding raw bytes, is refused naming it -- flattening is a reshape, and a
36
+ reshape belongs to a jq step that knows what the flattening should mean.
37
+
38
+ A conversion is a storage-object operation, like `storage.copy`: it reads the object at
39
+ `source` and writes the one at `target`, and neither side is ever a value the step carries.
40
+ A run that means to look at the records reads them with `storage.read`.
41
+
42
+ The family ships separately because `convert.std` is deliberately on the standard library
43
+ and nothing else; this codec stands on [pyarrow](https://arrow.apache.org/docs/python/).
44
+ Nothing else in the workspace may depend on it.
@@ -0,0 +1,30 @@
1
+ # dirigent-block-parquet
2
+
3
+ The parquet block family: one codec, `convert.arrow`, trading parquet with the
4
+ three text spellings `convert.std` already trades between.
5
+
6
+ | From | To | What it does |
7
+ | --- | --- | --- |
8
+ | `parquet` | `json`, `ndjson`, `csv` | Each row becomes one record, every value in its JSON spelling. |
9
+ | `json`, `ndjson`, `csv` | `parquet` | Each record becomes one row, under a schema inferred from the records. |
10
+
11
+ The record model is `convert.std`'s: a sequence of flat records. What parquet adds is
12
+ types, and the codec is honest about them in both directions:
13
+
14
+ - **Writing** infers one type per column -- boolean, int64, float64, or string -- from the
15
+ records. Integers and floats unify to a float column, because JSON calls both a number;
16
+ any other mix is refused naming the column. A column every row leaves null is written as
17
+ a nullable string column. A csv source carries no types, so csv to parquet writes string
18
+ columns and nothing else.
19
+ - **Reading** gives every value its JSON spelling: timestamps, dates and times come back as
20
+ ISO strings, decimals as strings, and a float that is NaN or infinite as null. A nested
21
+ column, or one holding raw bytes, is refused naming it -- flattening is a reshape, and a
22
+ reshape belongs to a jq step that knows what the flattening should mean.
23
+
24
+ A conversion is a storage-object operation, like `storage.copy`: it reads the object at
25
+ `source` and writes the one at `target`, and neither side is ever a value the step carries.
26
+ A run that means to look at the records reads them with `storage.read`.
27
+
28
+ The family ships separately because `convert.std` is deliberately on the standard library
29
+ and nothing else; this codec stands on [pyarrow](https://arrow.apache.org/docs/python/).
30
+ Nothing else in the workspace may depend on it.
@@ -0,0 +1,30 @@
1
+ [project]
2
+ name = "dirigent-block-parquet"
3
+ version = "0.17.2"
4
+ description = "The parquet block family for dirigent: the convert.arrow codec, on pyarrow."
5
+ readme = "README.md"
6
+ requires-python = ">=3.13"
7
+ license = "LicenseRef-Proprietary"
8
+ license-files = ["LICENSE"]
9
+ classifiers = [
10
+ "Programming Language :: Python :: 3",
11
+ "Programming Language :: Python :: 3.13",
12
+ ]
13
+ dependencies = [
14
+ "dirigent-common==0.17.2",
15
+ "dirigent-plugin==0.17.2",
16
+ "pyarrow>=21.0.0",
17
+ ]
18
+
19
+ [project.entry-points."dirigent.plugins.v1"]
20
+ block-parquet = "dirigent_block_parquet:plugin"
21
+
22
+ [build-system]
23
+ requires = ["uv_build>=0.12.0,<0.13.0"]
24
+ build-backend = "uv_build"
25
+
26
+ [tool.uv.sources.dirigent-plugin]
27
+ workspace = true
28
+
29
+ [tool.uv.sources.dirigent-common]
30
+ workspace = true
@@ -0,0 +1,28 @@
1
+ [project]
2
+ name = "dirigent-block-parquet"
3
+ version = "0.17.2"
4
+ description = "The parquet block family for dirigent: the convert.arrow codec, on pyarrow."
5
+ readme = "README.md"
6
+ requires-python = ">=3.13"
7
+ license = "LicenseRef-Proprietary"
8
+ license-files = ["LICENSE"]
9
+ classifiers = [
10
+ "Programming Language :: Python :: 3",
11
+ "Programming Language :: Python :: 3.13",
12
+ ]
13
+ dependencies = [
14
+ "dirigent-common==0.17.2",
15
+ "dirigent-plugin==0.17.2",
16
+ "pyarrow>=21.0.0",
17
+ ]
18
+
19
+ [project.entry-points."dirigent.plugins.v1"]
20
+ block-parquet = "dirigent_block_parquet:plugin"
21
+
22
+ [build-system]
23
+ requires = ["uv_build>=0.12.0,<0.13.0"]
24
+ build-backend = "uv_build"
25
+
26
+ [tool.uv.sources]
27
+ dirigent-plugin = { workspace = true }
28
+ dirigent-common = { workspace = true }
@@ -0,0 +1,24 @@
1
+ """The parquet block family: the ``convert.arrow`` codec, on pyarrow."""
2
+
3
+ from dirigent_block_parquet.arrow import TEXT_FORMATS, UNIT, ArrowConverter
4
+ from dirigent_plugin import Contribution, extension
5
+
6
+
7
+ class ParquetBlocks:
8
+ """The plugin object the host discovers under the dirigent.plugins.v1 entry-point group."""
9
+
10
+ @extension
11
+ def contribute(self) -> Contribution:
12
+ """Contribute the ``convert.arrow`` codec."""
13
+ return Contribution(operators=[ArrowConverter()])
14
+
15
+
16
+ plugin = ParquetBlocks()
17
+
18
+ __all__ = [
19
+ "TEXT_FORMATS",
20
+ "UNIT",
21
+ "ArrowConverter",
22
+ "ParquetBlocks",
23
+ "plugin",
24
+ ]
@@ -0,0 +1,261 @@
1
+ """The ``arrow`` engine for the convert verb: parquet against the text formats, on pyarrow."""
2
+
3
+ import csv
4
+ import io
5
+ import json
6
+ from typing import Final
7
+
8
+ import pyarrow as pa
9
+ import pyarrow.parquet as pq
10
+ from pydantic import JsonValue
11
+
12
+ from dirigent_block_parquet.messages import (
13
+ BINARY_COLUMN,
14
+ CSV_HEADER_REPEATED,
15
+ CSV_HEADER_UNNAMED,
16
+ CSV_ROW_TOO_WIDE,
17
+ LINE_NOT_JSON,
18
+ MIXED_COLUMN,
19
+ NESTED_COLUMN,
20
+ NESTED_VALUE,
21
+ NO_JSON_SPELLING,
22
+ NOT_A_JSON_ARRAY,
23
+ NOT_JSON,
24
+ NOT_PARQUET,
25
+ NOT_UTF8,
26
+ RECORD_NOT_AN_OBJECT,
27
+ )
28
+ from dirigent_common import spelled
29
+ from dirigent_plugin import Converter, TransformError
30
+
31
+ #: The text spellings this engine trades parquet with.
32
+ TEXT_FORMATS: Final = ("json", "ndjson", "csv")
33
+
34
+ #: What a target format calls one element of the sequence it writes.
35
+ UNIT: Final = {"json": "element", "ndjson": "line", "csv": "row", "parquet": "row"}
36
+
37
+
38
+ class ArrowConverter(Converter):
39
+ """Re-encodes a sequence of records between parquet and the text formats.
40
+
41
+ The record model is ``convert.std``'s: a sequence of flat records, read out of one
42
+ spelling and written in another. What parquet adds is types -- a column knows whether it
43
+ holds numbers or text -- so writing infers a schema from the records, and a column whose
44
+ rows disagree about their type is refused naming it rather than coerced. Reading gives
45
+ every value its JSON spelling: timestamps, dates and times come back as ISO strings,
46
+ decimals as strings, and a float that is NaN or infinite as null, because JSON has no
47
+ other words for them.
48
+
49
+ A csv carries no types, so csv to parquet writes string columns and nothing else.
50
+ """
51
+
52
+ kind = "arrow"
53
+ summary = "Convert between parquet and the text formats."
54
+ pairs = frozenset({("parquet", text) for text in TEXT_FORMATS} | {(text, "parquet") for text in TEXT_FORMATS})
55
+
56
+ def convert(self, source: bytes, *, source_format: str, target_format: str) -> bytes:
57
+ """Read the records out of the source format and write them in the target format."""
58
+ if source_format == "parquet":
59
+ records = _read_parquet(source, target_format)
60
+ else:
61
+ records = _read_text(source, source_format, target_format)
62
+ if target_format == "parquet":
63
+ return _write_parquet(records)
64
+ return _write_text(records, target_format).encode()
65
+
66
+
67
+ def _read_parquet(source: bytes, target_format: str) -> list[dict[str, JsonValue]]:
68
+ """Read a parquet payload as records, refusing what the target has no spelling for."""
69
+ try:
70
+ table = pq.read_table(pa.BufferReader(source)) # pyright: ignore[reportUnknownMemberType]
71
+ except pa.ArrowInvalid as error:
72
+ raise TransformError(NOT_PARQUET.render(detail=str(error))) from error
73
+ for name, kind in zip(table.schema.names, table.schema.types, strict=True):
74
+ if pa.types.is_nested(kind):
75
+ raise TransformError(
76
+ NESTED_COLUMN.render(
77
+ column=repr(name), kind=kind, target_format=target_format, unit=UNIT[target_format]
78
+ )
79
+ )
80
+ if pa.types.is_binary(kind) or pa.types.is_large_binary(kind) or pa.types.is_fixed_size_binary(kind):
81
+ raise TransformError(BINARY_COLUMN.render(column=repr(name)))
82
+ return [{key: _spelled(value) for key, value in row.items()} for row in table.to_pylist()]
83
+
84
+
85
+ def _spelled(value: object) -> JsonValue:
86
+ """Give one arrow value its JSON spelling, in the one house conversion."""
87
+ try:
88
+ return spelled(value)
89
+ except ValueError as error:
90
+ # A schema check above rules out nested and binary columns, so nothing else arrives.
91
+ raise TransformError(NO_JSON_SPELLING.render(detail=str(error))) from error
92
+
93
+
94
+ def _read_text(source: bytes, source_format: str, target_format: str) -> list[dict[str, JsonValue]]:
95
+ """Read the sequence of records a text format spells out, refusing rows that nest."""
96
+ try:
97
+ text = source.decode("utf-8")
98
+ except UnicodeDecodeError as error:
99
+ raise TransformError(NOT_UTF8.render(reason=error.reason, position=error.start)) from error
100
+ if source_format == "csv":
101
+ return _read_csv(text)
102
+ values = _read_json_values(text, source_format, target_format)
103
+ records: list[dict[str, JsonValue]] = []
104
+ for number, value in enumerate(values, start=1):
105
+ if not isinstance(value, dict):
106
+ raise TransformError(RECORD_NOT_AN_OBJECT.render(unit=UNIT[source_format], number=number))
107
+ records.append(value)
108
+ return records
109
+
110
+
111
+ def _read_json_values(text: str, source_format: str, target_format: str) -> list[JsonValue]:
112
+ """Read a JSON array, or one JSON value per line, the way convert.std reads them."""
113
+ if source_format == "json":
114
+ try:
115
+ parsed: JsonValue = json.loads(text)
116
+ except ValueError as error:
117
+ raise TransformError(NOT_JSON.render(detail=str(error))) from error
118
+ if not isinstance(parsed, list):
119
+ raise TransformError(NOT_A_JSON_ARRAY.render(target_format=target_format, unit=UNIT[target_format]))
120
+ return parsed
121
+ values: list[JsonValue] = []
122
+ for number, line in enumerate(text.splitlines(), start=1):
123
+ if not line.strip():
124
+ continue
125
+ try:
126
+ values.append(json.loads(line))
127
+ except ValueError as error:
128
+ raise TransformError(LINE_NOT_JSON.render(number=number, detail=str(error))) from error
129
+ return values
130
+
131
+
132
+ def _read_csv(text: str) -> list[dict[str, JsonValue]]:
133
+ """Read the header row and its rows, as objects whose every value is a string."""
134
+ rows = csv.reader(io.StringIO(text, newline=""))
135
+ header = next(rows, None)
136
+ if header is None:
137
+ return []
138
+ _check_header(header)
139
+ records: list[dict[str, JsonValue]] = []
140
+ for number, row in enumerate(rows, start=1):
141
+ if not row:
142
+ continue
143
+ if len(row) > len(header):
144
+ raise TransformError(CSV_ROW_TOO_WIDE.render(number=number))
145
+ records.append({name: row[index] if index < len(row) else "" for index, name in enumerate(header)})
146
+ return records
147
+
148
+
149
+ def _columns_phrase(positions: list[int]) -> str:
150
+ """Word one or more column positions, counting from one."""
151
+ listed = ", ".join(str(position) for position in positions)
152
+ return f"column {listed}" if len(positions) == 1 else f"columns {listed}"
153
+
154
+
155
+ def _check_header(header: list[str]) -> None:
156
+ """Refuse a header that leaves a name empty or repeats one.
157
+
158
+ A record key names its column, so neither has a record spelling: keeping one of two
159
+ columns that share a name, or keying a column by the empty string, drops content while
160
+ reporting success.
161
+ """
162
+ positions: dict[str, list[int]] = {}
163
+ for index, name in enumerate(header, start=1):
164
+ positions.setdefault(name, []).append(index)
165
+ unnamed = positions.pop("", None)
166
+ if unnamed is not None:
167
+ raise TransformError(CSV_HEADER_UNNAMED.render(columns=_columns_phrase(unnamed)))
168
+ repeated = [(name, where) for name, where in positions.items() if len(where) > 1]
169
+ if repeated:
170
+ listed = "; ".join(f"{name!r} at {_columns_phrase(where)}" for name, where in repeated)
171
+ raise TransformError(CSV_HEADER_REPEATED.render(listed=listed))
172
+
173
+
174
+ #: The arrow type each column kind writes as. A column every row leaves null carries no
175
+ #: kind at all and is written as a nullable string column.
176
+ KIND_TYPES: Final[dict[str, pa.DataType]] = {
177
+ "boolean": pa.bool_(),
178
+ "integer": pa.int64(),
179
+ "number": pa.float64(),
180
+ "string": pa.string(),
181
+ }
182
+
183
+
184
+ def _write_parquet(records: list[dict[str, JsonValue]]) -> bytes:
185
+ """Infer one type per column from the records, then write them as parquet bytes."""
186
+ schema = pa.schema([(name, KIND_TYPES[kind]) for name, kind in _columns(records).items()])
187
+ sink = io.BytesIO()
188
+ pq.write_table(pa.Table.from_pylist(records, schema=schema), sink) # pyright: ignore[reportUnknownMemberType]
189
+ return sink.getvalue()
190
+
191
+
192
+ def _columns(records: list[dict[str, JsonValue]]) -> dict[str, str]:
193
+ """Name each column's one kind, refusing a column whose rows disagree.
194
+
195
+ Integers and floats unify to a float column, because JSON calls both a number; any
196
+ other mix is two types in one column, which parquet does not write and a codec that
197
+ coerced would not preserve.
198
+ """
199
+ kinds: dict[str, str] = {}
200
+ for number, record in enumerate(records, start=1):
201
+ for key, value in record.items():
202
+ kind = _kind(value, number, key)
203
+ if kind is None:
204
+ continue
205
+ settled = kinds.get(key)
206
+ if settled is None:
207
+ kinds[key] = kind
208
+ elif {settled, kind} == {"integer", "number"}:
209
+ kinds[key] = "number"
210
+ elif settled != kind:
211
+ raise TransformError(MIXED_COLUMN.render(column=repr(key), settled=settled, kind=kind))
212
+ for key in {name for record in records for name in record}:
213
+ kinds.setdefault(key, "string")
214
+ return kinds
215
+
216
+
217
+ def _kind(value: JsonValue, number: int, key: str) -> str | None:
218
+ """Name one value's column kind, refusing the nested ones parquet is not asked to hold."""
219
+ if value is None:
220
+ return None
221
+ if isinstance(value, dict | list):
222
+ raise TransformError(NESTED_VALUE.render(number=number, key=repr(key)))
223
+ # Before the integer check, because a bool is an int in Python and is not one in JSON.
224
+ if isinstance(value, bool):
225
+ return "boolean"
226
+ if isinstance(value, int):
227
+ return "integer"
228
+ if isinstance(value, float):
229
+ return "number"
230
+ return "string"
231
+
232
+
233
+ def _write_text(records: list[dict[str, JsonValue]], target_format: str) -> str:
234
+ """Write the records in a text spelling, the way convert.std writes them."""
235
+ if target_format == "json":
236
+ return json.dumps(records, separators=(",", ":"))
237
+ if target_format == "ndjson":
238
+ return "".join(f"{json.dumps(record, separators=(',', ':'))}\n" for record in records)
239
+ return _write_csv(records)
240
+
241
+
242
+ def _write_csv(records: list[dict[str, JsonValue]]) -> str:
243
+ """Write a header of every key any row has, in first-seen order, and one row per record."""
244
+ header: list[str] = []
245
+ for record in records:
246
+ header.extend(key for key in record if key not in header)
247
+ out = io.StringIO(newline="")
248
+ writer = csv.writer(out, lineterminator="\n")
249
+ writer.writerow(header)
250
+ for record in records:
251
+ writer.writerow([_cell(record.get(key)) for key in header])
252
+ return out.getvalue()
253
+
254
+
255
+ def _cell(value: JsonValue) -> str:
256
+ """Render one value as csv text. Reading ruled the nested ones out already."""
257
+ if value is None:
258
+ return ""
259
+ if isinstance(value, str):
260
+ return value
261
+ return json.dumps(value)
@@ -0,0 +1,70 @@
1
+ """Every refusal the parquet codec makes, catalogued under the ``parquet`` prefix.
2
+
3
+ A codec refusal reaches an attempt as ``plugin.transform_failed``: the convert frame owns
4
+ the failure, and the codec's sentence rides on it as the ``detail`` param.
5
+ """
6
+
7
+ from dirigent_common import Catalogue
8
+
9
+ PARQUET = Catalogue("parquet")
10
+
11
+ NOT_PARQUET = PARQUET.define("not_parquet", "the input is not parquet: {detail}")
12
+
13
+ NESTED_COLUMN = PARQUET.define(
14
+ "nested_column",
15
+ "column {column} is {kind}, and a nested column has no {target_format} {unit} spelling; "
16
+ "flatten it before converting",
17
+ )
18
+
19
+ BINARY_COLUMN = PARQUET.define(
20
+ "binary_column",
21
+ "column {column} holds raw bytes, which have no JSON spelling; decode or drop it before converting",
22
+ )
23
+
24
+ NO_JSON_SPELLING = PARQUET.define("no_json_spelling", "{detail}")
25
+
26
+ NOT_UTF8 = PARQUET.define(
27
+ "not_utf8",
28
+ "the input is not UTF-8 text: {reason} at byte {position}; a text format is UTF-8",
29
+ )
30
+
31
+ RECORD_NOT_AN_OBJECT = PARQUET.define(
32
+ "record_not_an_object",
33
+ "{unit} {number} of the input is not an object, and a parquet row is a flat object",
34
+ )
35
+
36
+ NOT_JSON = PARQUET.define("not_json", "the input is not JSON: {detail}")
37
+
38
+ NOT_A_JSON_ARRAY = PARQUET.define(
39
+ "not_a_json_array",
40
+ "json to {target_format} writes one {unit} per element, so the input has to be a JSON array",
41
+ )
42
+
43
+ LINE_NOT_JSON = PARQUET.define("line_not_json", "line {number} of the input is not JSON: {detail}")
44
+
45
+ CSV_ROW_TOO_WIDE = PARQUET.define(
46
+ "csv_row_too_wide",
47
+ "row {number} of the csv has more cells than the header names columns, "
48
+ "and a cell no column names has nowhere to go",
49
+ )
50
+
51
+ CSV_HEADER_UNNAMED = PARQUET.define(
52
+ "csv_header_unnamed",
53
+ "the csv header has no name at {columns}, and a record key names its column; name it before converting",
54
+ )
55
+
56
+ CSV_HEADER_REPEATED = PARQUET.define(
57
+ "csv_header_repeated",
58
+ "the csv header repeats a column name: {listed}; a record key names one column, so rename them before converting",
59
+ )
60
+
61
+ MIXED_COLUMN = PARQUET.define(
62
+ "mixed_column",
63
+ "column {column} holds both {settled} and {kind} values, and a parquet "
64
+ "column carries one type; reshape it before converting",
65
+ )
66
+
67
+ NESTED_VALUE = PARQUET.define(
68
+ "nested_value",
69
+ "row {number} has a nested value at {key}, and this codec writes flat columns; flatten it before converting",
70
+ )