dirigent-parquet 0.9.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,18 @@
1
+ Copyright (c) 2026 Morten Olav Hansen <morten@winterop.com>. All rights reserved.
2
+
3
+ This source code and accompanying documentation are the property of
4
+ Morten Olav Hansen. No license, express or implied, is granted to use, copy,
5
+ modify, merge, publish, distribute, sublicense, or sell copies of this
6
+ software or its derivatives.
7
+
8
+ The source is published for reference only. Any use beyond reading
9
+ requires written permission from the copyright holder.
10
+
11
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS
12
+ OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF
13
+ MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE, AND NONINFRINGEMENT.
14
+ IN NO EVENT SHALL THE COPYRIGHT HOLDER BE LIABLE FOR ANY CLAIM, DAMAGES,
15
+ OR OTHER LIABILITY ARISING FROM THE USE OF THE SOFTWARE.
16
+
17
+ Third-party components redistributed with this software, and the licences they
18
+ carry, are listed in THIRD_PARTY_NOTICES.md.
@@ -0,0 +1,42 @@
1
+ Metadata-Version: 2.4
2
+ Name: dirigent-parquet
3
+ Version: 0.9.0
4
+ Summary: The parquet format pack for dirigent: the convert.arrow codec, on pyarrow.
5
+ License-Expression: LicenseRef-Proprietary
6
+ License-File: LICENSE
7
+ Requires-Dist: dirigent-common
8
+ Requires-Dist: dirigent-plugin
9
+ Requires-Dist: pyarrow>=21.0.0
10
+ Requires-Python: >=3.13
11
+ Description-Content-Type: text/markdown
12
+
13
+ # dirigent-parquet
14
+
15
+ The parquet format pack for dirigent: one codec, `convert.arrow`, trading parquet with the
16
+ three text spellings `convert.std` already trades between.
17
+
18
+ | From | To | What it does |
19
+ | --- | --- | --- |
20
+ | `parquet` | `json`, `ndjson`, `csv` | Each row becomes one record, every value in its JSON spelling. |
21
+ | `json`, `ndjson`, `csv` | `parquet` | Each record becomes one row, under a schema inferred from the records. |
22
+
23
+ The record model is `convert.std`'s: a sequence of flat records. What parquet adds is
24
+ types, and the codec is honest about them in both directions:
25
+
26
+ - **Writing** infers one type per column -- boolean, int64, float64, or string -- from the
27
+ records. Integers and floats unify to a float column, because JSON calls both a number;
28
+ any other mix is refused naming the column. A column every row leaves null is written as
29
+ a nullable string column. A csv source carries no types, so csv to parquet writes string
30
+ columns and nothing else.
31
+ - **Reading** gives every value its JSON spelling: timestamps, dates and times come back as
32
+ ISO strings, decimals as strings, and a float that is NaN or infinite as null. A nested
33
+ column, or one holding raw bytes, is refused naming it -- flattening is a reshape, and a
34
+ reshape belongs to a jq step that knows what the flattening should mean.
35
+
36
+ Parquet is bytes rather than text, so it always travels by uri: `input_uri` in, and
37
+ `save_to` required when parquet is the target. The inline `input` field is for text
38
+ sources only, and a config that breaks either rule is refused at apply.
39
+
40
+ The pack ships separately because `convert.std` is deliberately on the standard library
41
+ and nothing else; this codec stands on [pyarrow](https://arrow.apache.org/docs/python/).
42
+ Nothing else in the workspace may depend on it.
@@ -0,0 +1,30 @@
1
+ # dirigent-parquet
2
+
3
+ The parquet format pack for dirigent: one codec, `convert.arrow`, trading parquet with the
4
+ three text spellings `convert.std` already trades between.
5
+
6
+ | From | To | What it does |
7
+ | --- | --- | --- |
8
+ | `parquet` | `json`, `ndjson`, `csv` | Each row becomes one record, every value in its JSON spelling. |
9
+ | `json`, `ndjson`, `csv` | `parquet` | Each record becomes one row, under a schema inferred from the records. |
10
+
11
+ The record model is `convert.std`'s: a sequence of flat records. What parquet adds is
12
+ types, and the codec is honest about them in both directions:
13
+
14
+ - **Writing** infers one type per column -- boolean, int64, float64, or string -- from the
15
+ records. Integers and floats unify to a float column, because JSON calls both a number;
16
+ any other mix is refused naming the column. A column every row leaves null is written as
17
+ a nullable string column. A csv source carries no types, so csv to parquet writes string
18
+ columns and nothing else.
19
+ - **Reading** gives every value its JSON spelling: timestamps, dates and times come back as
20
+ ISO strings, decimals as strings, and a float that is NaN or infinite as null. A nested
21
+ column, or one holding raw bytes, is refused naming it -- flattening is a reshape, and a
22
+ reshape belongs to a jq step that knows what the flattening should mean.
23
+
24
+ Parquet is bytes rather than text, so it always travels by uri: `input_uri` in, and
25
+ `save_to` required when parquet is the target. The inline `input` field is for text
26
+ sources only, and a config that breaks either rule is refused at apply.
27
+
28
+ The pack ships separately because `convert.std` is deliberately on the standard library
29
+ and nothing else; this codec stands on [pyarrow](https://arrow.apache.org/docs/python/).
30
+ Nothing else in the workspace may depend on it.
@@ -0,0 +1,26 @@
1
+ [project]
2
+ name = "dirigent-parquet"
3
+ version = "0.9.0"
4
+ description = "The parquet format pack for dirigent: the convert.arrow codec, on pyarrow."
5
+ readme = "README.md"
6
+ requires-python = ">=3.13"
7
+ license = "LicenseRef-Proprietary"
8
+ license-files = ["LICENSE"]
9
+ dependencies = [
10
+ "dirigent-common",
11
+ "dirigent-plugin",
12
+ "pyarrow>=21.0.0",
13
+ ]
14
+
15
+ [project.entry-points."dirigent.plugins.v1"]
16
+ parquet = "dirigent_parquet:plugin"
17
+
18
+ [build-system]
19
+ requires = ["uv_build>=0.12.0,<0.13.0"]
20
+ build-backend = "uv_build"
21
+
22
+ [tool.uv.sources.dirigent-plugin]
23
+ workspace = true
24
+
25
+ [tool.uv.sources.dirigent-common]
26
+ workspace = true
@@ -0,0 +1,24 @@
1
+ [project]
2
+ name = "dirigent-parquet"
3
+ version = "0.9.0"
4
+ description = "The parquet format pack for dirigent: the convert.arrow codec, on pyarrow."
5
+ readme = "README.md"
6
+ requires-python = ">=3.13"
7
+ license = "LicenseRef-Proprietary"
8
+ license-files = ["LICENSE"]
9
+ dependencies = [
10
+ "dirigent-common",
11
+ "dirigent-plugin",
12
+ "pyarrow>=21.0.0",
13
+ ]
14
+
15
+ [project.entry-points."dirigent.plugins.v1"]
16
+ parquet = "dirigent_parquet:plugin"
17
+
18
+ [build-system]
19
+ requires = ["uv_build>=0.12.0,<0.13.0"]
20
+ build-backend = "uv_build"
21
+
22
+ [tool.uv.sources]
23
+ dirigent-plugin = { workspace = true }
24
+ dirigent-common = { workspace = true }
@@ -0,0 +1,25 @@
1
+ """The parquet format pack: the ``convert.arrow`` codec, on pyarrow."""
2
+
3
+ from dirigent_parquet.arrow import BYTES_BY_URI, TEXT_FORMATS, UNIT, ArrowConverter
4
+ from dirigent_plugin import Contribution, extension
5
+
6
+
7
+ class ParquetPlugin:
8
+ """The plugin object the host discovers under the dirigent.plugins.v1 entry-point group."""
9
+
10
+ @extension
11
+ def contribute(self) -> Contribution:
12
+ """Contribute the ``convert.arrow`` codec."""
13
+ return Contribution(operators=[ArrowConverter()])
14
+
15
+
16
+ plugin = ParquetPlugin()
17
+
18
+ __all__ = [
19
+ "BYTES_BY_URI",
20
+ "TEXT_FORMATS",
21
+ "UNIT",
22
+ "ArrowConverter",
23
+ "ParquetPlugin",
24
+ "plugin",
25
+ ]
@@ -0,0 +1,311 @@
1
+ """The ``arrow`` engine for the convert verb: parquet against the text formats, on pyarrow."""
2
+
3
+ import csv
4
+ import io
5
+ import json
6
+ from typing import Final
7
+
8
+ import pyarrow as pa
9
+ import pyarrow.parquet as pq
10
+ from pydantic import BaseModel, JsonValue
11
+
12
+ from dirigent_common import spelled
13
+ from dirigent_plugin import (
14
+ BlockFailure,
15
+ ConvertConfig,
16
+ Converter,
17
+ ConvertOutput,
18
+ ErrorClass,
19
+ RemoteHandle,
20
+ StepContext,
21
+ TransformError,
22
+ )
23
+
24
+ #: The text spellings this engine trades parquet with.
25
+ TEXT_FORMATS: Final = ("json", "ndjson", "csv")
26
+
27
+ #: What a target format calls one element of the sequence it writes.
28
+ UNIT: Final = {"json": "element", "ndjson": "line", "csv": "row", "parquet": "row"}
29
+
30
+ #: Why parquet only ever travels by URI, said once and reused by both refusals.
31
+ BYTES_BY_URI: Final = "parquet is bytes, and bytes travel by uri"
32
+
33
+
34
+ class ArrowConverter(Converter):
35
+ """Re-encodes a sequence of records between parquet and the text formats.
36
+
37
+ The record model is ``convert.std``'s: a sequence of flat records, read out of one
38
+ spelling and written in another. What parquet adds is types -- a column knows whether it
39
+ holds numbers or text -- so writing infers a schema from the records, and a column whose
40
+ rows disagree about their type is refused naming it rather than coerced. Reading gives
41
+ every value its JSON spelling: timestamps, dates and times come back as ISO strings,
42
+ decimals as strings, and a float that is NaN or infinite as null, because JSON has no
43
+ other words for them.
44
+
45
+ A csv carries no types, so csv to parquet writes string columns and nothing else.
46
+
47
+ Parquet is bytes rather than text, so it always travels by uri: the parquet side of a
48
+ conversion is ``input_uri`` in and ``save_to`` out, never the inline ``input`` field.
49
+ """
50
+
51
+ kind = "arrow"
52
+ summary = "Convert between parquet and the text formats."
53
+ pairs = frozenset({("parquet", text) for text in TEXT_FORMATS} | {(text, "parquet") for text in TEXT_FORMATS})
54
+
55
+ def check_config(self, config: BaseModel) -> list[str]:
56
+ """Refuse the pair, and a parquet side asked to travel inline, at apply."""
57
+ issues = super().check_config(config)
58
+ if isinstance(config, ConvertConfig):
59
+ issues.extend(_inline_refusals(config))
60
+ return issues
61
+
62
+ async def execute(self, config: ConvertConfig, ctx: StepContext) -> ConvertOutput | RemoteHandle:
63
+ """Guard the uri rule again at run time, then let the frame do its work.
64
+
65
+ A config whose formats arrived through references is checked here for the first
66
+ time, because apply deferred it.
67
+ """
68
+ refused = _inline_refusals(config)
69
+ if refused:
70
+ raise BlockFailure(refused[0], error_class=ErrorClass.REJECTED)
71
+ return await super().execute(config, ctx)
72
+
73
+ def convert(self, source: bytes, *, source_format: str, target_format: str) -> bytes:
74
+ """Read the records out of the source format and write them in the target format."""
75
+ if source_format == "parquet":
76
+ records = _read_parquet(source, target_format)
77
+ else:
78
+ records = _read_text(source, source_format, target_format)
79
+ if target_format == "parquet":
80
+ return _write_parquet(records)
81
+ return _write_text(records, target_format).encode()
82
+
83
+
84
+ def _inline_refusals(config: ConvertConfig) -> list[str]:
85
+ """Word the refusals of a parquet payload written or asked for inline."""
86
+ issues: list[str] = []
87
+ if config.from_format == "parquet" and config.input is not None:
88
+ issues.append(f"a parquet input is read from input_uri, not written inline: {BYTES_BY_URI}")
89
+ if config.to_format == "parquet" and config.save_to is None:
90
+ issues.append(f"a parquet result needs save_to, because it cannot inline: {BYTES_BY_URI}")
91
+ return issues
92
+
93
+
94
+ def _read_parquet(source: bytes, target_format: str) -> list[dict[str, JsonValue]]:
95
+ """Read a parquet payload as records, refusing what the target has no spelling for."""
96
+ try:
97
+ table = pq.read_table(pa.BufferReader(source)) # pyright: ignore[reportUnknownMemberType]
98
+ except pa.ArrowInvalid as error:
99
+ raise TransformError(f"the input is not parquet: {error}") from error
100
+ for name, kind in zip(table.schema.names, table.schema.types, strict=True):
101
+ if pa.types.is_nested(kind):
102
+ raise TransformError(
103
+ f"column {name!r} is {kind}, and a nested column has no "
104
+ f"{target_format} {UNIT[target_format]} spelling; flatten it before converting"
105
+ )
106
+ if pa.types.is_binary(kind) or pa.types.is_large_binary(kind) or pa.types.is_fixed_size_binary(kind):
107
+ raise TransformError(
108
+ f"column {name!r} holds raw bytes, which have no JSON spelling; decode or drop it before converting"
109
+ )
110
+ return [{key: _spelled(value) for key, value in row.items()} for row in table.to_pylist()]
111
+
112
+
113
+ def _spelled(value: object) -> JsonValue:
114
+ """Give one arrow value its JSON spelling, in the one house conversion."""
115
+ try:
116
+ return spelled(value)
117
+ except ValueError as error:
118
+ # A schema check above rules out nested and binary columns, so nothing else arrives.
119
+ raise TransformError(str(error)) from error
120
+
121
+
122
+ def _read_text(source: bytes, source_format: str, target_format: str) -> list[dict[str, JsonValue]]:
123
+ """Read the sequence of records a text format spells out, refusing rows that nest."""
124
+ try:
125
+ text = source.decode("utf-8")
126
+ except UnicodeDecodeError as error:
127
+ raise TransformError(
128
+ f"the input is not UTF-8 text: {error.reason} at byte {error.start}; a text format is UTF-8"
129
+ ) from error
130
+ if source_format == "csv":
131
+ return _read_csv(text)
132
+ values = _read_json_values(text, source_format, target_format)
133
+ records: list[dict[str, JsonValue]] = []
134
+ for number, value in enumerate(values, start=1):
135
+ if not isinstance(value, dict):
136
+ raise TransformError(
137
+ f"{UNIT[source_format]} {number} of the input is not an object, and a parquet row is a flat object"
138
+ )
139
+ records.append(value)
140
+ return records
141
+
142
+
143
+ def _read_json_values(text: str, source_format: str, target_format: str) -> list[JsonValue]:
144
+ """Read a JSON array, or one JSON value per line, the way convert.std reads them."""
145
+ if source_format == "json":
146
+ try:
147
+ parsed: JsonValue = json.loads(text)
148
+ except ValueError as error:
149
+ raise TransformError(f"the input is not JSON: {error}") from error
150
+ if not isinstance(parsed, list):
151
+ raise TransformError(
152
+ f"json to {target_format} writes one {UNIT[target_format]} per element, "
153
+ f"so the input has to be a JSON array"
154
+ )
155
+ return parsed
156
+ values: list[JsonValue] = []
157
+ for number, line in enumerate(text.splitlines(), start=1):
158
+ if not line.strip():
159
+ continue
160
+ try:
161
+ values.append(json.loads(line))
162
+ except ValueError as error:
163
+ raise TransformError(f"line {number} of the input is not JSON: {error}") from error
164
+ return values
165
+
166
+
167
+ def _read_csv(text: str) -> list[dict[str, JsonValue]]:
168
+ """Read the header row and its rows, as objects whose every value is a string."""
169
+ rows = csv.reader(io.StringIO(text, newline=""))
170
+ header = next(rows, None)
171
+ if header is None:
172
+ return []
173
+ _check_header(header)
174
+ records: list[dict[str, JsonValue]] = []
175
+ for number, row in enumerate(rows, start=1):
176
+ if not row:
177
+ continue
178
+ if len(row) > len(header):
179
+ raise TransformError(
180
+ f"row {number} of the csv has more cells than the header names columns, "
181
+ f"and a cell no column names has nowhere to go"
182
+ )
183
+ records.append({name: row[index] if index < len(row) else "" for index, name in enumerate(header)})
184
+ return records
185
+
186
+
187
+ def _columns_phrase(positions: list[int]) -> str:
188
+ """Word one or more column positions, counting from one."""
189
+ listed = ", ".join(str(position) for position in positions)
190
+ return f"column {listed}" if len(positions) == 1 else f"columns {listed}"
191
+
192
+
193
+ def _check_header(header: list[str]) -> None:
194
+ """Refuse a header that leaves a name empty or repeats one.
195
+
196
+ A record key names its column, so neither has a record spelling: keeping one of two
197
+ columns that share a name, or keying a column by the empty string, drops content while
198
+ reporting success.
199
+ """
200
+ positions: dict[str, list[int]] = {}
201
+ for index, name in enumerate(header, start=1):
202
+ positions.setdefault(name, []).append(index)
203
+ unnamed = positions.pop("", None)
204
+ if unnamed is not None:
205
+ raise TransformError(
206
+ f"the csv header has no name at {_columns_phrase(unnamed)}, and a record key names "
207
+ f"its column; name it before converting"
208
+ )
209
+ repeated = [(name, where) for name, where in positions.items() if len(where) > 1]
210
+ if repeated:
211
+ listed = "; ".join(f"{name!r} at {_columns_phrase(where)}" for name, where in repeated)
212
+ raise TransformError(
213
+ f"the csv header repeats a column name: {listed}; a record key names one column, "
214
+ f"so rename them before converting"
215
+ )
216
+
217
+
218
+ #: The arrow type each column kind writes as. A column every row leaves null carries no
219
+ #: kind at all and is written as a nullable string column.
220
+ KIND_TYPES: Final[dict[str, pa.DataType]] = {
221
+ "boolean": pa.bool_(),
222
+ "integer": pa.int64(),
223
+ "number": pa.float64(),
224
+ "string": pa.string(),
225
+ }
226
+
227
+
228
+ def _write_parquet(records: list[dict[str, JsonValue]]) -> bytes:
229
+ """Infer one type per column from the records, then write them as parquet bytes."""
230
+ schema = pa.schema([(name, KIND_TYPES[kind]) for name, kind in _columns(records).items()])
231
+ sink = io.BytesIO()
232
+ pq.write_table(pa.Table.from_pylist(records, schema=schema), sink) # pyright: ignore[reportUnknownMemberType]
233
+ return sink.getvalue()
234
+
235
+
236
+ def _columns(records: list[dict[str, JsonValue]]) -> dict[str, str]:
237
+ """Name each column's one kind, refusing a column whose rows disagree.
238
+
239
+ Integers and floats unify to a float column, because JSON calls both a number; any
240
+ other mix is two types in one column, which parquet does not write and a codec that
241
+ coerced would not preserve.
242
+ """
243
+ kinds: dict[str, str] = {}
244
+ for number, record in enumerate(records, start=1):
245
+ for key, value in record.items():
246
+ kind = _kind(value, number, key)
247
+ if kind is None:
248
+ continue
249
+ settled = kinds.get(key)
250
+ if settled is None:
251
+ kinds[key] = kind
252
+ elif {settled, kind} == {"integer", "number"}:
253
+ kinds[key] = "number"
254
+ elif settled != kind:
255
+ raise TransformError(
256
+ f"column {key!r} holds both {settled} and {kind} values, and a parquet "
257
+ f"column carries one type; reshape it before converting"
258
+ )
259
+ for key in {name for record in records for name in record}:
260
+ kinds.setdefault(key, "string")
261
+ return kinds
262
+
263
+
264
+ def _kind(value: JsonValue, number: int, key: str) -> str | None:
265
+ """Name one value's column kind, refusing the nested ones parquet is not asked to hold."""
266
+ if value is None:
267
+ return None
268
+ if isinstance(value, dict | list):
269
+ raise TransformError(
270
+ f"row {number} has a nested value at {key!r}, and this codec writes flat "
271
+ f"columns; flatten it before converting"
272
+ )
273
+ # Before the integer check, because a bool is an int in Python and is not one in JSON.
274
+ if isinstance(value, bool):
275
+ return "boolean"
276
+ if isinstance(value, int):
277
+ return "integer"
278
+ if isinstance(value, float):
279
+ return "number"
280
+ return "string"
281
+
282
+
283
+ def _write_text(records: list[dict[str, JsonValue]], target_format: str) -> str:
284
+ """Write the records in a text spelling, the way convert.std writes them."""
285
+ if target_format == "json":
286
+ return json.dumps(records, separators=(",", ":"))
287
+ if target_format == "ndjson":
288
+ return "".join(f"{json.dumps(record, separators=(',', ':'))}\n" for record in records)
289
+ return _write_csv(records)
290
+
291
+
292
+ def _write_csv(records: list[dict[str, JsonValue]]) -> str:
293
+ """Write a header of every key any row has, in first-seen order, and one row per record."""
294
+ header: list[str] = []
295
+ for record in records:
296
+ header.extend(key for key in record if key not in header)
297
+ out = io.StringIO(newline="")
298
+ writer = csv.writer(out, lineterminator="\n")
299
+ writer.writerow(header)
300
+ for record in records:
301
+ writer.writerow([_cell(record.get(key)) for key in header])
302
+ return out.getvalue()
303
+
304
+
305
+ def _cell(value: JsonValue) -> str:
306
+ """Render one value as csv text. Reading ruled the nested ones out already."""
307
+ if value is None:
308
+ return ""
309
+ if isinstance(value, str):
310
+ return value
311
+ return json.dumps(value)
File without changes