hypertabular 0.7.0__cp311-abi3-win_arm64.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- hypertabular/__init__.py +489 -0
- hypertabular/_native.pyd +0 -0
- hypertabular/_native.pyi +126 -0
- hypertabular/py.typed +0 -0
- hypertabular-0.7.0.dist-info/METADATA +133 -0
- hypertabular-0.7.0.dist-info/RECORD +9 -0
- hypertabular-0.7.0.dist-info/WHEEL +4 -0
- hypertabular-0.7.0.dist-info/licenses/LICENSE +21 -0
- hypertabular-0.7.0.dist-info/sboms/hypertabular.cyclonedx.json +747 -0
hypertabular/__init__.py
ADDED
|
@@ -0,0 +1,489 @@
|
|
|
1
|
+
"""Delimited text — CSV, TSV, any single-byte ASCII separator — and workbooks — XLSX and
|
|
2
|
+
ODS — read a batch at a time into typed columns, with a
|
|
3
|
+
`HyperCast <https://github.com/SkunkWerkx/HyperCast>`_ verdict for every cell.
|
|
4
|
+
|
|
5
|
+
The native core owns no memory and reads no files. It is linked straight into a CPython
|
|
6
|
+
extension module (``hypertabular._native``, PyO3), and the extension is the binding: it
|
|
7
|
+
allocates a value array and a verdict array per column, once, and the core fills them in
|
|
8
|
+
one call per batch, with the interpreter released. A column then reaches Python whole — a
|
|
9
|
+
read-only ``memoryview`` of the door's own item type over the batch's copy of that array,
|
|
10
|
+
which anything that takes a buffer takes — with no Python call per cell::
|
|
11
|
+
|
|
12
|
+
from hypertabular import Column, DelimitedReader, Dialect, Fault, Success
|
|
13
|
+
|
|
14
|
+
plan = [Column.i32(0), Column.text(1), Column.f64(2)]
|
|
15
|
+
with DelimitedReader.open("orders.csv", Dialect.CSV, plan) as reader:
|
|
16
|
+
for batch in reader:
|
|
17
|
+
ids, names, scores = batch.columns
|
|
18
|
+
|
|
19
|
+
# A column at a time, as the core wrote it…
|
|
20
|
+
total = sum(scores.values) # memoryview of C doubles
|
|
21
|
+
if scores.fault_count:
|
|
22
|
+
for row, fault in scores.faults():
|
|
23
|
+
print(row, fault.reason.name, scores.raw(row))
|
|
24
|
+
|
|
25
|
+
# …or a cell at a time, as HyperCast's union.
|
|
26
|
+
for row in range(batch.rows):
|
|
27
|
+
match scores[row]:
|
|
28
|
+
case Success(value):
|
|
29
|
+
print(names.values[row], value)
|
|
30
|
+
case Fault(reason, offset, length):
|
|
31
|
+
print(reason.name, "on line", batch.line(row), "in", scores.raw(row))
|
|
32
|
+
|
|
33
|
+
# A workbook reads into the same batch.
|
|
34
|
+
book = Workbook.open("orders.xlsx")
|
|
35
|
+
for batch in book.sheet("Orders", SheetOptions(), plan):
|
|
36
|
+
...
|
|
37
|
+
|
|
38
|
+
- **Nothing is sniffed.** The :class:`Dialect` states the separator, the quoting and the
|
|
39
|
+
header; :class:`SheetOptions` states a sheet's header and whether empty rows are skipped;
|
|
40
|
+
the plan states each column's door and, for numbers, its ``NumFormat``.
|
|
41
|
+
- **HyperCast is the judge.** :class:`Success`, :class:`Fault`, :class:`CastFailure`,
|
|
42
|
+
:class:`NumFormat`, :class:`UnixPrecision`, :class:`DateOrder` and :class:`ExcelEpoch`
|
|
43
|
+
are the ``hypercast`` package's own objects, re-exported here unchanged — not copies. A
|
|
44
|
+
text cell means exactly what ``hypercast.cast_*`` says of the same text, a typed workbook
|
|
45
|
+
cell is converted by the door directly, and its value is the
|
|
46
|
+
Python type that door gives: ``int``, ``float``, ``bool``, ``decimal.Decimal``,
|
|
47
|
+
``uuid.UUID``, ``datetime``, ``date``, ``time``, ``timedelta`` (and ``str`` for text).
|
|
48
|
+
- **A bad value is a verdict; a broken file is an exception.** A cell that does not cast
|
|
49
|
+
is a ``Fault`` in its column and the read goes on. A record of the wrong width, input
|
|
50
|
+
that ends inside a quoted cell, a workbook whose container or parts cannot be read,
|
|
51
|
+
raises :class:`TabularError` — after every intact row before it has been delivered.
|
|
52
|
+
- **A batch owns what it shows.** It stays valid after the reader has moved on.
|
|
53
|
+
"""
|
|
54
|
+
|
|
55
|
+
from __future__ import annotations
|
|
56
|
+
|
|
57
|
+
from dataclasses import dataclass
|
|
58
|
+
from enum import IntEnum
|
|
59
|
+
from typing import ClassVar
|
|
60
|
+
|
|
61
|
+
from hypercast import (
|
|
62
|
+
CastFailure,
|
|
63
|
+
DateOrder,
|
|
64
|
+
ExcelEpoch,
|
|
65
|
+
Fault,
|
|
66
|
+
NumFormat,
|
|
67
|
+
Success,
|
|
68
|
+
UnixPrecision,
|
|
69
|
+
Verdict,
|
|
70
|
+
)
|
|
71
|
+
|
|
72
|
+
from . import _native
|
|
73
|
+
|
|
74
|
+
#: Which backend this process loaded: always ``"native"``, the PyO3 extension that links the
|
|
75
|
+
#: Rust core straight into CPython — the only backend, and the one every wheel ships.
|
|
76
|
+
BACKEND: str = "native"
|
|
77
|
+
|
|
78
|
+
__all__ = [
|
|
79
|
+
"BACKEND",
|
|
80
|
+
"native_version",
|
|
81
|
+
"DelimitedReader",
|
|
82
|
+
"Workbook",
|
|
83
|
+
"WorkbookFormat",
|
|
84
|
+
"Sheet",
|
|
85
|
+
"SheetInfo",
|
|
86
|
+
"SheetOptions",
|
|
87
|
+
"Batch",
|
|
88
|
+
"ColumnData",
|
|
89
|
+
"Dialect",
|
|
90
|
+
"Column",
|
|
91
|
+
"Door",
|
|
92
|
+
"TabularError",
|
|
93
|
+
"TabularFailure",
|
|
94
|
+
# HyperCast's own, re-exported unchanged.
|
|
95
|
+
"CastFailure",
|
|
96
|
+
"Success",
|
|
97
|
+
"Fault",
|
|
98
|
+
"Verdict",
|
|
99
|
+
"NumFormat",
|
|
100
|
+
"UnixPrecision",
|
|
101
|
+
"DateOrder",
|
|
102
|
+
"ExcelEpoch",
|
|
103
|
+
]
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
class Door(IntEnum):
|
|
107
|
+
"""The door a column is cast through: HyperCast's, plus :data:`TEXT` for the bytes
|
|
108
|
+
themselves. The values are the core's own door codes."""
|
|
109
|
+
|
|
110
|
+
BOOL = 1
|
|
111
|
+
I8 = 2
|
|
112
|
+
I16 = 3
|
|
113
|
+
I32 = 4
|
|
114
|
+
I64 = 5
|
|
115
|
+
U8 = 6
|
|
116
|
+
U16 = 7
|
|
117
|
+
U32 = 8
|
|
118
|
+
U64 = 9
|
|
119
|
+
F32 = 10
|
|
120
|
+
F64 = 11
|
|
121
|
+
UUID = 12
|
|
122
|
+
TIMESTAMP = 13
|
|
123
|
+
UNIX = 14
|
|
124
|
+
DATE = 15
|
|
125
|
+
TIME = 16
|
|
126
|
+
DURATION = 17
|
|
127
|
+
TEXT = 18
|
|
128
|
+
DECIMAL = 19
|
|
129
|
+
DATE_ORDERED = 20
|
|
130
|
+
DATETIME = 21
|
|
131
|
+
EXCEL_SERIAL = 22
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
class TabularFailure(IntEnum):
|
|
135
|
+
"""Why an input could not be read as rows at all. The values are the core's codes."""
|
|
136
|
+
|
|
137
|
+
UNCLOSED_QUOTE = 1
|
|
138
|
+
"""The input ended inside a quoted cell."""
|
|
139
|
+
COLUMN_COUNT = 2
|
|
140
|
+
"""A record's cell count disagrees with the first record's."""
|
|
141
|
+
ROW_TOO_LONG = 3
|
|
142
|
+
"""A single record is larger than :data:`DelimitedReader.MAX_ROW_BYTES`."""
|
|
143
|
+
NOT_A_ZIP = 16
|
|
144
|
+
"""The workbook's container is not a zip file."""
|
|
145
|
+
CONTAINER = 17
|
|
146
|
+
"""The zip's own structure is broken."""
|
|
147
|
+
ENCRYPTED = 18
|
|
148
|
+
"""The workbook is encrypted."""
|
|
149
|
+
METHOD = 19
|
|
150
|
+
"""A part is compressed by a method other than stored or deflate."""
|
|
151
|
+
MISSING_PART = 20
|
|
152
|
+
"""A part the workbook cannot be read without is missing."""
|
|
153
|
+
XML = 21
|
|
154
|
+
"""A part's XML ends inside a construct."""
|
|
155
|
+
DEFLATE = 22
|
|
156
|
+
"""A part's bytes are not a deflate stream, or stop before the stream does."""
|
|
157
|
+
NOT_A_WORKBOOK = 23
|
|
158
|
+
"""The zip is neither an XLSX nor an ODS workbook."""
|
|
159
|
+
SHARED_STRING = 24
|
|
160
|
+
"""A cell names a shared string the table does not have."""
|
|
161
|
+
TOO_LARGE = 25
|
|
162
|
+
"""More text than can be addressed: over 4 GiB in a batch or in the shared strings, or
|
|
163
|
+
2 GiB in a cell."""
|
|
164
|
+
|
|
165
|
+
|
|
166
|
+
class WorkbookFormat(IntEnum):
|
|
167
|
+
"""Which kind of workbook a :class:`Workbook` is."""
|
|
168
|
+
|
|
169
|
+
XLSX = 1
|
|
170
|
+
"""Office Open XML: ``.xlsx``, ``.xlsm``."""
|
|
171
|
+
ODS = 2
|
|
172
|
+
"""OpenDocument: ``.ods``."""
|
|
173
|
+
|
|
174
|
+
|
|
175
|
+
class TabularError(Exception):
|
|
176
|
+
"""A structural failure: the input is not rows of cells — a record of the wrong width,
|
|
177
|
+
input that ends inside a quoted cell, a workbook whose container or parts cannot be
|
|
178
|
+
read. Never a cell's verdict: a value that does not cast is a ``Fault`` in its column,
|
|
179
|
+
and the read goes on. A structural failure ends the input, after every intact row
|
|
180
|
+
before it has been delivered.
|
|
181
|
+
|
|
182
|
+
For a workbook, :attr:`record` is the part the failure is in, :attr:`line` the sheet
|
|
183
|
+
row and :attr:`byte` the offset within the part's inflated bytes.
|
|
184
|
+
"""
|
|
185
|
+
|
|
186
|
+
kind: TabularFailure
|
|
187
|
+
"""What is wrong."""
|
|
188
|
+
record: int
|
|
189
|
+
"""Zero-based index of the offending record — the header and skipped blank lines included."""
|
|
190
|
+
line: int
|
|
191
|
+
"""One-based line the offending record starts on."""
|
|
192
|
+
byte: int
|
|
193
|
+
"""Absolute byte offset of the offending record's start."""
|
|
194
|
+
expected: int
|
|
195
|
+
"""Cells in the first record, for :data:`TabularFailure.COLUMN_COUNT`."""
|
|
196
|
+
found: int
|
|
197
|
+
"""Cells in this record, for :data:`TabularFailure.COLUMN_COUNT`."""
|
|
198
|
+
|
|
199
|
+
def __init__(
|
|
200
|
+
self,
|
|
201
|
+
kind: TabularFailure,
|
|
202
|
+
record: int,
|
|
203
|
+
line: int,
|
|
204
|
+
byte: int,
|
|
205
|
+
expected: int = 0,
|
|
206
|
+
found: int = 0,
|
|
207
|
+
) -> None:
|
|
208
|
+
"""Builds the failure the reader raises; the arguments are its attributes."""
|
|
209
|
+
# The arguments themselves, not the message, so the exception copies and pickles.
|
|
210
|
+
super().__init__(kind, record, line, byte, expected, found)
|
|
211
|
+
self.kind = TabularFailure(kind)
|
|
212
|
+
self.record = record
|
|
213
|
+
self.line = line
|
|
214
|
+
self.byte = byte
|
|
215
|
+
self.expected = expected
|
|
216
|
+
self.found = found
|
|
217
|
+
|
|
218
|
+
def __str__(self) -> str:
|
|
219
|
+
"""The failure in words, with its position."""
|
|
220
|
+
where = f"record {self.record} (line {self.line}, byte {self.byte})"
|
|
221
|
+
part = f"part {self.record} of the workbook"
|
|
222
|
+
match self.kind:
|
|
223
|
+
case TabularFailure.COLUMN_COUNT:
|
|
224
|
+
return f"{where} has {self.found} cells; the first record had {self.expected}"
|
|
225
|
+
case TabularFailure.UNCLOSED_QUOTE:
|
|
226
|
+
return f"the input ended inside a quoted cell in {where}"
|
|
227
|
+
case TabularFailure.ROW_TOO_LONG:
|
|
228
|
+
limit = _native.DelimitedReader.MAX_ROW_BYTES
|
|
229
|
+
return f"{where} exceeds the {limit}-byte row ceiling"
|
|
230
|
+
case TabularFailure.NOT_A_ZIP:
|
|
231
|
+
return "the workbook is not a zip file"
|
|
232
|
+
case TabularFailure.ENCRYPTED:
|
|
233
|
+
return "the workbook is encrypted"
|
|
234
|
+
case TabularFailure.METHOD:
|
|
235
|
+
return f"{part} is compressed by method {self.found}, neither stored nor deflate"
|
|
236
|
+
case TabularFailure.MISSING_PART:
|
|
237
|
+
return f"part {self.record}, which the workbook cannot be read without, is missing"
|
|
238
|
+
case TabularFailure.XML:
|
|
239
|
+
return f"{part} ends inside an XML construct (byte {self.byte})"
|
|
240
|
+
case TabularFailure.DEFLATE:
|
|
241
|
+
return f"{part} is not a whole deflate stream (byte {self.byte})"
|
|
242
|
+
case TabularFailure.NOT_A_WORKBOOK:
|
|
243
|
+
return "the zip is neither an XLSX nor an ODS workbook"
|
|
244
|
+
case TabularFailure.SHARED_STRING:
|
|
245
|
+
return f"row {self.line} names shared string {self.found}; the table has {self.expected}"
|
|
246
|
+
case TabularFailure.TOO_LARGE:
|
|
247
|
+
return "the workbook holds more text than a batch can address"
|
|
248
|
+
case _:
|
|
249
|
+
return "the workbook's zip structure is broken"
|
|
250
|
+
|
|
251
|
+
|
|
252
|
+
@dataclass(frozen=True, slots=True)
|
|
253
|
+
class Dialect:
|
|
254
|
+
"""How the text is delimited, declared by the caller. Nothing is sniffed: the separator
|
|
255
|
+
is stated, quoting is stated, the header is stated — the same stance HyperCast's
|
|
256
|
+
``NumFormat`` takes for numeric notation.
|
|
257
|
+
"""
|
|
258
|
+
|
|
259
|
+
separator: str
|
|
260
|
+
"""The single-byte separator: tab, or any printable ASCII character except ``"``."""
|
|
261
|
+
quoting: bool = True
|
|
262
|
+
"""Whether ``"`` quotes cells (RFC 4180, ``""`` for a literal quote). Off, a quote is
|
|
263
|
+
an ordinary byte."""
|
|
264
|
+
has_header: bool = True
|
|
265
|
+
"""Whether the first record is a header, exposed through
|
|
266
|
+
:attr:`DelimitedReader.header` and never delivered as a row."""
|
|
267
|
+
skip_blank_lines: bool = True
|
|
268
|
+
"""Whether a completely empty line is skipped rather than read as a one-cell row."""
|
|
269
|
+
|
|
270
|
+
CSV: ClassVar[Dialect]
|
|
271
|
+
"""Comma-separated, quoted, with a header, blank lines skipped."""
|
|
272
|
+
TSV: ClassVar[Dialect]
|
|
273
|
+
"""Tab-separated, otherwise as :data:`CSV`."""
|
|
274
|
+
PSV: ClassVar[Dialect]
|
|
275
|
+
"""Pipe-separated, otherwise as :data:`CSV`."""
|
|
276
|
+
|
|
277
|
+
def __post_init__(self) -> None:
|
|
278
|
+
"""Refuses a separator the scanner cannot honour — a caller bug, not a data verdict."""
|
|
279
|
+
if not isinstance(self.separator, str):
|
|
280
|
+
raise TypeError(f"separator must be a str, not {type(self.separator).__name__}")
|
|
281
|
+
valid = len(self.separator) == 1 and (
|
|
282
|
+
self.separator == "\t" or (" " <= self.separator <= "~" and self.separator != '"')
|
|
283
|
+
)
|
|
284
|
+
if not valid:
|
|
285
|
+
raise ValueError(
|
|
286
|
+
f"separator {self.separator!r} is not tab or a printable ASCII character other than '\"'"
|
|
287
|
+
)
|
|
288
|
+
|
|
289
|
+
|
|
290
|
+
Dialect.CSV = Dialect(",")
|
|
291
|
+
Dialect.TSV = Dialect("\t")
|
|
292
|
+
Dialect.PSV = Dialect("|")
|
|
293
|
+
|
|
294
|
+
|
|
295
|
+
@dataclass(frozen=True, slots=True, repr=False)
|
|
296
|
+
class Column:
|
|
297
|
+
"""One output column of a plan: which source column it reads, the door it casts
|
|
298
|
+
through, and — for the numeric doors — the notation. Build one with the factory named
|
|
299
|
+
for its door (the names are HyperCast's: ``Column.i32`` is what ``hypercast.cast_i32``
|
|
300
|
+
would make of each cell).
|
|
301
|
+
|
|
302
|
+
A plan is a projection: a forty-column file can be read into five typed columns, in
|
|
303
|
+
any order, and a source column can be read through more than one door.
|
|
304
|
+
"""
|
|
305
|
+
|
|
306
|
+
ordinal: int
|
|
307
|
+
"""Zero-based ordinal of the source column. Past a record's last cell reads as empty."""
|
|
308
|
+
door: Door
|
|
309
|
+
"""The door."""
|
|
310
|
+
format: NumFormat = NumFormat.INVARIANT
|
|
311
|
+
"""The numeric notation, read by the numeric doors."""
|
|
312
|
+
declared: UnixPrecision | DateOrder | ExcelEpoch | None = None
|
|
313
|
+
"""What the door declares beside itself: the Unix precision, the date order, or the
|
|
314
|
+
Excel date system — never guessed."""
|
|
315
|
+
|
|
316
|
+
def __post_init__(self) -> None:
|
|
317
|
+
"""Refuses an ordinal no source column can have."""
|
|
318
|
+
if not isinstance(self.ordinal, int) or isinstance(self.ordinal, bool):
|
|
319
|
+
raise TypeError(f"ordinal must be an int, not {type(self.ordinal).__name__}")
|
|
320
|
+
if not 0 <= self.ordinal < 2**31:
|
|
321
|
+
raise ValueError(f"ordinal must be between 0 and 2**31 - 1, not {self.ordinal}")
|
|
322
|
+
|
|
323
|
+
def __repr__(self) -> str:
|
|
324
|
+
"""The column as the factory call that makes it."""
|
|
325
|
+
said = [str(self.ordinal)]
|
|
326
|
+
if self.declared is not None:
|
|
327
|
+
said.append(f"{type(self.declared).__name__}.{self.declared.name}")
|
|
328
|
+
if self.format is not NumFormat.INVARIANT:
|
|
329
|
+
said.append(repr(self.format))
|
|
330
|
+
return f"Column.{self.door.name.lower()}({', '.join(said)})"
|
|
331
|
+
|
|
332
|
+
@classmethod
|
|
333
|
+
def _numeric(cls, ordinal: int, door: Door, fmt: NumFormat) -> Column:
|
|
334
|
+
if not isinstance(fmt, NumFormat):
|
|
335
|
+
raise TypeError(f"fmt must be a hypercast.NumFormat, not {type(fmt).__name__}")
|
|
336
|
+
return cls(ordinal, door, fmt)
|
|
337
|
+
|
|
338
|
+
@classmethod
|
|
339
|
+
def bool(cls, ordinal: int) -> Column:
|
|
340
|
+
"""A ``bool`` column: HyperCast's boolean lexicon."""
|
|
341
|
+
return cls(ordinal, Door.BOOL)
|
|
342
|
+
|
|
343
|
+
@classmethod
|
|
344
|
+
def i8(cls, ordinal: int, fmt: NumFormat = NumFormat.INVARIANT) -> Column:
|
|
345
|
+
"""A signed 8-bit integer column under the declared notation."""
|
|
346
|
+
return cls._numeric(ordinal, Door.I8, fmt)
|
|
347
|
+
|
|
348
|
+
@classmethod
|
|
349
|
+
def i16(cls, ordinal: int, fmt: NumFormat = NumFormat.INVARIANT) -> Column:
|
|
350
|
+
"""A signed 16-bit integer column under the declared notation."""
|
|
351
|
+
return cls._numeric(ordinal, Door.I16, fmt)
|
|
352
|
+
|
|
353
|
+
@classmethod
|
|
354
|
+
def i32(cls, ordinal: int, fmt: NumFormat = NumFormat.INVARIANT) -> Column:
|
|
355
|
+
"""A signed 32-bit integer column under the declared notation."""
|
|
356
|
+
return cls._numeric(ordinal, Door.I32, fmt)
|
|
357
|
+
|
|
358
|
+
@classmethod
|
|
359
|
+
def i64(cls, ordinal: int, fmt: NumFormat = NumFormat.INVARIANT) -> Column:
|
|
360
|
+
"""A signed 64-bit integer column under the declared notation."""
|
|
361
|
+
return cls._numeric(ordinal, Door.I64, fmt)
|
|
362
|
+
|
|
363
|
+
@classmethod
|
|
364
|
+
def u8(cls, ordinal: int, fmt: NumFormat = NumFormat.INVARIANT) -> Column:
|
|
365
|
+
"""An unsigned 8-bit integer column under the declared notation."""
|
|
366
|
+
return cls._numeric(ordinal, Door.U8, fmt)
|
|
367
|
+
|
|
368
|
+
@classmethod
|
|
369
|
+
def u16(cls, ordinal: int, fmt: NumFormat = NumFormat.INVARIANT) -> Column:
|
|
370
|
+
"""An unsigned 16-bit integer column under the declared notation."""
|
|
371
|
+
return cls._numeric(ordinal, Door.U16, fmt)
|
|
372
|
+
|
|
373
|
+
@classmethod
|
|
374
|
+
def u32(cls, ordinal: int, fmt: NumFormat = NumFormat.INVARIANT) -> Column:
|
|
375
|
+
"""An unsigned 32-bit integer column under the declared notation."""
|
|
376
|
+
return cls._numeric(ordinal, Door.U32, fmt)
|
|
377
|
+
|
|
378
|
+
@classmethod
|
|
379
|
+
def u64(cls, ordinal: int, fmt: NumFormat = NumFormat.INVARIANT) -> Column:
|
|
380
|
+
"""An unsigned 64-bit integer column under the declared notation."""
|
|
381
|
+
return cls._numeric(ordinal, Door.U64, fmt)
|
|
382
|
+
|
|
383
|
+
@classmethod
|
|
384
|
+
def f32(cls, ordinal: int, fmt: NumFormat = NumFormat.INVARIANT) -> Column:
|
|
385
|
+
"""An IEEE single column under the declared notation."""
|
|
386
|
+
return cls._numeric(ordinal, Door.F32, fmt)
|
|
387
|
+
|
|
388
|
+
@classmethod
|
|
389
|
+
def f64(cls, ordinal: int, fmt: NumFormat = NumFormat.INVARIANT) -> Column:
|
|
390
|
+
"""An IEEE double column under the declared notation."""
|
|
391
|
+
return cls._numeric(ordinal, Door.F64, fmt)
|
|
392
|
+
|
|
393
|
+
@classmethod
|
|
394
|
+
def decimal(cls, ordinal: int, fmt: NumFormat = NumFormat.INVARIANT) -> Column:
|
|
395
|
+
"""An exact ``decimal.Decimal`` column under the declared notation; no float is
|
|
396
|
+
ever formed."""
|
|
397
|
+
return cls._numeric(ordinal, Door.DECIMAL, fmt)
|
|
398
|
+
|
|
399
|
+
@classmethod
|
|
400
|
+
def uuid(cls, ordinal: int) -> Column:
|
|
401
|
+
"""A ``uuid.UUID`` column."""
|
|
402
|
+
return cls(ordinal, Door.UUID)
|
|
403
|
+
|
|
404
|
+
@classmethod
|
|
405
|
+
def timestamp(cls, ordinal: int) -> Column:
|
|
406
|
+
"""An RFC 3339 instant column, as an aware UTC ``datetime``."""
|
|
407
|
+
return cls(ordinal, Door.TIMESTAMP)
|
|
408
|
+
|
|
409
|
+
@classmethod
|
|
410
|
+
def unix(cls, ordinal: int, precision: UnixPrecision) -> Column:
|
|
411
|
+
"""A Unix-epoch column at the declared precision — never guessed from magnitude —
|
|
412
|
+
as an aware UTC ``datetime``."""
|
|
413
|
+
return cls(ordinal, Door.UNIX, declared=UnixPrecision(precision))
|
|
414
|
+
|
|
415
|
+
@classmethod
|
|
416
|
+
def excel_serial(cls, ordinal: int, epoch: ExcelEpoch) -> Column:
|
|
417
|
+
"""An Excel date-serial column under the declared date system, as an aware UTC
|
|
418
|
+
``datetime``."""
|
|
419
|
+
return cls(ordinal, Door.EXCEL_SERIAL, declared=ExcelEpoch(epoch))
|
|
420
|
+
|
|
421
|
+
@classmethod
|
|
422
|
+
def date(cls, ordinal: int, order: DateOrder | None = None) -> Column:
|
|
423
|
+
"""A ``date`` column: strict ISO ``yyyy-MM-dd`` with no order, the separated forms
|
|
424
|
+
under a declared one (which is :meth:`date_ordered`) — as ``hypercast.cast_date``."""
|
|
425
|
+
if order is not None:
|
|
426
|
+
return cls.date_ordered(ordinal, order)
|
|
427
|
+
return cls(ordinal, Door.DATE)
|
|
428
|
+
|
|
429
|
+
@classmethod
|
|
430
|
+
def date_ordered(cls, ordinal: int, order: DateOrder) -> Column:
|
|
431
|
+
"""A separated-date column under the declared field order, as a ``date``."""
|
|
432
|
+
return cls(ordinal, Door.DATE_ORDERED, declared=DateOrder(order))
|
|
433
|
+
|
|
434
|
+
@classmethod
|
|
435
|
+
def datetime(cls, ordinal: int, order: DateOrder) -> Column:
|
|
436
|
+
"""A zone-less civil date-time column under the declared field order, as a naive
|
|
437
|
+
``datetime`` — the text named no zone and none is invented."""
|
|
438
|
+
return cls(ordinal, Door.DATETIME, declared=DateOrder(order))
|
|
439
|
+
|
|
440
|
+
@classmethod
|
|
441
|
+
def time(cls, ordinal: int) -> Column:
|
|
442
|
+
"""A 24-hour time-of-day column, as a ``time``."""
|
|
443
|
+
return cls(ordinal, Door.TIME)
|
|
444
|
+
|
|
445
|
+
@classmethod
|
|
446
|
+
def duration(cls, ordinal: int) -> Column:
|
|
447
|
+
"""A duration column, as a ``timedelta``."""
|
|
448
|
+
return cls(ordinal, Door.DURATION)
|
|
449
|
+
|
|
450
|
+
@classmethod
|
|
451
|
+
def text(cls, ordinal: int) -> Column:
|
|
452
|
+
"""A text column: the cell's bytes themselves, untrimmed, as a ``str``. A cell
|
|
453
|
+
with no bytes at all is the one way text fails (``CastFailure.EMPTY``)."""
|
|
454
|
+
return cls(ordinal, Door.TEXT)
|
|
455
|
+
|
|
456
|
+
|
|
457
|
+
@dataclass(frozen=True, slots=True)
|
|
458
|
+
class SheetOptions:
|
|
459
|
+
"""How a sheet of a :class:`Workbook` is read."""
|
|
460
|
+
|
|
461
|
+
has_header: bool = True
|
|
462
|
+
"""Whether the sheet's first row is a header, exposed through :attr:`Sheet.header` and
|
|
463
|
+
never delivered as a row."""
|
|
464
|
+
skip_empty_rows: bool = True
|
|
465
|
+
"""Whether a row with no cells is skipped rather than delivered as a row of empty
|
|
466
|
+
cells."""
|
|
467
|
+
batch_rows: int = 4096
|
|
468
|
+
"""Rows per batch, at most."""
|
|
469
|
+
|
|
470
|
+
def __post_init__(self) -> None:
|
|
471
|
+
"""Refuses a batch size no batch can have."""
|
|
472
|
+
if not isinstance(self.batch_rows, int) or isinstance(self.batch_rows, bool):
|
|
473
|
+
raise TypeError(f"batch_rows must be an int, not {type(self.batch_rows).__name__}")
|
|
474
|
+
if self.batch_rows < 1:
|
|
475
|
+
raise ValueError(f"batch_rows must be at least 1, not {self.batch_rows}")
|
|
476
|
+
|
|
477
|
+
|
|
478
|
+
# The extension's own classes are the package surface. _bind hands it this package's plan
|
|
479
|
+
# column, dialect, structural failure and workbook format; it imports HyperCast's verdict
|
|
480
|
+
# types itself.
|
|
481
|
+
_native._bind(Column, Dialect, TabularError, TabularFailure, WorkbookFormat)
|
|
482
|
+
|
|
483
|
+
DelimitedReader = _native.DelimitedReader
|
|
484
|
+
Workbook = _native.Workbook
|
|
485
|
+
Sheet = _native.Sheet
|
|
486
|
+
SheetInfo = _native.SheetInfo
|
|
487
|
+
Batch = _native.Batch
|
|
488
|
+
ColumnData = _native.ColumnData
|
|
489
|
+
native_version = _native.native_version
|
hypertabular/_native.pyd
ADDED
|
Binary file
|
hypertabular/_native.pyi
ADDED
|
@@ -0,0 +1,126 @@
|
|
|
1
|
+
"""The typed surface of ``hypertabular._native``, the PyO3 extension (``tests/test_typing.py``
|
|
2
|
+
holds the loaded module to this file). ``hypertabular`` re-exports all of it directly, so
|
|
3
|
+
this is what a type checker sees behind a reader, a batch and a column — and behind a cell,
|
|
4
|
+
which is HyperCast's own ``Success[...] | Fault``, so a ``match`` over the two cases can be
|
|
5
|
+
checked for exhaustiveness with ``typing.assert_never``.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from collections.abc import Iterable, Iterator, Sequence
|
|
9
|
+
from os import PathLike
|
|
10
|
+
from typing import Any, ClassVar, Protocol, final
|
|
11
|
+
|
|
12
|
+
from hypercast import ExcelEpoch, Fault, Success
|
|
13
|
+
|
|
14
|
+
from . import Column, Dialect, SheetOptions, TabularError, TabularFailure, WorkbookFormat
|
|
15
|
+
|
|
16
|
+
class _Readable(Protocol):
|
|
17
|
+
def read(self, size: int, /) -> bytes: ...
|
|
18
|
+
|
|
19
|
+
@final
|
|
20
|
+
class ColumnData:
|
|
21
|
+
@property
|
|
22
|
+
def column(self) -> Column: ...
|
|
23
|
+
@property
|
|
24
|
+
def values(self) -> Sequence[Any]: ...
|
|
25
|
+
@property
|
|
26
|
+
def verdicts(self) -> memoryview: ...
|
|
27
|
+
@property
|
|
28
|
+
def fault_count(self) -> int: ...
|
|
29
|
+
def faults(self) -> list[tuple[int, Fault]]: ...
|
|
30
|
+
def raw(self, row: int) -> bytes: ...
|
|
31
|
+
def __len__(self) -> int: ...
|
|
32
|
+
def __getitem__(self, row: int) -> Success[Any] | Fault: ...
|
|
33
|
+
def __iter__(self) -> Iterator[Success[Any] | Fault]: ...
|
|
34
|
+
|
|
35
|
+
@final
|
|
36
|
+
class Batch:
|
|
37
|
+
@property
|
|
38
|
+
def rows(self) -> int: ...
|
|
39
|
+
@property
|
|
40
|
+
def columns(self) -> tuple[ColumnData, ...]: ...
|
|
41
|
+
def column(self, index: int) -> ColumnData: ...
|
|
42
|
+
def line(self, row: int) -> int: ...
|
|
43
|
+
def raw(self, column: int, row: int) -> bytes: ...
|
|
44
|
+
def __len__(self) -> int: ...
|
|
45
|
+
|
|
46
|
+
@final
|
|
47
|
+
class DelimitedReader:
|
|
48
|
+
DEFAULT_BATCH_ROWS: ClassVar[int]
|
|
49
|
+
DEFAULT_BUFFER_BYTES: ClassVar[int]
|
|
50
|
+
MAX_ROW_BYTES: ClassVar[int]
|
|
51
|
+
def __new__(
|
|
52
|
+
cls,
|
|
53
|
+
source: bytes | _Readable,
|
|
54
|
+
dialect: Dialect,
|
|
55
|
+
plan: Iterable[Column],
|
|
56
|
+
*,
|
|
57
|
+
batch_rows: int = 4096,
|
|
58
|
+
buffer_bytes: int = 262144,
|
|
59
|
+
) -> DelimitedReader: ...
|
|
60
|
+
@staticmethod
|
|
61
|
+
def open(
|
|
62
|
+
path: str | PathLike[str],
|
|
63
|
+
dialect: Dialect,
|
|
64
|
+
plan: Iterable[Column],
|
|
65
|
+
*,
|
|
66
|
+
batch_rows: int = 4096,
|
|
67
|
+
buffer_bytes: int = 262144,
|
|
68
|
+
) -> DelimitedReader: ...
|
|
69
|
+
@property
|
|
70
|
+
def header(self) -> tuple[str, ...] | None: ...
|
|
71
|
+
@property
|
|
72
|
+
def dialect(self) -> Dialect: ...
|
|
73
|
+
@property
|
|
74
|
+
def plan(self) -> tuple[Column, ...]: ...
|
|
75
|
+
@property
|
|
76
|
+
def batch_rows(self) -> int: ...
|
|
77
|
+
@property
|
|
78
|
+
def records(self) -> int: ...
|
|
79
|
+
def read(self) -> Batch | None: ...
|
|
80
|
+
def close(self) -> None: ...
|
|
81
|
+
def __iter__(self) -> DelimitedReader: ...
|
|
82
|
+
def __next__(self) -> Batch: ...
|
|
83
|
+
def __enter__(self) -> DelimitedReader: ...
|
|
84
|
+
def __exit__(self, *_exc: object) -> None: ...
|
|
85
|
+
|
|
86
|
+
@final
|
|
87
|
+
class SheetInfo:
|
|
88
|
+
@property
|
|
89
|
+
def name(self) -> str: ...
|
|
90
|
+
@property
|
|
91
|
+
def hidden(self) -> bool: ...
|
|
92
|
+
|
|
93
|
+
@final
|
|
94
|
+
class Workbook:
|
|
95
|
+
def __new__(cls, data: bytes) -> Workbook: ...
|
|
96
|
+
@staticmethod
|
|
97
|
+
def open(path: str | PathLike[str]) -> Workbook: ...
|
|
98
|
+
@property
|
|
99
|
+
def format(self) -> WorkbookFormat: ...
|
|
100
|
+
@property
|
|
101
|
+
def date_system(self) -> ExcelEpoch: ...
|
|
102
|
+
@property
|
|
103
|
+
def sheets(self) -> tuple[SheetInfo, ...]: ...
|
|
104
|
+
def sheet(self, which: int | str, options: SheetOptions, plan: Iterable[Column]) -> Sheet: ...
|
|
105
|
+
|
|
106
|
+
@final
|
|
107
|
+
class Sheet:
|
|
108
|
+
@property
|
|
109
|
+
def options(self) -> SheetOptions: ...
|
|
110
|
+
@property
|
|
111
|
+
def plan(self) -> tuple[Column, ...]: ...
|
|
112
|
+
@property
|
|
113
|
+
def header(self) -> tuple[str, ...] | None: ...
|
|
114
|
+
def read(self) -> Batch | None: ...
|
|
115
|
+
def __iter__(self) -> Sheet: ...
|
|
116
|
+
def __next__(self) -> Batch: ...
|
|
117
|
+
|
|
118
|
+
def native_version() -> str: ...
|
|
119
|
+
def _bind(
|
|
120
|
+
column: type[Column],
|
|
121
|
+
dialect: type[Dialect],
|
|
122
|
+
error: type[TabularError],
|
|
123
|
+
failure: type[TabularFailure],
|
|
124
|
+
workbook_format: type[WorkbookFormat],
|
|
125
|
+
) -> None: ...
|
|
126
|
+
def _limits(reader: DelimitedReader, max_row_bytes: int, window_bytes: int) -> None: ...
|
hypertabular/py.typed
ADDED
|
File without changes
|