polypress 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- polypress/__init__.py +13 -0
- polypress/caccel.py +164 -0
- polypress/cli.py +255 -0
- polypress/codec.py +343 -0
- polypress/dtz.py +978 -0
- polypress/fast.py +1397 -0
- polypress/stream.py +438 -0
- polypress/tcz.c +207 -0
- polypress/turbo.py +1515 -0
- polypress-0.2.0.dist-info/METADATA +121 -0
- polypress-0.2.0.dist-info/RECORD +15 -0
- polypress-0.2.0.dist-info/WHEEL +5 -0
- polypress-0.2.0.dist-info/entry_points.txt +2 -0
- polypress-0.2.0.dist-info/licenses/LICENSE +21 -0
- polypress-0.2.0.dist-info/top_level.txt +1 -0
polypress/stream.py
ADDED
|
@@ -0,0 +1,438 @@
|
|
|
1
|
+
"""Bounded-memory compression for files larger than RAM.
|
|
2
|
+
|
|
3
|
+
The single-shot codec in fast.py holds the whole table as Python strings.
|
|
4
|
+
Measured expansion is about 8x -- a 291 MB CSV peaks at 2.4 GB -- so a 50 GB
|
|
5
|
+
file would need hundreds of gigabytes and simply will not run.
|
|
6
|
+
|
|
7
|
+
This splits the table into row blocks, compresses each independently, and
|
|
8
|
+
concatenates them behind an index. Peak memory is one block, not one file,
|
|
9
|
+
and it is settable. Blocks are independent, so this is also the thing that
|
|
10
|
+
makes parallelism possible later.
|
|
11
|
+
|
|
12
|
+
The cost is real and worth stating: the cross-column reordering only sees
|
|
13
|
+
correlations *inside* a block, so smaller blocks compress slightly worse.
|
|
14
|
+
Measure with `--rows` before choosing a small one.
|
|
15
|
+
|
|
16
|
+
python3 tzip.py stream-compress big.csv -o big.ppz --budget 1.0
|
|
17
|
+
python3 tzip.py stream-restore big.ppz -o back.csv
|
|
18
|
+
python3 tzip.py stream-info big.ppz
|
|
19
|
+
|
|
20
|
+
Or as a module, which is the same code path:
|
|
21
|
+
|
|
22
|
+
python3 -m polypress.stream compress big.csv --budget 1.0
|
|
23
|
+
"""
|
|
24
|
+
|
|
25
|
+
from __future__ import annotations
|
|
26
|
+
|
|
27
|
+
import argparse
|
|
28
|
+
import csv
|
|
29
|
+
import json
|
|
30
|
+
import lzma
|
|
31
|
+
import os
|
|
32
|
+
import sys
|
|
33
|
+
import time
|
|
34
|
+
from typing import Iterator, List, Optional
|
|
35
|
+
|
|
36
|
+
sys.path.insert(0, os.path.dirname(os.path.dirname(
|
|
37
|
+
os.path.abspath(__file__))))
|
|
38
|
+
|
|
39
|
+
from . import dtz
|
|
40
|
+
from . import fast
|
|
41
|
+
|
|
42
|
+
MAGIC = b"PPZS"
|
|
43
|
+
DEFAULT_BUDGET_GB = 1.0
|
|
44
|
+
# Python strings cost roughly this much more than the CSV bytes they came
|
|
45
|
+
# from; measured at 8.1x on NHAMCS (291 MB file -> 2.36 GB resident).
|
|
46
|
+
EXPANSION = 8.5
|
|
47
|
+
# Verification decodes the block and compares, so the original block and the
|
|
48
|
+
# decoded copy are both resident at once. Without this the budget lied by
|
|
49
|
+
# roughly 2x: asking for 1 GB gave a 2.07 GB peak.
|
|
50
|
+
VERIFY_FACTOR = 2.2
|
|
51
|
+
MIN_ROWS = 1000
|
|
52
|
+
MAX_ROWS = 2_000_000
|
|
53
|
+
|
|
54
|
+
csv.field_size_limit(1 << 31)
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
def _decode_guard(rows, path: str, enc: str):
|
|
58
|
+
"""Turn a lazy decode failure into the same refusal dtz raises.
|
|
59
|
+
|
|
60
|
+
The single-shot reader decodes the whole file inside one call, so a bad
|
|
61
|
+
byte surfaces there. Here the file is read block by block, so it surfaces
|
|
62
|
+
hundreds of rows in -- possibly after several blocks have already been
|
|
63
|
+
written. Same error, same message, wherever it lands."""
|
|
64
|
+
try:
|
|
65
|
+
for row in rows:
|
|
66
|
+
yield row
|
|
67
|
+
except UnicodeDecodeError as exc:
|
|
68
|
+
raise dtz.EncodingRefused(
|
|
69
|
+
dtz.encoding_error(path, exc, enc)) from None
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def _reader(path: str, encoding: Optional[str] = None):
|
|
73
|
+
ext = os.path.splitext(path)[1].lower()
|
|
74
|
+
enc = encoding or dtz.sniff_encoding(path)
|
|
75
|
+
fh = dtz.open_text(path, enc)
|
|
76
|
+
try:
|
|
77
|
+
delimiter = dtz.EXT_DELIMITER.get(ext)
|
|
78
|
+
if delimiter is None:
|
|
79
|
+
delimiter = dtz._sniff_delimiter(fh.read(65536))
|
|
80
|
+
fh.seek(0)
|
|
81
|
+
except UnicodeDecodeError as exc:
|
|
82
|
+
fh.close()
|
|
83
|
+
raise dtz.EncodingRefused(
|
|
84
|
+
dtz.encoding_error(path, exc, enc)) from None
|
|
85
|
+
return fh, _decode_guard(csv.reader(fh, delimiter=delimiter), path, enc)
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
def plan_rows(path: str, budget_gb: float, verify: bool = True,
|
|
89
|
+
encoding: Optional[str] = None) -> tuple:
|
|
90
|
+
"""Rows per block that keep peak memory near the budget.
|
|
91
|
+
|
|
92
|
+
Estimated from the real average row width of the file rather than a
|
|
93
|
+
guess, because a 209-column survey row and a 5-column sensor row differ
|
|
94
|
+
by two orders of magnitude."""
|
|
95
|
+
size = os.path.getsize(path)
|
|
96
|
+
fh, rd = _reader(path, encoding)
|
|
97
|
+
try:
|
|
98
|
+
header = next(rd, [])
|
|
99
|
+
n, seen = 0, 0
|
|
100
|
+
for row in rd:
|
|
101
|
+
seen += sum(len(c) for c in row) + len(row)
|
|
102
|
+
n += 1
|
|
103
|
+
if n >= 2000:
|
|
104
|
+
break
|
|
105
|
+
finally:
|
|
106
|
+
fh.close()
|
|
107
|
+
per_row = max(1.0, seen / max(n, 1))
|
|
108
|
+
factor = EXPANSION * (VERIFY_FACTOR if verify else 1.0)
|
|
109
|
+
rows = int((budget_gb * (1 << 30)) / (per_row * factor))
|
|
110
|
+
rows = max(MIN_ROWS, min(MAX_ROWS, rows))
|
|
111
|
+
est_blocks = max(1, int(size / (per_row * rows)) + 1)
|
|
112
|
+
return rows, per_row, est_blocks, header
|
|
113
|
+
|
|
114
|
+
|
|
115
|
+
def _blocks(path: str, rows_per_block: int,
|
|
116
|
+
encoding: Optional[str] = None) -> Iterator[dtz.Table]:
|
|
117
|
+
fh, rd = _reader(path, encoding)
|
|
118
|
+
try:
|
|
119
|
+
header = next(rd, None)
|
|
120
|
+
if header is None:
|
|
121
|
+
return
|
|
122
|
+
width = len(header)
|
|
123
|
+
batch: List[List[str]] = []
|
|
124
|
+
yielded = False
|
|
125
|
+
for row in rd:
|
|
126
|
+
if len(row) != width: # ragged rows, same rule as dtz
|
|
127
|
+
row = (row + [""] * width)[:width]
|
|
128
|
+
batch.append(row)
|
|
129
|
+
if len(batch) >= rows_per_block:
|
|
130
|
+
yield dtz.Table(list(header), batch)
|
|
131
|
+
yielded = True
|
|
132
|
+
batch = []
|
|
133
|
+
# Only emit a trailing block if it holds rows. The exception is a file
|
|
134
|
+
# with a header and no rows at all: that still needs one empty block,
|
|
135
|
+
# or the archive would record no columns. Yielding unconditionally --
|
|
136
|
+
# as an earlier `if batch or True` did -- appended a wasted ~178-byte
|
|
137
|
+
# encoding of nothing whenever the row count divided evenly.
|
|
138
|
+
if batch or not yielded:
|
|
139
|
+
yield dtz.Table(list(header), batch)
|
|
140
|
+
finally:
|
|
141
|
+
fh.close()
|
|
142
|
+
|
|
143
|
+
|
|
144
|
+
def compress(src: str, dst: str, budget_gb: float = DEFAULT_BUDGET_GB,
|
|
145
|
+
rows: Optional[int] = None, verify: bool = True,
|
|
146
|
+
progress=None, encoding: Optional[str] = None) -> dict:
|
|
147
|
+
if rows is None:
|
|
148
|
+
rows, per_row, est, _ = plan_rows(src, budget_gb, verify, encoding)
|
|
149
|
+
else:
|
|
150
|
+
per_row, est = 0.0, 0
|
|
151
|
+
sizes: List[int] = []
|
|
152
|
+
total_rows = 0
|
|
153
|
+
columns: List[str] = []
|
|
154
|
+
|
|
155
|
+
with open(dst, "wb") as out:
|
|
156
|
+
out.write(b"\0" * 8) # header length patched at the end
|
|
157
|
+
for i, table in enumerate(_blocks(src, rows, encoding)):
|
|
158
|
+
if not columns:
|
|
159
|
+
columns = table.columns
|
|
160
|
+
blob = fast.encode(table)
|
|
161
|
+
if verify:
|
|
162
|
+
back = fast.decode(blob)
|
|
163
|
+
if back.columns != table.columns or back.rows != table.rows:
|
|
164
|
+
raise RuntimeError(
|
|
165
|
+
"block {} failed verification; nothing usable written"
|
|
166
|
+
.format(i))
|
|
167
|
+
out.write(blob)
|
|
168
|
+
sizes.append(len(blob))
|
|
169
|
+
total_rows += len(table.rows)
|
|
170
|
+
if progress:
|
|
171
|
+
progress(i + 1, total_rows, sum(sizes))
|
|
172
|
+
|
|
173
|
+
header = {"columns": columns, "nrows": total_rows,
|
|
174
|
+
"rows_per_block": rows, "blocks": sizes}
|
|
175
|
+
hb = lzma.compress(json.dumps(header, separators=(",", ":")).encode(),
|
|
176
|
+
**fast.XZ)
|
|
177
|
+
out.write(hb)
|
|
178
|
+
out.write(len(hb).to_bytes(8, "big"))
|
|
179
|
+
|
|
180
|
+
# Rewrite the magic now that the file is complete: a truncated run leaves
|
|
181
|
+
# eight zero bytes at the front rather than something that looks valid.
|
|
182
|
+
with open(dst, "r+b") as out:
|
|
183
|
+
out.seek(0)
|
|
184
|
+
out.write(MAGIC + b"\0\0\0\0")
|
|
185
|
+
return {"rows": total_rows, "blocks": len(sizes),
|
|
186
|
+
"rows_per_block": rows, "bytes": os.path.getsize(dst),
|
|
187
|
+
"per_row": per_row}
|
|
188
|
+
|
|
189
|
+
|
|
190
|
+
def _read_header(fh):
|
|
191
|
+
fh.seek(0)
|
|
192
|
+
if fh.read(4) != MAGIC:
|
|
193
|
+
raise ValueError("not a Polypress stream archive")
|
|
194
|
+
fh.seek(-8, os.SEEK_END)
|
|
195
|
+
hlen = int.from_bytes(fh.read(8), "big")
|
|
196
|
+
fh.seek(-8 - hlen, os.SEEK_END)
|
|
197
|
+
return json.loads(lzma.decompress(fh.read(hlen), **fast.XZ))
|
|
198
|
+
|
|
199
|
+
|
|
200
|
+
def iter_blocks(src: str) -> Iterator[dtz.Table]:
|
|
201
|
+
"""Yield each block's table in order, holding one block at a time."""
|
|
202
|
+
with open(src, "rb") as fh:
|
|
203
|
+
header = _read_header(fh)
|
|
204
|
+
fh.seek(8)
|
|
205
|
+
for size in header["blocks"]:
|
|
206
|
+
yield fast.decode(fh.read(size))
|
|
207
|
+
|
|
208
|
+
|
|
209
|
+
# Restoring has to honour the output extension the same way tzip.py does --
|
|
210
|
+
# `restore -o out.parquet` must produce parquet, not a CSV wearing the name.
|
|
211
|
+
# But it cannot call dtz.write_any, which needs the whole table in memory; the
|
|
212
|
+
# entire point here is that the table does not fit. So each format gets a
|
|
213
|
+
# writer that consumes one block at a time and holds nothing else.
|
|
214
|
+
|
|
215
|
+
class _DelimitedOut:
|
|
216
|
+
def __init__(self, path, delim):
|
|
217
|
+
self.fh = open(path, "w", newline="", encoding="utf-8")
|
|
218
|
+
self.w = csv.writer(self.fh, delimiter=delim, lineterminator="\n")
|
|
219
|
+
self.first = True
|
|
220
|
+
|
|
221
|
+
def block(self, table):
|
|
222
|
+
if self.first:
|
|
223
|
+
self.w.writerow(table.columns)
|
|
224
|
+
self.first = False
|
|
225
|
+
self.w.writerows(table.rows)
|
|
226
|
+
|
|
227
|
+
def close(self):
|
|
228
|
+
self.fh.close()
|
|
229
|
+
|
|
230
|
+
|
|
231
|
+
class _JsonlOut:
|
|
232
|
+
def __init__(self, path, columns):
|
|
233
|
+
dtz._require_unique_columns(dtz.Table(list(columns), []), "jsonl")
|
|
234
|
+
self.columns = columns
|
|
235
|
+
self.path = path
|
|
236
|
+
self.fh = open(path, "w", encoding="utf-8")
|
|
237
|
+
self.rows = 0
|
|
238
|
+
|
|
239
|
+
def block(self, table):
|
|
240
|
+
dump = json.dumps
|
|
241
|
+
for row in table.rows:
|
|
242
|
+
self.fh.write(dump(dict(zip(self.columns, row)),
|
|
243
|
+
ensure_ascii=False, separators=(",", ":")))
|
|
244
|
+
self.fh.write("\n")
|
|
245
|
+
self.rows += 1
|
|
246
|
+
|
|
247
|
+
def close(self):
|
|
248
|
+
self.fh.close()
|
|
249
|
+
if not self.rows and self.columns:
|
|
250
|
+
# dtz's contract is that a FormatLimit leaves nothing behind. It
|
|
251
|
+
# can check up front; we only learn the table was empty after the
|
|
252
|
+
# last block, so remove the file we opened.
|
|
253
|
+
os.unlink(self.path)
|
|
254
|
+
raise dtz.FormatLimit(
|
|
255
|
+
"jsonl cannot carry column names for a zero-row table; "
|
|
256
|
+
"write .csv, .json or .parquet instead")
|
|
257
|
+
|
|
258
|
+
|
|
259
|
+
class _JsonOut:
|
|
260
|
+
"""A JSON array emitted incrementally, so the records never coexist."""
|
|
261
|
+
|
|
262
|
+
def __init__(self, path, columns):
|
|
263
|
+
dtz._require_unique_columns(dtz.Table(list(columns), []), "json")
|
|
264
|
+
self.columns = columns
|
|
265
|
+
self.path = path
|
|
266
|
+
self.fh = open(path, "w", encoding="utf-8")
|
|
267
|
+
self.rows = 0
|
|
268
|
+
|
|
269
|
+
def block(self, table):
|
|
270
|
+
dump = json.dumps
|
|
271
|
+
for row in table.rows:
|
|
272
|
+
self.fh.write("[\n " if not self.rows else ",\n ")
|
|
273
|
+
self.fh.write(dump(dict(zip(self.columns, row)),
|
|
274
|
+
ensure_ascii=False))
|
|
275
|
+
self.rows += 1
|
|
276
|
+
|
|
277
|
+
def close(self):
|
|
278
|
+
if self.rows:
|
|
279
|
+
self.fh.write("\n]")
|
|
280
|
+
self.fh.close()
|
|
281
|
+
return
|
|
282
|
+
# zero rows: a record list cannot carry headers, so use the column
|
|
283
|
+
# form, exactly as dtz.write_json does
|
|
284
|
+
self.fh.close()
|
|
285
|
+
with open(self.path, "w", encoding="utf-8") as fh:
|
|
286
|
+
json.dump({c: [] for c in self.columns}, fh, ensure_ascii=False)
|
|
287
|
+
|
|
288
|
+
|
|
289
|
+
class _ParquetOut:
|
|
290
|
+
"""One row group per block -- the block structure maps straight onto it."""
|
|
291
|
+
|
|
292
|
+
def __init__(self, path, columns):
|
|
293
|
+
if dtz.pq is None:
|
|
294
|
+
raise RuntimeError("writing parquet needs pyarrow installed")
|
|
295
|
+
dtz._require_unique_columns(dtz.Table(list(columns), []), "parquet")
|
|
296
|
+
self.columns = columns
|
|
297
|
+
# Every cell is a string in a Table, so the schema is fixed up front
|
|
298
|
+
# and stays identical across blocks -- which is what lets the row
|
|
299
|
+
# groups append without buffering.
|
|
300
|
+
self.schema = dtz.pa.schema([(c, dtz.pa.string()) for c in columns])
|
|
301
|
+
self.writer = dtz.pq.ParquetWriter(path, self.schema,
|
|
302
|
+
compression="zstd")
|
|
303
|
+
|
|
304
|
+
def block(self, table):
|
|
305
|
+
if not table.rows:
|
|
306
|
+
return
|
|
307
|
+
arrays = [dtz.pa.array(table.column(i))
|
|
308
|
+
for i in range(len(self.columns))]
|
|
309
|
+
self.writer.write_table(dtz.pa.table(arrays, schema=self.schema))
|
|
310
|
+
|
|
311
|
+
def close(self):
|
|
312
|
+
self.writer.close()
|
|
313
|
+
|
|
314
|
+
|
|
315
|
+
def _open_writer(dst: str, columns: List[str]):
|
|
316
|
+
ext = os.path.splitext(dst)[1].lower()
|
|
317
|
+
if ext == ".tsv":
|
|
318
|
+
return _DelimitedOut(dst, "\t")
|
|
319
|
+
if ext == ".json":
|
|
320
|
+
return _JsonOut(dst, columns)
|
|
321
|
+
if ext in (".jsonl", ".ndjson"):
|
|
322
|
+
return _JsonlOut(dst, columns)
|
|
323
|
+
if ext in (".parquet", ".pq"):
|
|
324
|
+
return _ParquetOut(dst, columns)
|
|
325
|
+
return _DelimitedOut(dst, dtz.EXT_DELIMITER.get(ext, ","))
|
|
326
|
+
|
|
327
|
+
|
|
328
|
+
def restore(src: str, dst: str, progress=None) -> dict:
|
|
329
|
+
with open(src, "rb") as fh:
|
|
330
|
+
header = _read_header(fh)
|
|
331
|
+
out = _open_writer(dst, header["columns"])
|
|
332
|
+
written = 0
|
|
333
|
+
try:
|
|
334
|
+
for table in iter_blocks(src):
|
|
335
|
+
out.block(table)
|
|
336
|
+
written += len(table.rows)
|
|
337
|
+
if progress:
|
|
338
|
+
progress(written)
|
|
339
|
+
finally:
|
|
340
|
+
out.close()
|
|
341
|
+
return {"rows": written, "blocks": len(header["blocks"]),
|
|
342
|
+
"bytes": os.path.getsize(dst)}
|
|
343
|
+
|
|
344
|
+
|
|
345
|
+
def info(src: str) -> dict:
|
|
346
|
+
with open(src, "rb") as fh:
|
|
347
|
+
header = _read_header(fh)
|
|
348
|
+
header["size"] = os.path.getsize(src)
|
|
349
|
+
return header
|
|
350
|
+
|
|
351
|
+
|
|
352
|
+
# ---------------------------------------------------------------------- cli
|
|
353
|
+
|
|
354
|
+
def human(n: float) -> str:
|
|
355
|
+
for unit in ("B", "KB", "MB", "GB"):
|
|
356
|
+
if abs(n) < 1024 or unit == "GB":
|
|
357
|
+
return "{:,.0f} {}".format(n, unit) if unit == "B" \
|
|
358
|
+
else "{:,.1f} {}".format(n, unit)
|
|
359
|
+
n /= 1024.0
|
|
360
|
+
return str(n)
|
|
361
|
+
|
|
362
|
+
|
|
363
|
+
def main(argv=None) -> int:
|
|
364
|
+
ap = argparse.ArgumentParser(prog="polypress-stream",
|
|
365
|
+
description=__doc__.split("\n")[0])
|
|
366
|
+
sub = ap.add_subparsers(dest="cmd", required=True)
|
|
367
|
+
|
|
368
|
+
c = sub.add_parser("compress")
|
|
369
|
+
c.add_argument("path")
|
|
370
|
+
c.add_argument("-o", "--output")
|
|
371
|
+
c.add_argument("--budget", type=float, default=DEFAULT_BUDGET_GB,
|
|
372
|
+
help="approximate peak memory in GB (default 1.0)")
|
|
373
|
+
c.add_argument("--rows", type=int, help="rows per block, overrides budget")
|
|
374
|
+
c.add_argument("--no-verify", action="store_true")
|
|
375
|
+
c.add_argument("--encoding",
|
|
376
|
+
help="text encoding of the input (default: utf-8, or "
|
|
377
|
+
"whatever a byte-order mark says)")
|
|
378
|
+
|
|
379
|
+
r = sub.add_parser("restore")
|
|
380
|
+
r.add_argument("path")
|
|
381
|
+
r.add_argument("-o", "--output")
|
|
382
|
+
|
|
383
|
+
i = sub.add_parser("info")
|
|
384
|
+
i.add_argument("path")
|
|
385
|
+
|
|
386
|
+
a = ap.parse_args(argv)
|
|
387
|
+
|
|
388
|
+
if a.cmd == "compress":
|
|
389
|
+
dst = a.output or a.path + ".ppz"
|
|
390
|
+
raw = os.path.getsize(a.path)
|
|
391
|
+
t0 = time.time()
|
|
392
|
+
|
|
393
|
+
def prog(nblk, nrows, nbytes):
|
|
394
|
+
sys.stderr.write("\r block {:>4} {:>10,} rows {:>10}"
|
|
395
|
+
.format(nblk, nrows, human(nbytes)))
|
|
396
|
+
sys.stderr.flush()
|
|
397
|
+
|
|
398
|
+
st = compress(a.path, dst, a.budget, a.rows, not a.no_verify, prog,
|
|
399
|
+
a.encoding)
|
|
400
|
+
sys.stderr.write("\r" + " " * 60 + "\r")
|
|
401
|
+
secs = time.time() - t0
|
|
402
|
+
print("{:,} rows in {} blocks of {:,} {} -> {} {:.2f}x "
|
|
403
|
+
"{:.1f} MB/s".format(
|
|
404
|
+
st["rows"], st["blocks"], st["rows_per_block"], human(raw),
|
|
405
|
+
human(st["bytes"]), raw / max(st["bytes"], 1),
|
|
406
|
+
raw / 1e6 / max(secs, 1e-9)))
|
|
407
|
+
print(dst)
|
|
408
|
+
return 0
|
|
409
|
+
|
|
410
|
+
if a.cmd == "restore":
|
|
411
|
+
dst = a.output or (a.path[:-4] if a.path.lower().endswith(".ppz")
|
|
412
|
+
else a.path + ".csv")
|
|
413
|
+
t0 = time.time()
|
|
414
|
+
st = restore(a.path, dst)
|
|
415
|
+
secs = time.time() - t0
|
|
416
|
+
print("{:,} rows from {} blocks {} {:.1f} MB/s".format(
|
|
417
|
+
st["rows"], st["blocks"], human(st["bytes"]),
|
|
418
|
+
st["bytes"] / 1e6 / max(secs, 1e-9)))
|
|
419
|
+
print(dst)
|
|
420
|
+
return 0
|
|
421
|
+
|
|
422
|
+
h = info(a.path)
|
|
423
|
+
print("file {}".format(a.path))
|
|
424
|
+
print("size {}".format(human(h["size"])))
|
|
425
|
+
print("rows {:,}".format(h["nrows"]))
|
|
426
|
+
print("columns {}".format(len(h["columns"])))
|
|
427
|
+
print("blocks {} of {:,} rows".format(
|
|
428
|
+
len(h["blocks"]), h["rows_per_block"]))
|
|
429
|
+
if h["blocks"]:
|
|
430
|
+
print("block sizes min {} median {} max {}".format(
|
|
431
|
+
human(min(h["blocks"])),
|
|
432
|
+
human(sorted(h["blocks"])[len(h["blocks"]) // 2]),
|
|
433
|
+
human(max(h["blocks"]))))
|
|
434
|
+
return 0
|
|
435
|
+
|
|
436
|
+
|
|
437
|
+
if __name__ == "__main__":
|
|
438
|
+
sys.exit(main())
|
polypress/tcz.c
ADDED
|
@@ -0,0 +1,207 @@
|
|
|
1
|
+
/* Hot loops for the table codec, in C.
|
|
2
|
+
*
|
|
3
|
+
* Only four things live here, and each earned its place by showing up in a
|
|
4
|
+
* profile:
|
|
5
|
+
*
|
|
6
|
+
* scan_decimals / parse_fixed / fmt_fixed
|
|
7
|
+
* Converting between decimal text and scaled integers was 70% of decode
|
|
8
|
+
* and a large share of encode. Both directions are here so the caller
|
|
9
|
+
* can verify exactness by parsing, re-formatting, and comparing bytes.
|
|
10
|
+
*
|
|
11
|
+
* pack_ints / unpack_ints
|
|
12
|
+
* One-byte varints with an escape.
|
|
13
|
+
*
|
|
14
|
+
* Deliberately plain C with no Python headers: it builds with a bare `cc`
|
|
15
|
+
* and is loaded through ctypes, so there is no extension-module toolchain to
|
|
16
|
+
* keep working. Python falls back to the numpy path if the library is
|
|
17
|
+
* missing, so this is an accelerator, never a requirement.
|
|
18
|
+
*/
|
|
19
|
+
|
|
20
|
+
#include <stdint.h>
|
|
21
|
+
#include <string.h>
|
|
22
|
+
|
|
23
|
+
#define ESCAPE 255
|
|
24
|
+
|
|
25
|
+
/* Largest magnitude we accept, matching INT_LIMIT on the Python side. */
|
|
26
|
+
static const int64_t LIMIT = ((int64_t)1) << 62;
|
|
27
|
+
|
|
28
|
+
/* ------------------------------------------------------------------ scan */
|
|
29
|
+
|
|
30
|
+
/* Scan a newline-separated buffer of numbers.
|
|
31
|
+
* Returns the maximum number of decimal places, or -1 if any field is not a
|
|
32
|
+
* plain decimal number (which sends the column down the text path).
|
|
33
|
+
* `count` receives the number of fields. */
|
|
34
|
+
int32_t scan_decimals(const char *buf, int64_t len, int64_t *count)
|
|
35
|
+
{
|
|
36
|
+
int32_t max_dec = 0;
|
|
37
|
+
int64_t fields = 0;
|
|
38
|
+
int64_t i = 0;
|
|
39
|
+
|
|
40
|
+
while (i <= len) {
|
|
41
|
+
int64_t start = i;
|
|
42
|
+
while (i < len && buf[i] != '\n') i++;
|
|
43
|
+
int64_t end = i;
|
|
44
|
+
int64_t p = start;
|
|
45
|
+
int digits = 0, dec = 0;
|
|
46
|
+
|
|
47
|
+
if (p < end && buf[p] == '-') p++;
|
|
48
|
+
while (p < end && buf[p] >= '0' && buf[p] <= '9') { p++; digits++; }
|
|
49
|
+
if (p < end && buf[p] == '.') {
|
|
50
|
+
p++;
|
|
51
|
+
while (p < end && buf[p] >= '0' && buf[p] <= '9') { p++; dec++; }
|
|
52
|
+
if (dec == 0) return -1; /* "12." is not accepted */
|
|
53
|
+
}
|
|
54
|
+
if (digits == 0 || p != end) return -1; /* empty or trailing junk */
|
|
55
|
+
if (dec > max_dec) max_dec = dec;
|
|
56
|
+
fields++;
|
|
57
|
+
|
|
58
|
+
if (i >= len) break;
|
|
59
|
+
i++; /* step over the newline */
|
|
60
|
+
}
|
|
61
|
+
if (count) *count = fields;
|
|
62
|
+
return max_dec > 18 ? -1 : max_dec;
|
|
63
|
+
}
|
|
64
|
+
|
|
65
|
+
/* ----------------------------------------------------------------- parse */
|
|
66
|
+
|
|
67
|
+
/* Parse `n` newline-separated decimals into integers scaled by 10^dec.
|
|
68
|
+
* Returns 0 on success, -1 on malformed input or overflow. */
|
|
69
|
+
int32_t parse_fixed(const char *buf, int64_t len, int32_t dec,
|
|
70
|
+
int64_t *out, int64_t n)
|
|
71
|
+
{
|
|
72
|
+
static const int64_t POW10[19] = {
|
|
73
|
+
1LL, 10LL, 100LL, 1000LL, 10000LL, 100000LL, 1000000LL,
|
|
74
|
+
10000000LL, 100000000LL, 1000000000LL, 10000000000LL,
|
|
75
|
+
100000000000LL, 1000000000000LL, 10000000000000LL,
|
|
76
|
+
100000000000000LL, 1000000000000000LL, 10000000000000000LL,
|
|
77
|
+
100000000000000000LL, 1000000000000000000LL};
|
|
78
|
+
|
|
79
|
+
int64_t i = 0, k = 0;
|
|
80
|
+
while (k < n) {
|
|
81
|
+
int64_t start = i;
|
|
82
|
+
while (i < len && buf[i] != '\n') i++;
|
|
83
|
+
int64_t end = i;
|
|
84
|
+
int64_t p = start;
|
|
85
|
+
int neg = 0;
|
|
86
|
+
int64_t v = 0;
|
|
87
|
+
int seen = 0;
|
|
88
|
+
|
|
89
|
+
/* Overflow has to be caught BEFORE it happens. Checking `v >= LIMIT`
|
|
90
|
+
* after `v = v*10 + d` is too late: signed overflow is undefined, and
|
|
91
|
+
* in practice it wraps negative, so the guard never fires. That let
|
|
92
|
+
* "-9223372036854775808" through as a valid value even though the
|
|
93
|
+
* documented limit is 2^62 -- and the numpy fallback rejected it, so
|
|
94
|
+
* the accelerator and the reference disagreed about what a column
|
|
95
|
+
* even was. Rearranged, the test is exact and never overflows. */
|
|
96
|
+
#define ACC_DIGIT(d) \
|
|
97
|
+
do { \
|
|
98
|
+
int dgt_ = (d); \
|
|
99
|
+
if (v > (LIMIT - 1 - dgt_) / 10) return -1; \
|
|
100
|
+
v = v * 10 + dgt_; \
|
|
101
|
+
} while (0)
|
|
102
|
+
|
|
103
|
+
if (p < end && buf[p] == '-') { neg = 1; p++; }
|
|
104
|
+
while (p < end && buf[p] >= '0' && buf[p] <= '9') {
|
|
105
|
+
ACC_DIGIT(buf[p] - '0');
|
|
106
|
+
p++; seen = 1;
|
|
107
|
+
}
|
|
108
|
+
if (!seen) return -1;
|
|
109
|
+
|
|
110
|
+
int used = 0;
|
|
111
|
+
if (p < end && buf[p] == '.') {
|
|
112
|
+
p++;
|
|
113
|
+
while (p < end && buf[p] >= '0' && buf[p] <= '9') {
|
|
114
|
+
if (used < dec) {
|
|
115
|
+
ACC_DIGIT(buf[p] - '0');
|
|
116
|
+
used++;
|
|
117
|
+
}
|
|
118
|
+
p++;
|
|
119
|
+
}
|
|
120
|
+
}
|
|
121
|
+
if (p != end) return -1;
|
|
122
|
+
while (used < dec) { /* pad to the column scale */
|
|
123
|
+
ACC_DIGIT(0);
|
|
124
|
+
used++;
|
|
125
|
+
}
|
|
126
|
+
#undef ACC_DIGIT
|
|
127
|
+
out[k++] = neg ? -v : v;
|
|
128
|
+
|
|
129
|
+
if (i >= len) break;
|
|
130
|
+
i++;
|
|
131
|
+
}
|
|
132
|
+
return k == n ? 0 : -1;
|
|
133
|
+
}
|
|
134
|
+
|
|
135
|
+
/* ---------------------------------------------------------------- format */
|
|
136
|
+
|
|
137
|
+
/* Write `n` scaled integers back as newline-separated decimal text.
|
|
138
|
+
* Returns bytes written, or -1 if `cap` was too small. */
|
|
139
|
+
int64_t fmt_fixed(const int64_t *in, int64_t n, int32_t dec,
|
|
140
|
+
char *out, int64_t cap)
|
|
141
|
+
{
|
|
142
|
+
char digits[24];
|
|
143
|
+
int64_t at = 0;
|
|
144
|
+
|
|
145
|
+
for (int64_t k = 0; k < n; k++) {
|
|
146
|
+
int64_t v = in[k];
|
|
147
|
+
/* worst case: sign + 19 integer digits + point + 18 decimals + \n */
|
|
148
|
+
if (at + 40 > cap) return -1;
|
|
149
|
+
|
|
150
|
+
if (v < 0) { out[at++] = '-'; v = -v; }
|
|
151
|
+
|
|
152
|
+
int64_t scale = 1;
|
|
153
|
+
for (int32_t d = 0; d < dec; d++) scale *= 10;
|
|
154
|
+
int64_t q = dec ? v / scale : v;
|
|
155
|
+
int64_t r = dec ? v % scale : 0;
|
|
156
|
+
|
|
157
|
+
int nd = 0;
|
|
158
|
+
if (q == 0) digits[nd++] = '0';
|
|
159
|
+
while (q > 0) { digits[nd++] = (char)('0' + (q % 10)); q /= 10; }
|
|
160
|
+
while (nd > 0) out[at++] = digits[--nd];
|
|
161
|
+
|
|
162
|
+
if (dec) {
|
|
163
|
+
out[at++] = '.';
|
|
164
|
+
for (int32_t d = dec - 1; d >= 0; d--) {
|
|
165
|
+
int64_t p = 1;
|
|
166
|
+
for (int32_t e = 0; e < d; e++) p *= 10;
|
|
167
|
+
out[at++] = (char)('0' + ((r / p) % 10));
|
|
168
|
+
}
|
|
169
|
+
}
|
|
170
|
+
out[at++] = '\n';
|
|
171
|
+
}
|
|
172
|
+
return at ? at - 1 : 0; /* drop the final newline */
|
|
173
|
+
}
|
|
174
|
+
|
|
175
|
+
/* ------------------------------------------------------------- varints */
|
|
176
|
+
|
|
177
|
+
/* Zigzag each value; small ones take a byte, the rest an escape plus a
|
|
178
|
+
* 64-bit tail entry. The tail must be 64 bits: a 32-bit tail silently
|
|
179
|
+
* truncated any value past 2^31, which is well inside the int64 range this
|
|
180
|
+
* codec accepts. */
|
|
181
|
+
int64_t pack_ints(const int64_t *in, int64_t n,
|
|
182
|
+
uint8_t *head, uint64_t *tail)
|
|
183
|
+
{
|
|
184
|
+
int64_t nbig = 0;
|
|
185
|
+
for (int64_t i = 0; i < n; i++) {
|
|
186
|
+
int64_t v = in[i];
|
|
187
|
+
uint64_t u = v < 0 ? (uint64_t)(-v) * 2 - 1 : (uint64_t)v * 2;
|
|
188
|
+
if (u < ESCAPE) {
|
|
189
|
+
head[i] = (uint8_t)u;
|
|
190
|
+
} else {
|
|
191
|
+
head[i] = ESCAPE;
|
|
192
|
+
tail[nbig++] = u;
|
|
193
|
+
}
|
|
194
|
+
}
|
|
195
|
+
return nbig;
|
|
196
|
+
}
|
|
197
|
+
|
|
198
|
+
void unpack_ints(const uint8_t *head, int64_t n,
|
|
199
|
+
const uint64_t *tail, int64_t *out)
|
|
200
|
+
{
|
|
201
|
+
int64_t nbig = 0;
|
|
202
|
+
for (int64_t i = 0; i < n; i++) {
|
|
203
|
+
uint64_t u = head[i];
|
|
204
|
+
if (u == ESCAPE) u = tail[nbig++];
|
|
205
|
+
out[i] = (u & 1) ? -(int64_t)((u + 1) / 2) : (int64_t)(u / 2);
|
|
206
|
+
}
|
|
207
|
+
}
|