polypress 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
polypress/stream.py ADDED
@@ -0,0 +1,438 @@
1
+ """Bounded-memory compression for files larger than RAM.
2
+
3
+ The single-shot codec in fast.py holds the whole table as Python strings.
4
+ Measured expansion is about 8x -- a 291 MB CSV peaks at 2.4 GB -- so a 50 GB
5
+ file would need hundreds of gigabytes and simply will not run.
6
+
7
+ This splits the table into row blocks, compresses each independently, and
8
+ concatenates them behind an index. Peak memory is one block, not one file,
9
+ and it is settable. Blocks are independent, so this is also the thing that
10
+ makes parallelism possible later.
11
+
12
+ The cost is real and worth stating: the cross-column reordering only sees
13
+ correlations *inside* a block, so smaller blocks compress slightly worse.
14
+ Measure with `--rows` before choosing a small one.
15
+
16
+ python3 tzip.py stream-compress big.csv -o big.ppz --budget 1.0
17
+ python3 tzip.py stream-restore big.ppz -o back.csv
18
+ python3 tzip.py stream-info big.ppz
19
+
20
+ Or as a module, which is the same code path:
21
+
22
+ python3 -m polypress.stream compress big.csv --budget 1.0
23
+ """
24
+
25
+ from __future__ import annotations
26
+
27
+ import argparse
28
+ import csv
29
+ import json
30
+ import lzma
31
+ import os
32
+ import sys
33
+ import time
34
+ from typing import Iterator, List, Optional
35
+
36
+ sys.path.insert(0, os.path.dirname(os.path.dirname(
37
+ os.path.abspath(__file__))))
38
+
39
+ from . import dtz
40
+ from . import fast
41
+
42
+ MAGIC = b"PPZS"
43
+ DEFAULT_BUDGET_GB = 1.0
44
+ # Python strings cost roughly this much more than the CSV bytes they came
45
+ # from; measured at 8.1x on NHAMCS (291 MB file -> 2.36 GB resident).
46
+ EXPANSION = 8.5
47
+ # Verification decodes the block and compares, so the original block and the
48
+ # decoded copy are both resident at once. Without this the budget lied by
49
+ # roughly 2x: asking for 1 GB gave a 2.07 GB peak.
50
+ VERIFY_FACTOR = 2.2
51
+ MIN_ROWS = 1000
52
+ MAX_ROWS = 2_000_000
53
+
54
+ csv.field_size_limit(1 << 31)
55
+
56
+
57
+ def _decode_guard(rows, path: str, enc: str):
58
+ """Turn a lazy decode failure into the same refusal dtz raises.
59
+
60
+ The single-shot reader decodes the whole file inside one call, so a bad
61
+ byte surfaces there. Here the file is read block by block, so it surfaces
62
+ hundreds of rows in -- possibly after several blocks have already been
63
+ written. Same error, same message, wherever it lands."""
64
+ try:
65
+ for row in rows:
66
+ yield row
67
+ except UnicodeDecodeError as exc:
68
+ raise dtz.EncodingRefused(
69
+ dtz.encoding_error(path, exc, enc)) from None
70
+
71
+
72
+ def _reader(path: str, encoding: Optional[str] = None):
73
+ ext = os.path.splitext(path)[1].lower()
74
+ enc = encoding or dtz.sniff_encoding(path)
75
+ fh = dtz.open_text(path, enc)
76
+ try:
77
+ delimiter = dtz.EXT_DELIMITER.get(ext)
78
+ if delimiter is None:
79
+ delimiter = dtz._sniff_delimiter(fh.read(65536))
80
+ fh.seek(0)
81
+ except UnicodeDecodeError as exc:
82
+ fh.close()
83
+ raise dtz.EncodingRefused(
84
+ dtz.encoding_error(path, exc, enc)) from None
85
+ return fh, _decode_guard(csv.reader(fh, delimiter=delimiter), path, enc)
86
+
87
+
88
+ def plan_rows(path: str, budget_gb: float, verify: bool = True,
89
+ encoding: Optional[str] = None) -> tuple:
90
+ """Rows per block that keep peak memory near the budget.
91
+
92
+ Estimated from the real average row width of the file rather than a
93
+ guess, because a 209-column survey row and a 5-column sensor row differ
94
+ by two orders of magnitude."""
95
+ size = os.path.getsize(path)
96
+ fh, rd = _reader(path, encoding)
97
+ try:
98
+ header = next(rd, [])
99
+ n, seen = 0, 0
100
+ for row in rd:
101
+ seen += sum(len(c) for c in row) + len(row)
102
+ n += 1
103
+ if n >= 2000:
104
+ break
105
+ finally:
106
+ fh.close()
107
+ per_row = max(1.0, seen / max(n, 1))
108
+ factor = EXPANSION * (VERIFY_FACTOR if verify else 1.0)
109
+ rows = int((budget_gb * (1 << 30)) / (per_row * factor))
110
+ rows = max(MIN_ROWS, min(MAX_ROWS, rows))
111
+ est_blocks = max(1, int(size / (per_row * rows)) + 1)
112
+ return rows, per_row, est_blocks, header
113
+
114
+
115
+ def _blocks(path: str, rows_per_block: int,
116
+ encoding: Optional[str] = None) -> Iterator[dtz.Table]:
117
+ fh, rd = _reader(path, encoding)
118
+ try:
119
+ header = next(rd, None)
120
+ if header is None:
121
+ return
122
+ width = len(header)
123
+ batch: List[List[str]] = []
124
+ yielded = False
125
+ for row in rd:
126
+ if len(row) != width: # ragged rows, same rule as dtz
127
+ row = (row + [""] * width)[:width]
128
+ batch.append(row)
129
+ if len(batch) >= rows_per_block:
130
+ yield dtz.Table(list(header), batch)
131
+ yielded = True
132
+ batch = []
133
+ # Only emit a trailing block if it holds rows. The exception is a file
134
+ # with a header and no rows at all: that still needs one empty block,
135
+ # or the archive would record no columns. Yielding unconditionally --
136
+ # as an earlier `if batch or True` did -- appended a wasted ~178-byte
137
+ # encoding of nothing whenever the row count divided evenly.
138
+ if batch or not yielded:
139
+ yield dtz.Table(list(header), batch)
140
+ finally:
141
+ fh.close()
142
+
143
+
144
+ def compress(src: str, dst: str, budget_gb: float = DEFAULT_BUDGET_GB,
145
+ rows: Optional[int] = None, verify: bool = True,
146
+ progress=None, encoding: Optional[str] = None) -> dict:
147
+ if rows is None:
148
+ rows, per_row, est, _ = plan_rows(src, budget_gb, verify, encoding)
149
+ else:
150
+ per_row, est = 0.0, 0
151
+ sizes: List[int] = []
152
+ total_rows = 0
153
+ columns: List[str] = []
154
+
155
+ with open(dst, "wb") as out:
156
+ out.write(b"\0" * 8) # header length patched at the end
157
+ for i, table in enumerate(_blocks(src, rows, encoding)):
158
+ if not columns:
159
+ columns = table.columns
160
+ blob = fast.encode(table)
161
+ if verify:
162
+ back = fast.decode(blob)
163
+ if back.columns != table.columns or back.rows != table.rows:
164
+ raise RuntimeError(
165
+ "block {} failed verification; nothing usable written"
166
+ .format(i))
167
+ out.write(blob)
168
+ sizes.append(len(blob))
169
+ total_rows += len(table.rows)
170
+ if progress:
171
+ progress(i + 1, total_rows, sum(sizes))
172
+
173
+ header = {"columns": columns, "nrows": total_rows,
174
+ "rows_per_block": rows, "blocks": sizes}
175
+ hb = lzma.compress(json.dumps(header, separators=(",", ":")).encode(),
176
+ **fast.XZ)
177
+ out.write(hb)
178
+ out.write(len(hb).to_bytes(8, "big"))
179
+
180
+ # Rewrite the magic now that the file is complete: a truncated run leaves
181
+ # eight zero bytes at the front rather than something that looks valid.
182
+ with open(dst, "r+b") as out:
183
+ out.seek(0)
184
+ out.write(MAGIC + b"\0\0\0\0")
185
+ return {"rows": total_rows, "blocks": len(sizes),
186
+ "rows_per_block": rows, "bytes": os.path.getsize(dst),
187
+ "per_row": per_row}
188
+
189
+
190
+ def _read_header(fh):
191
+ fh.seek(0)
192
+ if fh.read(4) != MAGIC:
193
+ raise ValueError("not a Polypress stream archive")
194
+ fh.seek(-8, os.SEEK_END)
195
+ hlen = int.from_bytes(fh.read(8), "big")
196
+ fh.seek(-8 - hlen, os.SEEK_END)
197
+ return json.loads(lzma.decompress(fh.read(hlen), **fast.XZ))
198
+
199
+
200
+ def iter_blocks(src: str) -> Iterator[dtz.Table]:
201
+ """Yield each block's table in order, holding one block at a time."""
202
+ with open(src, "rb") as fh:
203
+ header = _read_header(fh)
204
+ fh.seek(8)
205
+ for size in header["blocks"]:
206
+ yield fast.decode(fh.read(size))
207
+
208
+
209
+ # Restoring has to honour the output extension the same way tzip.py does --
210
+ # `restore -o out.parquet` must produce parquet, not a CSV wearing the name.
211
+ # But it cannot call dtz.write_any, which needs the whole table in memory; the
212
+ # entire point here is that the table does not fit. So each format gets a
213
+ # writer that consumes one block at a time and holds nothing else.
214
+
215
+ class _DelimitedOut:
216
+ def __init__(self, path, delim):
217
+ self.fh = open(path, "w", newline="", encoding="utf-8")
218
+ self.w = csv.writer(self.fh, delimiter=delim, lineterminator="\n")
219
+ self.first = True
220
+
221
+ def block(self, table):
222
+ if self.first:
223
+ self.w.writerow(table.columns)
224
+ self.first = False
225
+ self.w.writerows(table.rows)
226
+
227
+ def close(self):
228
+ self.fh.close()
229
+
230
+
231
+ class _JsonlOut:
232
+ def __init__(self, path, columns):
233
+ dtz._require_unique_columns(dtz.Table(list(columns), []), "jsonl")
234
+ self.columns = columns
235
+ self.path = path
236
+ self.fh = open(path, "w", encoding="utf-8")
237
+ self.rows = 0
238
+
239
+ def block(self, table):
240
+ dump = json.dumps
241
+ for row in table.rows:
242
+ self.fh.write(dump(dict(zip(self.columns, row)),
243
+ ensure_ascii=False, separators=(",", ":")))
244
+ self.fh.write("\n")
245
+ self.rows += 1
246
+
247
+ def close(self):
248
+ self.fh.close()
249
+ if not self.rows and self.columns:
250
+ # dtz's contract is that a FormatLimit leaves nothing behind. It
251
+ # can check up front; we only learn the table was empty after the
252
+ # last block, so remove the file we opened.
253
+ os.unlink(self.path)
254
+ raise dtz.FormatLimit(
255
+ "jsonl cannot carry column names for a zero-row table; "
256
+ "write .csv, .json or .parquet instead")
257
+
258
+
259
+ class _JsonOut:
260
+ """A JSON array emitted incrementally, so the records never coexist."""
261
+
262
+ def __init__(self, path, columns):
263
+ dtz._require_unique_columns(dtz.Table(list(columns), []), "json")
264
+ self.columns = columns
265
+ self.path = path
266
+ self.fh = open(path, "w", encoding="utf-8")
267
+ self.rows = 0
268
+
269
+ def block(self, table):
270
+ dump = json.dumps
271
+ for row in table.rows:
272
+ self.fh.write("[\n " if not self.rows else ",\n ")
273
+ self.fh.write(dump(dict(zip(self.columns, row)),
274
+ ensure_ascii=False))
275
+ self.rows += 1
276
+
277
+ def close(self):
278
+ if self.rows:
279
+ self.fh.write("\n]")
280
+ self.fh.close()
281
+ return
282
+ # zero rows: a record list cannot carry headers, so use the column
283
+ # form, exactly as dtz.write_json does
284
+ self.fh.close()
285
+ with open(self.path, "w", encoding="utf-8") as fh:
286
+ json.dump({c: [] for c in self.columns}, fh, ensure_ascii=False)
287
+
288
+
289
+ class _ParquetOut:
290
+ """One row group per block -- the block structure maps straight onto it."""
291
+
292
+ def __init__(self, path, columns):
293
+ if dtz.pq is None:
294
+ raise RuntimeError("writing parquet needs pyarrow installed")
295
+ dtz._require_unique_columns(dtz.Table(list(columns), []), "parquet")
296
+ self.columns = columns
297
+ # Every cell is a string in a Table, so the schema is fixed up front
298
+ # and stays identical across blocks -- which is what lets the row
299
+ # groups append without buffering.
300
+ self.schema = dtz.pa.schema([(c, dtz.pa.string()) for c in columns])
301
+ self.writer = dtz.pq.ParquetWriter(path, self.schema,
302
+ compression="zstd")
303
+
304
+ def block(self, table):
305
+ if not table.rows:
306
+ return
307
+ arrays = [dtz.pa.array(table.column(i))
308
+ for i in range(len(self.columns))]
309
+ self.writer.write_table(dtz.pa.table(arrays, schema=self.schema))
310
+
311
+ def close(self):
312
+ self.writer.close()
313
+
314
+
315
+ def _open_writer(dst: str, columns: List[str]):
316
+ ext = os.path.splitext(dst)[1].lower()
317
+ if ext == ".tsv":
318
+ return _DelimitedOut(dst, "\t")
319
+ if ext == ".json":
320
+ return _JsonOut(dst, columns)
321
+ if ext in (".jsonl", ".ndjson"):
322
+ return _JsonlOut(dst, columns)
323
+ if ext in (".parquet", ".pq"):
324
+ return _ParquetOut(dst, columns)
325
+ return _DelimitedOut(dst, dtz.EXT_DELIMITER.get(ext, ","))
326
+
327
+
328
+ def restore(src: str, dst: str, progress=None) -> dict:
329
+ with open(src, "rb") as fh:
330
+ header = _read_header(fh)
331
+ out = _open_writer(dst, header["columns"])
332
+ written = 0
333
+ try:
334
+ for table in iter_blocks(src):
335
+ out.block(table)
336
+ written += len(table.rows)
337
+ if progress:
338
+ progress(written)
339
+ finally:
340
+ out.close()
341
+ return {"rows": written, "blocks": len(header["blocks"]),
342
+ "bytes": os.path.getsize(dst)}
343
+
344
+
345
+ def info(src: str) -> dict:
346
+ with open(src, "rb") as fh:
347
+ header = _read_header(fh)
348
+ header["size"] = os.path.getsize(src)
349
+ return header
350
+
351
+
352
+ # ---------------------------------------------------------------------- cli
353
+
354
+ def human(n: float) -> str:
355
+ for unit in ("B", "KB", "MB", "GB"):
356
+ if abs(n) < 1024 or unit == "GB":
357
+ return "{:,.0f} {}".format(n, unit) if unit == "B" \
358
+ else "{:,.1f} {}".format(n, unit)
359
+ n /= 1024.0
360
+ return str(n)
361
+
362
+
363
+ def main(argv=None) -> int:
364
+ ap = argparse.ArgumentParser(prog="polypress-stream",
365
+ description=__doc__.split("\n")[0])
366
+ sub = ap.add_subparsers(dest="cmd", required=True)
367
+
368
+ c = sub.add_parser("compress")
369
+ c.add_argument("path")
370
+ c.add_argument("-o", "--output")
371
+ c.add_argument("--budget", type=float, default=DEFAULT_BUDGET_GB,
372
+ help="approximate peak memory in GB (default 1.0)")
373
+ c.add_argument("--rows", type=int, help="rows per block, overrides budget")
374
+ c.add_argument("--no-verify", action="store_true")
375
+ c.add_argument("--encoding",
376
+ help="text encoding of the input (default: utf-8, or "
377
+ "whatever a byte-order mark says)")
378
+
379
+ r = sub.add_parser("restore")
380
+ r.add_argument("path")
381
+ r.add_argument("-o", "--output")
382
+
383
+ i = sub.add_parser("info")
384
+ i.add_argument("path")
385
+
386
+ a = ap.parse_args(argv)
387
+
388
+ if a.cmd == "compress":
389
+ dst = a.output or a.path + ".ppz"
390
+ raw = os.path.getsize(a.path)
391
+ t0 = time.time()
392
+
393
+ def prog(nblk, nrows, nbytes):
394
+ sys.stderr.write("\r block {:>4} {:>10,} rows {:>10}"
395
+ .format(nblk, nrows, human(nbytes)))
396
+ sys.stderr.flush()
397
+
398
+ st = compress(a.path, dst, a.budget, a.rows, not a.no_verify, prog,
399
+ a.encoding)
400
+ sys.stderr.write("\r" + " " * 60 + "\r")
401
+ secs = time.time() - t0
402
+ print("{:,} rows in {} blocks of {:,} {} -> {} {:.2f}x "
403
+ "{:.1f} MB/s".format(
404
+ st["rows"], st["blocks"], st["rows_per_block"], human(raw),
405
+ human(st["bytes"]), raw / max(st["bytes"], 1),
406
+ raw / 1e6 / max(secs, 1e-9)))
407
+ print(dst)
408
+ return 0
409
+
410
+ if a.cmd == "restore":
411
+ dst = a.output or (a.path[:-4] if a.path.lower().endswith(".ppz")
412
+ else a.path + ".csv")
413
+ t0 = time.time()
414
+ st = restore(a.path, dst)
415
+ secs = time.time() - t0
416
+ print("{:,} rows from {} blocks {} {:.1f} MB/s".format(
417
+ st["rows"], st["blocks"], human(st["bytes"]),
418
+ st["bytes"] / 1e6 / max(secs, 1e-9)))
419
+ print(dst)
420
+ return 0
421
+
422
+ h = info(a.path)
423
+ print("file {}".format(a.path))
424
+ print("size {}".format(human(h["size"])))
425
+ print("rows {:,}".format(h["nrows"]))
426
+ print("columns {}".format(len(h["columns"])))
427
+ print("blocks {} of {:,} rows".format(
428
+ len(h["blocks"]), h["rows_per_block"]))
429
+ if h["blocks"]:
430
+ print("block sizes min {} median {} max {}".format(
431
+ human(min(h["blocks"])),
432
+ human(sorted(h["blocks"])[len(h["blocks"]) // 2]),
433
+ human(max(h["blocks"]))))
434
+ return 0
435
+
436
+
437
+ if __name__ == "__main__":
438
+ sys.exit(main())
polypress/tcz.c ADDED
@@ -0,0 +1,207 @@
1
+ /* Hot loops for the table codec, in C.
2
+ *
3
+ * Only four things live here, and each earned its place by showing up in a
4
+ * profile:
5
+ *
6
+ * scan_decimals / parse_fixed / fmt_fixed
7
+ * Converting between decimal text and scaled integers was 70% of decode
8
+ * and a large share of encode. Both directions are here so the caller
9
+ * can verify exactness by parsing, re-formatting, and comparing bytes.
10
+ *
11
+ * pack_ints / unpack_ints
12
+ * One-byte varints with an escape.
13
+ *
14
+ * Deliberately plain C with no Python headers: it builds with a bare `cc`
15
+ * and is loaded through ctypes, so there is no extension-module toolchain to
16
+ * keep working. Python falls back to the numpy path if the library is
17
+ * missing, so this is an accelerator, never a requirement.
18
+ */
19
+
20
+ #include <stdint.h>
21
+ #include <string.h>
22
+
23
+ #define ESCAPE 255
24
+
25
+ /* Largest magnitude we accept, matching INT_LIMIT on the Python side. */
26
+ static const int64_t LIMIT = ((int64_t)1) << 62;
27
+
28
+ /* ------------------------------------------------------------------ scan */
29
+
30
+ /* Scan a newline-separated buffer of numbers.
31
+ * Returns the maximum number of decimal places, or -1 if any field is not a
32
+ * plain decimal number (which sends the column down the text path).
33
+ * `count` receives the number of fields. */
34
+ int32_t scan_decimals(const char *buf, int64_t len, int64_t *count)
35
+ {
36
+ int32_t max_dec = 0;
37
+ int64_t fields = 0;
38
+ int64_t i = 0;
39
+
40
+ while (i <= len) {
41
+ int64_t start = i;
42
+ while (i < len && buf[i] != '\n') i++;
43
+ int64_t end = i;
44
+ int64_t p = start;
45
+ int digits = 0, dec = 0;
46
+
47
+ if (p < end && buf[p] == '-') p++;
48
+ while (p < end && buf[p] >= '0' && buf[p] <= '9') { p++; digits++; }
49
+ if (p < end && buf[p] == '.') {
50
+ p++;
51
+ while (p < end && buf[p] >= '0' && buf[p] <= '9') { p++; dec++; }
52
+ if (dec == 0) return -1; /* "12." is not accepted */
53
+ }
54
+ if (digits == 0 || p != end) return -1; /* empty or trailing junk */
55
+ if (dec > max_dec) max_dec = dec;
56
+ fields++;
57
+
58
+ if (i >= len) break;
59
+ i++; /* step over the newline */
60
+ }
61
+ if (count) *count = fields;
62
+ return max_dec > 18 ? -1 : max_dec;
63
+ }
64
+
65
+ /* ----------------------------------------------------------------- parse */
66
+
67
+ /* Parse `n` newline-separated decimals into integers scaled by 10^dec.
68
+ * Returns 0 on success, -1 on malformed input or overflow. */
69
+ int32_t parse_fixed(const char *buf, int64_t len, int32_t dec,
70
+ int64_t *out, int64_t n)
71
+ {
72
+ static const int64_t POW10[19] = {
73
+ 1LL, 10LL, 100LL, 1000LL, 10000LL, 100000LL, 1000000LL,
74
+ 10000000LL, 100000000LL, 1000000000LL, 10000000000LL,
75
+ 100000000000LL, 1000000000000LL, 10000000000000LL,
76
+ 100000000000000LL, 1000000000000000LL, 10000000000000000LL,
77
+ 100000000000000000LL, 1000000000000000000LL};
78
+
79
+ int64_t i = 0, k = 0;
80
+ while (k < n) {
81
+ int64_t start = i;
82
+ while (i < len && buf[i] != '\n') i++;
83
+ int64_t end = i;
84
+ int64_t p = start;
85
+ int neg = 0;
86
+ int64_t v = 0;
87
+ int seen = 0;
88
+
89
+ /* Overflow has to be caught BEFORE it happens. Checking `v >= LIMIT`
90
+ * after `v = v*10 + d` is too late: signed overflow is undefined, and
91
+ * in practice it wraps negative, so the guard never fires. That let
92
+ * "-9223372036854775808" through as a valid value even though the
93
+ * documented limit is 2^62 -- and the numpy fallback rejected it, so
94
+ * the accelerator and the reference disagreed about what a column
95
+ * even was. Rearranged, the test is exact and never overflows. */
96
+ #define ACC_DIGIT(d) \
97
+ do { \
98
+ int dgt_ = (d); \
99
+ if (v > (LIMIT - 1 - dgt_) / 10) return -1; \
100
+ v = v * 10 + dgt_; \
101
+ } while (0)
102
+
103
+ if (p < end && buf[p] == '-') { neg = 1; p++; }
104
+ while (p < end && buf[p] >= '0' && buf[p] <= '9') {
105
+ ACC_DIGIT(buf[p] - '0');
106
+ p++; seen = 1;
107
+ }
108
+ if (!seen) return -1;
109
+
110
+ int used = 0;
111
+ if (p < end && buf[p] == '.') {
112
+ p++;
113
+ while (p < end && buf[p] >= '0' && buf[p] <= '9') {
114
+ if (used < dec) {
115
+ ACC_DIGIT(buf[p] - '0');
116
+ used++;
117
+ }
118
+ p++;
119
+ }
120
+ }
121
+ if (p != end) return -1;
122
+ while (used < dec) { /* pad to the column scale */
123
+ ACC_DIGIT(0);
124
+ used++;
125
+ }
126
+ #undef ACC_DIGIT
127
+ out[k++] = neg ? -v : v;
128
+
129
+ if (i >= len) break;
130
+ i++;
131
+ }
132
+ return k == n ? 0 : -1;
133
+ }
134
+
135
+ /* ---------------------------------------------------------------- format */
136
+
137
+ /* Write `n` scaled integers back as newline-separated decimal text.
138
+ * Returns bytes written, or -1 if `cap` was too small. */
139
+ int64_t fmt_fixed(const int64_t *in, int64_t n, int32_t dec,
140
+ char *out, int64_t cap)
141
+ {
142
+ char digits[24];
143
+ int64_t at = 0;
144
+
145
+ for (int64_t k = 0; k < n; k++) {
146
+ int64_t v = in[k];
147
+ /* worst case: sign + 19 integer digits + point + 18 decimals + \n */
148
+ if (at + 40 > cap) return -1;
149
+
150
+ if (v < 0) { out[at++] = '-'; v = -v; }
151
+
152
+ int64_t scale = 1;
153
+ for (int32_t d = 0; d < dec; d++) scale *= 10;
154
+ int64_t q = dec ? v / scale : v;
155
+ int64_t r = dec ? v % scale : 0;
156
+
157
+ int nd = 0;
158
+ if (q == 0) digits[nd++] = '0';
159
+ while (q > 0) { digits[nd++] = (char)('0' + (q % 10)); q /= 10; }
160
+ while (nd > 0) out[at++] = digits[--nd];
161
+
162
+ if (dec) {
163
+ out[at++] = '.';
164
+ for (int32_t d = dec - 1; d >= 0; d--) {
165
+ int64_t p = 1;
166
+ for (int32_t e = 0; e < d; e++) p *= 10;
167
+ out[at++] = (char)('0' + ((r / p) % 10));
168
+ }
169
+ }
170
+ out[at++] = '\n';
171
+ }
172
+ return at ? at - 1 : 0; /* drop the final newline */
173
+ }
174
+
175
+ /* ------------------------------------------------------------- varints */
176
+
177
+ /* Zigzag each value; small ones take a byte, the rest an escape plus a
178
+ * 64-bit tail entry. The tail must be 64 bits: a 32-bit tail silently
179
+ * truncated any value past 2^31, which is well inside the int64 range this
180
+ * codec accepts. */
181
+ int64_t pack_ints(const int64_t *in, int64_t n,
182
+ uint8_t *head, uint64_t *tail)
183
+ {
184
+ int64_t nbig = 0;
185
+ for (int64_t i = 0; i < n; i++) {
186
+ int64_t v = in[i];
187
+ uint64_t u = v < 0 ? (uint64_t)(-v) * 2 - 1 : (uint64_t)v * 2;
188
+ if (u < ESCAPE) {
189
+ head[i] = (uint8_t)u;
190
+ } else {
191
+ head[i] = ESCAPE;
192
+ tail[nbig++] = u;
193
+ }
194
+ }
195
+ return nbig;
196
+ }
197
+
198
+ void unpack_ints(const uint8_t *head, int64_t n,
199
+ const uint64_t *tail, int64_t *out)
200
+ {
201
+ int64_t nbig = 0;
202
+ for (int64_t i = 0; i < n; i++) {
203
+ uint64_t u = head[i];
204
+ if (u == ESCAPE) u = tail[nbig++];
205
+ out[i] = (u & 1) ? -(int64_t)((u + 1) / 2) : (int64_t)(u / 2);
206
+ }
207
+ }