herringbone 0.1.0 → 0.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/README.md +198 -74
- data/bin/herringbone +38 -26
- data/lib/herringbone/active_record.rb +48 -19
- data/lib/herringbone/bloom_filter.rb +370 -0
- data/lib/herringbone/byte_values.rb +29 -0
- data/lib/herringbone/codecs/lz4.rb +66 -3
- data/lib/herringbone/codecs/snappy.rb +132 -3
- data/lib/herringbone/compression.rb +121 -16
- data/lib/herringbone/encodings/delta.rb +76 -19
- data/lib/herringbone/encodings/plain.rb +29 -7
- data/lib/herringbone/encodings/rle.rb +55 -3
- data/lib/herringbone/format.rb +203 -0
- data/lib/herringbone/inspector.rb +1784 -0
- data/lib/herringbone/io_buffer_support.rb +3 -1
- data/lib/herringbone/reader/column_chunk_reader.rb +388 -0
- data/lib/herringbone/reader/column_cursor.rb +225 -0
- data/lib/herringbone/reader/numo.rb +365 -0
- data/lib/herringbone/reader/page_stream.rb +615 -0
- data/lib/herringbone/reader/scan.rb +350 -0
- data/lib/herringbone/reader.rb +585 -272
- data/lib/herringbone/schema.rb +326 -90
- data/lib/herringbone/thrift.rb +155 -4
- data/lib/herringbone/types.rb +177 -42
- data/lib/herringbone/version.rb +2 -1
- data/lib/herringbone/visualizer.rb +1136 -0
- data/lib/herringbone/writer.rb +550 -105
- data/lib/herringbone/xxhash.rb +425 -0
- data/lib/herringbone.rb +64 -9
- metadata +14 -33
|
@@ -1,33 +1,103 @@
|
|
|
1
1
|
# frozen_string_literal: true
|
|
2
2
|
|
|
3
3
|
module Herringbone
|
|
4
|
+
# Pure-Ruby compression codecs for Parquet pages: Snappy, and LZ4 (raw blocks and the Hadoop framing).
|
|
5
|
+
# Compression dispatches to these; GZIP uses Zlib and ZSTD/BROTLI use external gems.
|
|
4
6
|
module Codecs
|
|
5
7
|
# Pure-Ruby implementation of the raw Snappy block format (as used by Parquet),
|
|
6
8
|
# see https://github.com/google/snappy/blob/main/format_description.txt
|
|
7
9
|
#
|
|
10
|
+
# When the `snappy` gem (a binding to Google's libsnappy) can be loaded, it is used instead:
|
|
11
|
+
# 12x faster decompression and 27x faster compression. It is optional and only a speedup: if
|
|
12
|
+
# it is missing, the pure-Ruby code below is used silently. Both produce raw Snappy blocks the
|
|
13
|
+
# other reads.
|
|
14
|
+
#
|
|
8
15
|
# 32-bit loads are done with four getbyte calls rather than unpack1(offset:) to stay
|
|
9
16
|
# compatible with Ruby 3.0 - the speed difference on MRI is marginal.
|
|
10
17
|
module Snappy
|
|
18
|
+
# Raised for corrupt or truncated Snappy input, and for input too large to compress.
|
|
11
19
|
class Error < StandardError; end
|
|
12
20
|
|
|
21
|
+
# The compressor works on independent 64 KiB fragments, as the reference implementation does.
|
|
13
22
|
BLOCK_SIZE = 1 << 16
|
|
23
|
+
# Size of the match-finder hash table, in bits (16384 entries).
|
|
14
24
|
HASH_BITS = 14
|
|
25
|
+
# Right shift that keeps the top HASH_BITS bits of the 32-bit multiplicative hash.
|
|
15
26
|
HASH_SHIFT = 32 - HASH_BITS
|
|
27
|
+
# Multiplier of the reference implementation's 4-byte hash.
|
|
16
28
|
HASH_MUL = 0x1e35a7bd
|
|
17
29
|
INPUT_MARGIN = 15 # bytes at the block end never searched for matches, as in the reference
|
|
30
|
+
# The length preamble is a 32-bit varint, so no block may decompress to more than this.
|
|
18
31
|
MAX_UNCOMPRESSED = (1 << 32) - 1
|
|
19
32
|
|
|
33
|
+
# Name of the optional native gem, passed to +require+.
|
|
34
|
+
NATIVE_GEM = "snappy"
|
|
35
|
+
|
|
20
36
|
module_function
|
|
21
37
|
|
|
22
38
|
# @param input [String] raw snappy block
|
|
23
39
|
# @return [String] decompressed bytes in ASCII-8BIT
|
|
40
|
+
# @raise [Error] if the block is corrupt or truncated
|
|
24
41
|
def decompress(input)
|
|
25
|
-
src = input.encoding == Encoding::BINARY ? input : input.b
|
|
42
|
+
src = (input.encoding == Encoding::BINARY) ? input : input.b
|
|
43
|
+
if (lib = native)
|
|
44
|
+
begin
|
|
45
|
+
return lib.inflate(src)
|
|
46
|
+
rescue lib::Error => e
|
|
47
|
+
raise Error, "corrupt snappy data (#{e.message})"
|
|
48
|
+
end
|
|
49
|
+
end
|
|
26
50
|
return decompress_io_buffer(src) if IOBufferSupport::AVAILABLE
|
|
27
51
|
decompress_string(src)
|
|
28
52
|
end
|
|
29
53
|
|
|
54
|
+
# The backend in use: :native (the snappy gem) or :ruby
|
|
55
|
+
# @return [Symbol] +:native+ or +:ruby+
|
|
56
|
+
def backend
|
|
57
|
+
native ? :native : :ruby
|
|
58
|
+
end
|
|
59
|
+
|
|
60
|
+
# For tests and benchmarks: :ruby forces pure Ruby, :native requires the snappy gem
|
|
61
|
+
# (UnsupportedError if it cannot be loaded), nil goes back to the default
|
|
62
|
+
# @param name [Symbol, nil] +:ruby+, +:native+ or nil
|
|
63
|
+
# @return [void]
|
|
64
|
+
# @raise [UnsupportedError] if +:native+ is requested and the gem cannot be loaded
|
|
65
|
+
# @raise [ArgumentError] for any other backend name
|
|
66
|
+
def backend=(name)
|
|
67
|
+
@native = case name
|
|
68
|
+
when :ruby then false
|
|
69
|
+
when :native
|
|
70
|
+
native_library || raise(UnsupportedError, "The \"#{NATIVE_GEM}\" gem could not be loaded")
|
|
71
|
+
when nil then nil
|
|
72
|
+
else raise ArgumentError, "Unknown Snappy backend #{name.inspect} (expected :ruby, :native or nil)"
|
|
73
|
+
end
|
|
74
|
+
end
|
|
75
|
+
|
|
76
|
+
# The native library to use, resolved (and memoized) on first call unless forced by backend=.
|
|
77
|
+
# @return [Module, nil] the +::Snappy+ module, or nil when the pure-Ruby code should run
|
|
78
|
+
def native
|
|
79
|
+
@native = native_library || false if @native.nil?
|
|
80
|
+
@native || nil
|
|
81
|
+
end
|
|
82
|
+
|
|
83
|
+
# Tries to load the snappy gem once and memoizes the result.
|
|
84
|
+
# @return [Module, false] the +::Snappy+ module, or false if it is missing or lacks inflate/deflate
|
|
85
|
+
def native_library
|
|
86
|
+
if @native_lib.nil?
|
|
87
|
+
@native_lib = begin
|
|
88
|
+
require NATIVE_GEM
|
|
89
|
+
(::Snappy.respond_to?(:inflate) && ::Snappy.respond_to?(:deflate)) ? ::Snappy : false
|
|
90
|
+
rescue LoadError
|
|
91
|
+
false
|
|
92
|
+
end
|
|
93
|
+
end
|
|
94
|
+
@native_lib
|
|
95
|
+
end
|
|
96
|
+
|
|
30
97
|
# Decompresses into a preallocated IO::Buffer: copies do not allocate intermediate Strings.
|
|
98
|
+
# @param src [String] raw snappy block in ASCII-8BIT
|
|
99
|
+
# @return [String] decompressed bytes in ASCII-8BIT
|
|
100
|
+
# @raise [Error] if the block is corrupt or truncated
|
|
31
101
|
def decompress_io_buffer(src)
|
|
32
102
|
n = src.bytesize
|
|
33
103
|
expected, pos = read_varint(src, n)
|
|
@@ -98,6 +168,10 @@ module Herringbone
|
|
|
98
168
|
inbuf&.free
|
|
99
169
|
end
|
|
100
170
|
|
|
171
|
+
# Decompresses by appending to a String; the fallback when IO::Buffer is not available.
|
|
172
|
+
# @param src [String] raw snappy block in ASCII-8BIT
|
|
173
|
+
# @return [String] decompressed bytes in ASCII-8BIT
|
|
174
|
+
# @raise [Error] if the block is corrupt or truncated
|
|
101
175
|
def decompress_string(src)
|
|
102
176
|
n = src.bytesize
|
|
103
177
|
expected, pos = read_varint(src, n)
|
|
@@ -162,10 +236,15 @@ module Herringbone
|
|
|
162
236
|
|
|
163
237
|
# @param input [String] bytes to compress
|
|
164
238
|
# @return [String] raw snappy block in ASCII-8BIT
|
|
239
|
+
# @raise [Error] if the input is larger than MAX_UNCOMPRESSED
|
|
165
240
|
def compress(input)
|
|
166
|
-
src = input.encoding == Encoding::BINARY ? input : input.b
|
|
241
|
+
src = (input.encoding == Encoding::BINARY) ? input : input.b
|
|
167
242
|
n = src.bytesize
|
|
243
|
+
# Checked for libsnappy too: it would truncate the 32-bit length preamble without complaint
|
|
168
244
|
raise Error, "input too large for snappy" if n > MAX_UNCOMPRESSED
|
|
245
|
+
if (lib = native)
|
|
246
|
+
return lib.deflate(src)
|
|
247
|
+
end
|
|
169
248
|
|
|
170
249
|
out = String.new(capacity: 32 + n + n / 6, encoding: Encoding::BINARY)
|
|
171
250
|
write_varint(out, n)
|
|
@@ -181,6 +260,11 @@ module Herringbone
|
|
|
181
260
|
out
|
|
182
261
|
end
|
|
183
262
|
|
|
263
|
+
# Reads the uncompressed-length preamble (a little-endian base-128 varint) at the start of +src+.
|
|
264
|
+
# @param src [String] raw snappy block
|
|
265
|
+
# @param n [Integer] byte size of +src+
|
|
266
|
+
# @return [Array(Integer, Integer)] the declared uncompressed length and the offset just past the varint
|
|
267
|
+
# @raise [Error] if the varint is truncated, longer than 5 bytes or exceeds MAX_UNCOMPRESSED
|
|
184
268
|
def read_varint(src, n)
|
|
185
269
|
value = 0
|
|
186
270
|
shift = 0
|
|
@@ -198,6 +282,10 @@ module Herringbone
|
|
|
198
282
|
[value, pos]
|
|
199
283
|
end
|
|
200
284
|
|
|
285
|
+
# Appends +value+ as a little-endian base-128 varint (the length preamble).
|
|
286
|
+
# @param out [String] binary output buffer, appended to
|
|
287
|
+
# @param value [Integer] non-negative length to encode
|
|
288
|
+
# @return [String] +out+
|
|
201
289
|
def write_varint(out, value)
|
|
202
290
|
while value >= 0x80
|
|
203
291
|
out << ((value & 0x7f) | 0x80)
|
|
@@ -210,6 +298,12 @@ module Herringbone
|
|
|
210
298
|
# positions relative to `base`; matches never cross the block boundary.
|
|
211
299
|
# The 4-byte little-endian word at every position of the block is unpacked up front
|
|
212
300
|
# (in C, via String#unpack) so hashing and match checks are single Array lookups.
|
|
301
|
+
# @param src [String] whole input in ASCII-8BIT
|
|
302
|
+
# @param base [Integer] offset of the fragment in +src+
|
|
303
|
+
# @param len [Integer] fragment length, at most BLOCK_SIZE
|
|
304
|
+
# @param out [String] binary output buffer, appended to
|
|
305
|
+
# @param table [Array<Integer>] zeroed hash table of 1 << HASH_BITS fragment-relative positions
|
|
306
|
+
# @return [void]
|
|
213
307
|
def compress_block(src, base, len, out, table)
|
|
214
308
|
ip_end = base + len
|
|
215
309
|
next_emit = base
|
|
@@ -280,12 +374,24 @@ module Herringbone
|
|
|
280
374
|
|
|
281
375
|
# words[k][j] is the 4-byte little-endian value at base + 4 * j + k, so the word at
|
|
282
376
|
# relative position i is words[i & 3][i >> 2]
|
|
377
|
+
# @param src [String] whole input in ASCII-8BIT
|
|
378
|
+
# @param base [Integer] offset of the fragment in +src+
|
|
379
|
+
# @param len [Integer] fragment length
|
|
380
|
+
# @return [Array<Array<Integer>>] four arrays of 32-bit words, one per byte phase
|
|
283
381
|
def block_words(src, base, len)
|
|
284
382
|
Array.new(4) { |k| src.byteslice(base + k, len - k).unpack("V*") }
|
|
285
383
|
end
|
|
286
384
|
|
|
287
385
|
# Like match_length, but compares 4 bytes at a time using the unpacked block words,
|
|
288
386
|
# which avoids allocating substrings for the (common) short matches.
|
|
387
|
+
# Falls back to match_length once a match reaches 64 bytes.
|
|
388
|
+
# @param words [Array<Array<Integer>>] fragment words from block_words
|
|
389
|
+
# @param src [String] whole input in ASCII-8BIT
|
|
390
|
+
# @param base [Integer] offset of the fragment in +src+
|
|
391
|
+
# @param s1 [Integer] absolute offset of the earlier occurrence
|
|
392
|
+
# @param s2 [Integer] absolute offset of the current position (s1 < s2)
|
|
393
|
+
# @param limit [Integer] absolute offset not to read past (the fragment end)
|
|
394
|
+
# @return [Integer] number of equal bytes starting at s1 and s2
|
|
289
395
|
def match_length_words(words, src, base, s1, s2, limit)
|
|
290
396
|
start = s2
|
|
291
397
|
r1 = s1 - base
|
|
@@ -307,6 +413,11 @@ module Herringbone
|
|
|
307
413
|
|
|
308
414
|
# Number of equal bytes at s1 and s2 (s1 < s2), not reading past limit. Gallops
|
|
309
415
|
# with byteslice comparisons (memcmp) to avoid per-byte loops on long matches.
|
|
416
|
+
# @param src [String] whole input in ASCII-8BIT
|
|
417
|
+
# @param s1 [Integer] absolute offset of the earlier occurrence
|
|
418
|
+
# @param s2 [Integer] absolute offset of the current position
|
|
419
|
+
# @param limit [Integer] absolute offset not to read past
|
|
420
|
+
# @return [Integer] number of equal bytes
|
|
310
421
|
def match_length(src, s1, s2, limit)
|
|
311
422
|
start = s2
|
|
312
423
|
return 0 if s2 >= limit || src.getbyte(s1) != src.getbyte(s2)
|
|
@@ -330,6 +441,13 @@ module Herringbone
|
|
|
330
441
|
s2 - start
|
|
331
442
|
end
|
|
332
443
|
|
|
444
|
+
# Appends a literal element: the tag (with a 1-4 byte length extension for long literals)
|
|
445
|
+
# followed by the bytes themselves. Does nothing for an empty literal.
|
|
446
|
+
# @param out [String] binary output buffer, appended to
|
|
447
|
+
# @param src [String] whole input in ASCII-8BIT
|
|
448
|
+
# @param pos [Integer] offset of the literal bytes in +src+
|
|
449
|
+
# @param len [Integer] number of literal bytes
|
|
450
|
+
# @return [String, nil] +out+, or nil when +len+ is zero
|
|
333
451
|
def emit_literal(out, src, pos, len)
|
|
334
452
|
return if len == 0
|
|
335
453
|
n = len - 1
|
|
@@ -348,6 +466,11 @@ module Herringbone
|
|
|
348
466
|
end
|
|
349
467
|
|
|
350
468
|
# Offsets are always < 64KB (matches stay within a block), so 4-byte offsets are never needed.
|
|
469
|
+
# Long matches are split into copies of at most 64 bytes, never leaving a remainder under 4.
|
|
470
|
+
# @param out [String] binary output buffer, appended to
|
|
471
|
+
# @param offset [Integer] backward distance to the match source, 1...65536
|
|
472
|
+
# @param len [Integer] match length, at least 4
|
|
473
|
+
# @return [String] +out+
|
|
351
474
|
def emit_copy(out, offset, len)
|
|
352
475
|
while len >= 68
|
|
353
476
|
emit_copy_upto64(out, offset, 64)
|
|
@@ -360,6 +483,12 @@ module Herringbone
|
|
|
360
483
|
emit_copy_upto64(out, offset, len)
|
|
361
484
|
end
|
|
362
485
|
|
|
486
|
+
# Appends one copy element: the 2-byte form (1-byte offset) when len is 4..11 and offset < 2048,
|
|
487
|
+
# otherwise the 3-byte form with a 2-byte offset.
|
|
488
|
+
# @param out [String] binary output buffer, appended to
|
|
489
|
+
# @param offset [Integer] backward distance to the match source, 1...65536
|
|
490
|
+
# @param len [Integer] copy length, 4..64
|
|
491
|
+
# @return [String] +out+
|
|
363
492
|
def emit_copy_upto64(out, offset, len)
|
|
364
493
|
if len < 12 && offset < 2048
|
|
365
494
|
out << (1 | ((len - 4) << 2) | ((offset >> 8) << 5)) << (offset & 0xff)
|
|
@@ -368,7 +497,7 @@ module Herringbone
|
|
|
368
497
|
end
|
|
369
498
|
end
|
|
370
499
|
|
|
371
|
-
private_class_method :decompress_io_buffer, :decompress_string, :read_varint, :write_varint, :compress_block, :block_words, :match_length_words,
|
|
500
|
+
private_class_method :native, :native_library, :decompress_io_buffer, :decompress_string, :read_varint, :write_varint, :compress_block, :block_words, :match_length_words,
|
|
372
501
|
:match_length, :emit_literal, :emit_copy, :emit_copy_upto64
|
|
373
502
|
end
|
|
374
503
|
end
|
|
@@ -2,27 +2,121 @@
|
|
|
2
2
|
|
|
3
3
|
require "zlib"
|
|
4
4
|
require "stringio"
|
|
5
|
-
require "zstd-ruby"
|
|
6
|
-
require "brotli"
|
|
7
5
|
|
|
8
6
|
module Herringbone
|
|
9
|
-
#
|
|
10
|
-
|
|
7
|
+
# Raised when a file uses (or a writer asks for) a codec whose library is not installed
|
|
8
|
+
class MissingCodecError < UnsupportedError
|
|
9
|
+
# @return [String] codec name (+"ZSTD"+) and name of the gem providing it (+"zstd-ruby"+)
|
|
10
|
+
attr_reader :codec, :gem_name
|
|
11
|
+
|
|
12
|
+
# @param codec [String] codec name, as shown in the message
|
|
13
|
+
# @param gem_name [String] gem to add to the Gemfile
|
|
14
|
+
# @param load_error [LoadError] the error from requiring the gem, whose message is included
|
|
15
|
+
def initialize(codec, gem_name, load_error)
|
|
16
|
+
@codec = codec
|
|
17
|
+
@gem_name = gem_name
|
|
18
|
+
super("#{codec} compression needs the \"#{gem_name}\" gem, which could not be loaded " \
|
|
19
|
+
"(#{load_error.message}). Add `gem \"#{gem_name}\"` to your Gemfile to use #{codec}.")
|
|
20
|
+
end
|
|
21
|
+
end
|
|
22
|
+
|
|
23
|
+
# Dispatches page (de)compression by Parquet codec id. Snappy and LZ4 are pure Ruby and GZIP
|
|
24
|
+
# uses zlib, so those always work. ZSTD and Brotli come from optional gems (zstd-ruby, brotli),
|
|
25
|
+
# which are required on first use; if they are missing, MissingCodecError says what to add.
|
|
26
|
+
# (Snappy also uses the optional snappy gem when it is installed, see Codecs::Snappy.)
|
|
11
27
|
module Compression
|
|
12
28
|
module_function
|
|
13
29
|
|
|
14
|
-
|
|
15
|
-
|
|
16
|
-
|
|
17
|
-
|
|
18
|
-
|
|
30
|
+
# Codecs backed by optional native gems: codec id => [name, gem, require path, constant]
|
|
31
|
+
LIBRARIES = {
|
|
32
|
+
Format::Codec::ZSTD => ["ZSTD", "zstd-ruby", "zstd-ruby", :Zstd],
|
|
33
|
+
Format::Codec::BROTLI => ["Brotli", "brotli", "brotli", :Brotli]
|
|
34
|
+
}.freeze
|
|
35
|
+
|
|
36
|
+
@libraries = {}
|
|
37
|
+
@library_mutex = Mutex.new
|
|
38
|
+
|
|
39
|
+
# The library module for a codec backed by an optional gem, requiring it on first use
|
|
40
|
+
#
|
|
41
|
+
# @param codec [Integer] a codec id that is a key of LIBRARIES
|
|
42
|
+
# @return [Module] the gem's module (+Zstd+ or +Brotli+)
|
|
43
|
+
# @raise [MissingCodecError] when the gem cannot be loaded
|
|
44
|
+
# @raise [KeyError] for a codec id not in LIBRARIES
|
|
45
|
+
def library(codec)
|
|
46
|
+
@libraries.fetch(codec) do
|
|
47
|
+
@library_mutex.synchronize do
|
|
48
|
+
@libraries.fetch(codec) do
|
|
49
|
+
name, gem_name, path, const = LIBRARIES.fetch(codec)
|
|
50
|
+
begin
|
|
51
|
+
require_library(path)
|
|
52
|
+
rescue LoadError => e
|
|
53
|
+
raise MissingCodecError.new(name, gem_name, e)
|
|
54
|
+
end
|
|
55
|
+
@libraries[codec] = Object.const_get(const)
|
|
56
|
+
end
|
|
57
|
+
end
|
|
58
|
+
end
|
|
59
|
+
end
|
|
60
|
+
|
|
61
|
+
# Requires a codec gem; a separate method so tests can stub it to simulate a missing gem
|
|
62
|
+
#
|
|
63
|
+
# @param path [String] require path of the gem
|
|
64
|
+
# @return [Boolean] whether the library was newly loaded
|
|
65
|
+
# @raise [LoadError] when the gem is not installed
|
|
66
|
+
def require_library(path)
|
|
67
|
+
require path
|
|
68
|
+
end
|
|
69
|
+
|
|
70
|
+
# Raises MissingCodecError (or UnsupportedError) unless +codec+ can be used
|
|
71
|
+
#
|
|
72
|
+
# @param codec [Integer, Symbol, String] codec id or name (see NAMES)
|
|
73
|
+
# @return [Module, nil] the gem's module for a gem-backed codec, nil for a built-in one
|
|
74
|
+
# @raise [MissingCodecError] when the codec's gem cannot be loaded
|
|
75
|
+
# @raise [UnsupportedError] for a codec herringbone does not implement (LZO)
|
|
76
|
+
# @raise [ArgumentError] for an unknown codec name
|
|
77
|
+
def ensure_available!(codec)
|
|
78
|
+
codec = codec_id(codec)
|
|
79
|
+
return library(codec) if LIBRARIES.key?(codec)
|
|
80
|
+
return if SUPPORTED.include?(codec)
|
|
81
|
+
raise UnsupportedError, "#{Format::Codec::NAMES[codec] || codec} compression is not supported"
|
|
82
|
+
end
|
|
83
|
+
|
|
84
|
+
# Codec ids that work without optional gems
|
|
85
|
+
SUPPORTED = [
|
|
86
|
+
Format::Codec::UNCOMPRESSED, Format::Codec::SNAPPY, Format::Codec::GZIP,
|
|
87
|
+
Format::Codec::LZ4_RAW, Format::Codec::LZ4
|
|
88
|
+
].freeze
|
|
89
|
+
|
|
90
|
+
# Codec id => name, as given to the writer's compression: option and listed by Herringbone.codecs
|
|
91
|
+
NAMES = {
|
|
92
|
+
Format::Codec::UNCOMPRESSED => :none, Format::Codec::SNAPPY => :snappy, Format::Codec::GZIP => :gzip,
|
|
93
|
+
Format::Codec::LZ4_RAW => :lz4, Format::Codec::LZ4 => :lz4_hadoop, Format::Codec::ZSTD => :zstd,
|
|
94
|
+
Format::Codec::BROTLI => :brotli, Format::Codec::LZO => :lzo
|
|
19
95
|
}.freeze
|
|
96
|
+
# Codec name => codec id, the inverse of NAMES
|
|
97
|
+
CODECS_BY_NAME = NAMES.invert.freeze
|
|
20
98
|
|
|
99
|
+
# Codec id for a codec name; Integers are taken to be ids already and returned unchecked
|
|
100
|
+
#
|
|
101
|
+
# @param name [Integer, Symbol, String] codec id, or a name from NAMES (case-insensitive)
|
|
102
|
+
# @return [Integer] the Format::Codec id
|
|
103
|
+
# @raise [ArgumentError] for an unknown name
|
|
21
104
|
def codec_id(name)
|
|
22
105
|
return name if name.is_a?(Integer)
|
|
23
|
-
CODECS_BY_NAME.fetch(name.to_s.downcase.to_sym)
|
|
106
|
+
CODECS_BY_NAME.fetch(name.to_s.downcase.to_sym) do
|
|
107
|
+
raise ArgumentError, "Unknown compression codec #{name.inspect}, expected one of #{NAMES.values.map(&:inspect).join(", ")}"
|
|
108
|
+
end
|
|
24
109
|
end
|
|
25
110
|
|
|
111
|
+
# Decompresses a page body, checking the result against the size the page header declares
|
|
112
|
+
#
|
|
113
|
+
# @param codec [Integer] Format::Codec id from the column chunk metadata
|
|
114
|
+
# @param data [String] compressed bytes
|
|
115
|
+
# @param uncompressed_size [Integer] expected decompressed size in bytes
|
|
116
|
+
# @return [String] decompressed bytes (binary)
|
|
117
|
+
# @raise [UnsupportedError] for a codec herringbone does not implement
|
|
118
|
+
# @raise [MissingCodecError] when the codec's gem cannot be loaded
|
|
119
|
+
# @raise [FormatError] when the decompressed size does not match +uncompressed_size+
|
|
26
120
|
def decompress(codec, data, uncompressed_size)
|
|
27
121
|
return "".b if uncompressed_size.zero? && data.empty?
|
|
28
122
|
out = case codec
|
|
@@ -31,18 +125,25 @@ module Herringbone
|
|
|
31
125
|
when Format::Codec::GZIP then gunzip(data)
|
|
32
126
|
when Format::Codec::LZ4_RAW then Codecs::LZ4.decompress_block(data, uncompressed_size)
|
|
33
127
|
when Format::Codec::LZ4 then Codecs::LZ4.decompress_hadoop(data, uncompressed_size)
|
|
34
|
-
when Format::Codec::ZSTD then
|
|
35
|
-
when Format::Codec::BROTLI then
|
|
128
|
+
when Format::Codec::ZSTD then library(codec).decompress(data)
|
|
129
|
+
when Format::Codec::BROTLI then library(codec).inflate(data)
|
|
36
130
|
else
|
|
37
131
|
raise UnsupportedError, "Unsupported compression codec #{Format::Codec::NAMES[codec] || codec}"
|
|
38
132
|
end
|
|
39
133
|
out = out.b
|
|
40
134
|
if out.bytesize != uncompressed_size
|
|
41
|
-
raise
|
|
135
|
+
raise FormatError, "Decompressed #{out.bytesize} bytes, expected #{uncompressed_size}"
|
|
42
136
|
end
|
|
43
137
|
out
|
|
44
138
|
end
|
|
45
139
|
|
|
140
|
+
# Compresses a page body
|
|
141
|
+
#
|
|
142
|
+
# @param codec [Integer] Format::Codec id
|
|
143
|
+
# @param data [String] bytes to compress
|
|
144
|
+
# @return [String] compressed bytes (binary)
|
|
145
|
+
# @raise [UnsupportedError] for a codec herringbone does not implement
|
|
146
|
+
# @raise [MissingCodecError] when the codec's gem cannot be loaded
|
|
46
147
|
def compress(codec, data)
|
|
47
148
|
case codec
|
|
48
149
|
when Format::Codec::UNCOMPRESSED then data
|
|
@@ -50,18 +151,22 @@ module Herringbone
|
|
|
50
151
|
when Format::Codec::GZIP then Zlib.gzip(data)
|
|
51
152
|
when Format::Codec::LZ4_RAW then Codecs::LZ4.compress_block(data)
|
|
52
153
|
when Format::Codec::LZ4 then Codecs::LZ4.compress_hadoop(data)
|
|
53
|
-
when Format::Codec::ZSTD then
|
|
54
|
-
when Format::Codec::BROTLI then
|
|
154
|
+
when Format::Codec::ZSTD then library(codec).compress(data)
|
|
155
|
+
when Format::Codec::BROTLI then library(codec).deflate(data)
|
|
55
156
|
else
|
|
56
157
|
raise UnsupportedError, "Unsupported compression codec #{Format::Codec::NAMES[codec] || codec}"
|
|
57
158
|
end.b
|
|
58
159
|
end
|
|
59
160
|
|
|
60
161
|
# Handles files whose gzip data consists of several concatenated members
|
|
162
|
+
#
|
|
163
|
+
# @param data [String] gzip data, one or more members
|
|
164
|
+
# @return [String] the members' decompressed bytes concatenated (binary)
|
|
165
|
+
# @raise [Zlib::GzipFile::Error] on corrupt gzip data
|
|
61
166
|
def gunzip(data)
|
|
62
167
|
io = StringIO.new(data)
|
|
63
168
|
out = String.new(encoding: Encoding::BINARY)
|
|
64
|
-
|
|
169
|
+
while true
|
|
65
170
|
gz = Zlib::GzipReader.new(io)
|
|
66
171
|
out << gz.read
|
|
67
172
|
unused = gz.unused
|
|
@@ -6,56 +6,84 @@ module Herringbone
|
|
|
6
6
|
module Delta
|
|
7
7
|
module_function
|
|
8
8
|
|
|
9
|
+
# Values per block written by the encoder (the size parquet-mr and Arrow use).
|
|
9
10
|
BLOCK_SIZE = 128
|
|
11
|
+
# Miniblocks per block written by the encoder, so 32 values per miniblock.
|
|
10
12
|
MINIBLOCKS = 4
|
|
11
13
|
|
|
14
|
+
# @param n [Integer] zigzag-encoded unsigned integer
|
|
15
|
+
# @return [Integer] the signed integer it represents
|
|
12
16
|
def zigzag_decode(n) = (n >> 1) ^ -(n & 1)
|
|
17
|
+
|
|
18
|
+
# @param n [Integer] signed integer
|
|
19
|
+
# @return [Integer] zigzag-encoded form (0, -1, 1, -2 map to 0, 1, 2, 3)
|
|
13
20
|
def zigzag_encode(n) = n.negative? ? ((-n) << 1) - 1 : n << 1
|
|
14
21
|
|
|
15
22
|
# Wraps an integer into the signed range of +bits+ bits
|
|
23
|
+
# @param v [Integer] integer to wrap
|
|
24
|
+
# @param bits [Integer] target width, 32 or 64
|
|
25
|
+
# @return [Integer] +v+ modulo 2^bits, as a two's complement signed integer
|
|
16
26
|
def wrap(v, bits)
|
|
17
27
|
half = 1 << (bits - 1)
|
|
18
28
|
((v + half) & ((1 << bits) - 1)) - half
|
|
19
29
|
end
|
|
20
30
|
|
|
21
31
|
# Decodes DELTA_BINARY_PACKED integers. +bits+ is 32 or 64 (for wraparound).
|
|
22
|
-
# Returns [values, new_pos]. If +count+ is nil, the total from the header is used.
|
|
32
|
+
# Returns [values, new_pos]. If +count+ is nil, the total from the header is used. A smaller
|
|
33
|
+
# +count+ decodes only that many values, but the returned offset is still the end of the
|
|
34
|
+
# whole encoded block, so data following it can be read from there.
|
|
35
|
+
# @param data [String] binary page data
|
|
36
|
+
# @param pos [Integer] byte offset of the block header
|
|
37
|
+
# @param bits [Integer] integer width that deltas wrap around in, 32 or 64
|
|
38
|
+
# @param count [Integer, nil] maximum number of values to decode
|
|
39
|
+
# @return [Array(Array<Integer>, Integer)] the decoded values and the offset just past the encoded block
|
|
40
|
+
# @raise [FormatError] on an invalid header or miniblock bit width, or truncated data
|
|
23
41
|
def decode_binary_packed(data, pos, bits = 64, count = nil)
|
|
24
42
|
block_size, pos = RLE.read_uleb(data, pos)
|
|
25
43
|
miniblocks, pos = RLE.read_uleb(data, pos)
|
|
26
44
|
total, pos = RLE.read_uleb(data, pos)
|
|
27
45
|
first, pos = RLE.read_uleb(data, pos)
|
|
28
|
-
raise
|
|
46
|
+
raise FormatError, "Invalid DELTA_BINARY_PACKED header" if miniblocks.zero? || block_size % miniblocks != 0
|
|
29
47
|
per_mini = block_size / miniblocks
|
|
30
|
-
raise
|
|
31
|
-
|
|
48
|
+
raise FormatError, "Invalid miniblock size #{per_mini}" if per_mini % 8 != 0
|
|
49
|
+
want = (count && count < total) ? count : total
|
|
32
50
|
values = []
|
|
33
51
|
return [values, pos] if total.zero?
|
|
34
52
|
last = zigzag_decode(first)
|
|
35
|
-
values << last
|
|
53
|
+
values << last if want.positive?
|
|
36
54
|
half = 1 << (bits - 1)
|
|
37
55
|
mask = (1 << bits) - 1
|
|
38
|
-
|
|
56
|
+
left = total - 1 # deltas still encoded, decoded or skipped
|
|
57
|
+
while left.positive?
|
|
39
58
|
min_delta, pos = RLE.read_uleb(data, pos)
|
|
40
59
|
min_delta = zigzag_decode(min_delta)
|
|
41
60
|
widths = data.byteslice(pos, miniblocks).unpack("C*")
|
|
42
61
|
pos += miniblocks
|
|
43
62
|
widths.each do |w|
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
|
|
63
|
+
# Miniblocks past the last value have a width byte but no body
|
|
64
|
+
break unless left.positive?
|
|
65
|
+
raise FormatError, "Invalid delta bit width #{w}" if w > bits
|
|
66
|
+
if values.size < want
|
|
67
|
+
deltas = RLE.unpack_bits(data, pos, per_mini, w)
|
|
68
|
+
take = want - values.size
|
|
69
|
+
deltas = deltas.first(take) if take < per_mini
|
|
70
|
+
deltas.each do |d|
|
|
71
|
+
last = ((last + min_delta + d + half) & mask) - half
|
|
72
|
+
values << last
|
|
73
|
+
end
|
|
53
74
|
end
|
|
75
|
+
pos += per_mini * w / 8
|
|
76
|
+
left -= per_mini
|
|
54
77
|
end
|
|
55
78
|
end
|
|
56
79
|
[values, pos]
|
|
57
80
|
end
|
|
58
81
|
|
|
82
|
+
# Encodes integers with DELTA_BINARY_PACKED, using BLOCK_SIZE values per block in MINIBLOCKS
|
|
83
|
+
# miniblocks. Deltas wrap at +bits+ so INT32 columns never need more than 32-bit widths.
|
|
84
|
+
# @param values [Array<Integer>] signed integers that fit in +bits+ bits
|
|
85
|
+
# @param bits [Integer] physical integer width, 32 or 64
|
|
86
|
+
# @return [String] encoded bytes in ASCII-8BIT
|
|
59
87
|
def encode_binary_packed(values, bits = 64)
|
|
60
88
|
out = String.new(encoding: Encoding::BINARY)
|
|
61
89
|
per_mini = BLOCK_SIZE / MINIBLOCKS
|
|
@@ -82,35 +110,54 @@ module Herringbone
|
|
|
82
110
|
out
|
|
83
111
|
end
|
|
84
112
|
|
|
113
|
+
# Decodes DELTA_LENGTH_BYTE_ARRAY: DELTA_BINARY_PACKED lengths followed by the concatenated bytes.
|
|
114
|
+
# @param data [String] binary page data
|
|
115
|
+
# @param pos [Integer] byte offset of the lengths block
|
|
116
|
+
# @param count [Integer] number of values to decode
|
|
117
|
+
# @return [Array(Array<String>, Integer)] binary slices of +data+ and the offset just past them
|
|
118
|
+
# @raise [FormatError] if there are too few lengths or a value runs past the end of +data+
|
|
85
119
|
def decode_length_byte_array(data, pos, count)
|
|
86
120
|
lengths, pos = decode_binary_packed(data, pos, 32, count)
|
|
87
|
-
raise
|
|
121
|
+
raise FormatError, "DELTA_LENGTH_BYTE_ARRAY has #{lengths.size} lengths, need #{count}" if lengths.size < count
|
|
88
122
|
out = Array.new(count)
|
|
89
123
|
size = data.bytesize
|
|
90
124
|
lengths.each_with_index do |len, i|
|
|
91
|
-
raise
|
|
125
|
+
raise FormatError, "DELTA_LENGTH_BYTE_ARRAY value overruns page" if len.negative? || pos + len > size
|
|
92
126
|
out[i] = data.byteslice(pos, len)
|
|
93
127
|
pos += len
|
|
94
128
|
end
|
|
95
129
|
[out, pos]
|
|
96
130
|
end
|
|
97
131
|
|
|
132
|
+
# Encodes byte strings with DELTA_LENGTH_BYTE_ARRAY.
|
|
133
|
+
# @param values [Array<String>] byte strings, in any encoding
|
|
134
|
+
# @return [String] encoded bytes in ASCII-8BIT
|
|
98
135
|
def encode_length_byte_array(values)
|
|
99
136
|
encode_binary_packed(values.map(&:bytesize), 32) << values.pack("a*" * values.size)
|
|
100
137
|
end
|
|
101
138
|
|
|
139
|
+
# Decodes DELTA_BYTE_ARRAY (incremental encoding): DELTA_BINARY_PACKED prefix lengths, then
|
|
140
|
+
# the suffixes as DELTA_LENGTH_BYTE_ARRAY. Each value is the previous value's prefix plus its suffix.
|
|
141
|
+
# @param data [String] binary page data
|
|
142
|
+
# @param pos [Integer] byte offset of the prefix lengths block
|
|
143
|
+
# @param count [Integer] number of values to decode
|
|
144
|
+
# @return [Array(Array<String>, Integer)] binary values and the offset just past them
|
|
145
|
+
# @raise [FormatError] if a prefix is negative or longer than the previous value, or the data is malformed
|
|
102
146
|
def decode_byte_array(data, pos, count)
|
|
103
147
|
prefixes, pos = decode_binary_packed(data, pos, 32, count)
|
|
104
148
|
suffixes, pos = decode_length_byte_array(data, pos, count)
|
|
105
149
|
prev = "".b
|
|
106
150
|
out = Array.new(count) do |i|
|
|
107
151
|
prefix = prefixes[i]
|
|
108
|
-
raise
|
|
152
|
+
raise FormatError, "DELTA_BYTE_ARRAY prefix length #{prefix} out of range" if prefix > prev.bytesize || prefix.negative?
|
|
109
153
|
prev = prefix.zero? ? suffixes[i] : prev.byteslice(0, prefix) + suffixes[i]
|
|
110
154
|
end
|
|
111
155
|
[out, pos]
|
|
112
156
|
end
|
|
113
157
|
|
|
158
|
+
# Encodes byte strings with DELTA_BYTE_ARRAY, sharing the common byte prefix with the previous value.
|
|
159
|
+
# @param values [Array<String>] byte strings, in any encoding
|
|
160
|
+
# @return [String] encoded bytes in ASCII-8BIT
|
|
114
161
|
def encode_byte_array(values)
|
|
115
162
|
prev = "".b
|
|
116
163
|
prefixes = []
|
|
@@ -132,13 +179,23 @@ module Herringbone
|
|
|
132
179
|
module_function
|
|
133
180
|
|
|
134
181
|
# Returns the value bytes re-interleaved into PLAIN layout
|
|
182
|
+
# @param data [String] binary page data
|
|
183
|
+
# @param pos [Integer] byte offset of the first stream
|
|
184
|
+
# @param count [Integer] number of values
|
|
185
|
+
# @param width [Integer] byte width of one value
|
|
186
|
+
# @return [Array(String, Integer)] PLAIN-layout bytes and the offset just past the streams
|
|
187
|
+
# @raise [FormatError] if fewer than count * width bytes remain
|
|
135
188
|
def decode(data, pos, count, width)
|
|
136
189
|
nbytes = count * width
|
|
137
|
-
raise
|
|
190
|
+
raise FormatError, "Truncated BYTE_STREAM_SPLIT data" if pos + nbytes > data.bytesize
|
|
138
191
|
streams = Array.new(width) { |k| data.byteslice(pos + k * count, count).unpack("C*") }
|
|
139
192
|
[streams[0].zip(*streams[1..]).flatten.pack("C*"), pos + nbytes]
|
|
140
193
|
end
|
|
141
194
|
|
|
195
|
+
# Splits PLAIN-layout fixed-width values into +width+ byte streams. Inverse of decode.
|
|
196
|
+
# @param plain [String] PLAIN-encoded values, a multiple of +width+ bytes
|
|
197
|
+
# @param width [Integer] byte width of one value
|
|
198
|
+
# @return [String] the concatenated streams in ASCII-8BIT
|
|
142
199
|
def encode(plain, width)
|
|
143
200
|
count = plain.bytesize / width
|
|
144
201
|
bytes = plain.unpack("C*")
|