herringbone 0.2.0 → 0.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/README.md +56 -4
- data/lib/herringbone/active_record.rb +44 -12
- data/lib/herringbone/bloom_filter.rb +112 -12
- data/lib/herringbone/byte_values.rb +29 -0
- data/lib/herringbone/codecs/lz4.rb +66 -3
- data/lib/herringbone/codecs/snappy.rb +132 -3
- data/lib/herringbone/compression.rb +48 -0
- data/lib/herringbone/encodings/delta.rb +70 -13
- data/lib/herringbone/encodings/plain.rb +22 -0
- data/lib/herringbone/encodings/rle.rb +53 -1
- data/lib/herringbone/format.rb +150 -0
- data/lib/herringbone/inspector.rb +477 -81
- data/lib/herringbone/io_buffer_support.rb +3 -1
- data/lib/herringbone/reader/column_chunk_reader.rb +98 -6
- data/lib/herringbone/reader/column_cursor.rb +46 -2
- data/lib/herringbone/reader/numo.rb +365 -0
- data/lib/herringbone/reader/page_stream.rb +280 -4
- data/lib/herringbone/reader/scan.rb +103 -9
- data/lib/herringbone/reader.rb +308 -17
- data/lib/herringbone/schema.rb +306 -15
- data/lib/herringbone/thrift.rb +155 -4
- data/lib/herringbone/types.rb +176 -41
- data/lib/herringbone/version.rb +2 -1
- data/lib/herringbone/visualizer.rb +51 -11
- data/lib/herringbone/writer.rb +308 -31
- data/lib/herringbone/xxhash.rb +108 -2
- data/lib/herringbone.rb +28 -0
- metadata +3 -2
|
@@ -5,17 +5,26 @@ module Herringbone
|
|
|
5
5
|
# Pure-Ruby LZ4: raw block format (Parquet LZ4_RAW), Hadoop-framed blocks
|
|
6
6
|
# (Parquet's deprecated LZ4) and a decoder for the LZ4 frame format.
|
|
7
7
|
module LZ4
|
|
8
|
+
# Raised for corrupt, truncated or unsupported LZ4 input.
|
|
8
9
|
class Error < StandardError; end
|
|
9
10
|
|
|
11
|
+
# Shortest match a sequence can encode; the token's match length is stored minus this.
|
|
10
12
|
MIN_MATCH = 4
|
|
11
13
|
LAST_LITERALS = 5 # the last 5 bytes of a block are always literals
|
|
12
14
|
MFLIMIT = 12 # the last match must start at least 12 bytes before the end
|
|
15
|
+
# Largest backward distance a 2-byte match offset can express.
|
|
13
16
|
MAX_OFFSET = 65_535
|
|
17
|
+
# Size of the compressor's hash table, in bits (16384 entries).
|
|
14
18
|
HASH_LOG = 14
|
|
19
|
+
# Right shift that keeps the top HASH_LOG bits of the 32-bit multiplicative hash.
|
|
15
20
|
HASH_SHIFT = 32 - HASH_LOG
|
|
21
|
+
# Controls how fast the compressor skips ahead through input without matches (as in the reference).
|
|
16
22
|
SKIP_STRENGTH = 6
|
|
23
|
+
# Magic number at the start of an LZ4 frame (little-endian).
|
|
17
24
|
FRAME_MAGIC = 0x184D2204
|
|
25
|
+
# Hadoop block header: big-endian 32-bit uncompressed size and compressed size.
|
|
18
26
|
HADOOP_PREFIX = 8
|
|
27
|
+
# Shorthand for Encoding::BINARY.
|
|
19
28
|
BINARY = Encoding::BINARY
|
|
20
29
|
# String#unpack1 accepts offset: since Ruby 3.1
|
|
21
30
|
UNPACK_OFFSET = begin
|
|
@@ -28,6 +37,10 @@ module Herringbone
|
|
|
28
37
|
module_function
|
|
29
38
|
|
|
30
39
|
# Decompress a raw LZ4 block that must expand to exactly uncompressed_size bytes.
|
|
40
|
+
# @param input [String] raw LZ4 block
|
|
41
|
+
# @param uncompressed_size [Integer] exact decompressed size, from the page header
|
|
42
|
+
# @return [String] decompressed bytes in ASCII-8BIT
|
|
43
|
+
# @raise [Error] if the block is corrupt or does not decode to +uncompressed_size+ bytes
|
|
31
44
|
def decompress_block(input, uncompressed_size)
|
|
32
45
|
src = binary(input)
|
|
33
46
|
out = String.new(capacity: uncompressed_size, encoding: BINARY)
|
|
@@ -40,6 +53,10 @@ module Herringbone
|
|
|
40
53
|
|
|
41
54
|
# Parquet LZ4 (codec 5). Tries Hadoop framing, then the LZ4 frame format,
|
|
42
55
|
# then a bare raw block (Arrow falls back hadoop -> raw; some writers emitted frames).
|
|
56
|
+
# @param input [String] compressed page data
|
|
57
|
+
# @param uncompressed_size [Integer] exact decompressed size, from the page header
|
|
58
|
+
# @return [String] decompressed bytes in ASCII-8BIT
|
|
59
|
+
# @raise [Error] if none of the three layouts decodes
|
|
43
60
|
def decompress_hadoop(input, uncompressed_size)
|
|
44
61
|
src = binary(input)
|
|
45
62
|
result = try_hadoop(src, uncompressed_size)
|
|
@@ -58,6 +75,10 @@ module Herringbone
|
|
|
58
75
|
|
|
59
76
|
# Decode LZ4 frame format data (one or more frames, skippable frames ignored).
|
|
60
77
|
# Checksums are skipped, not verified.
|
|
78
|
+
# @param input [String] one or more concatenated LZ4 frames
|
|
79
|
+
# @param max_size [Integer] upper bound on the decompressed size
|
|
80
|
+
# @return [String] decompressed bytes in ASCII-8BIT, possibly shorter than +max_size+
|
|
81
|
+
# @raise [Error] on bad magic, truncation, unsupported frame features or output over +max_size+
|
|
61
82
|
def decompress_frame(input, max_size)
|
|
62
83
|
src = binary(input)
|
|
63
84
|
n = src.bytesize
|
|
@@ -79,6 +100,8 @@ module Herringbone
|
|
|
79
100
|
end
|
|
80
101
|
|
|
81
102
|
# Compress into a single raw LZ4 block.
|
|
103
|
+
# @param input [String] bytes to compress
|
|
104
|
+
# @return [String] raw LZ4 block in ASCII-8BIT
|
|
82
105
|
def compress_block(input)
|
|
83
106
|
src = binary(input)
|
|
84
107
|
n = src.bytesize
|
|
@@ -121,7 +144,7 @@ module Herringbone
|
|
|
121
144
|
# Emit the sequence: token, literal run, offset, extended match length
|
|
122
145
|
lit_len = ip - anchor
|
|
123
146
|
ml = len - MIN_MATCH
|
|
124
|
-
out << (((lit_len < 15 ? lit_len : 15) << 4) | (ml < 15 ? ml : 15))
|
|
147
|
+
out << ((((lit_len < 15) ? lit_len : 15) << 4) | ((ml < 15) ? ml : 15))
|
|
125
148
|
write_length(out, lit_len - 15) if lit_len >= 15
|
|
126
149
|
out << src.byteslice(anchor, lit_len) if lit_len > 0
|
|
127
150
|
offset = ip - ref
|
|
@@ -144,6 +167,8 @@ module Herringbone
|
|
|
144
167
|
end
|
|
145
168
|
|
|
146
169
|
# Single Hadoop-framed block: [BE uncompressed size][BE compressed size][raw block]
|
|
170
|
+
# @param input [String] bytes to compress
|
|
171
|
+
# @return [String] Hadoop-framed LZ4 data in ASCII-8BIT
|
|
147
172
|
def compress_hadoop(input)
|
|
148
173
|
src = binary(input)
|
|
149
174
|
block = compress_block(src)
|
|
@@ -152,18 +177,31 @@ module Herringbone
|
|
|
152
177
|
|
|
153
178
|
# -- internals --
|
|
154
179
|
|
|
180
|
+
# @param str [String] input in any encoding
|
|
181
|
+
# @return [String] +str+ itself if already binary, otherwise a binary copy
|
|
155
182
|
def binary(str)
|
|
156
|
-
str.encoding == BINARY ? str : str.b
|
|
183
|
+
(str.encoding == BINARY) ? str : str.b
|
|
157
184
|
end
|
|
158
185
|
|
|
186
|
+
# @param src [String] binary input
|
|
187
|
+
# @param i [Integer] byte offset
|
|
188
|
+
# @return [Integer] unsigned little-endian 32-bit value at +i+
|
|
159
189
|
def le32(src, i)
|
|
160
190
|
src.byteslice(i, 4).unpack1("V")
|
|
161
191
|
end
|
|
162
192
|
|
|
193
|
+
# Same as le32 but via getbyte, for Rubies without unpack1(offset:).
|
|
194
|
+
# @param src [String] binary input
|
|
195
|
+
# @param i [Integer] byte offset
|
|
196
|
+
# @return [Integer] unsigned little-endian 32-bit value at +i+
|
|
163
197
|
def u32(src, i)
|
|
164
198
|
src.getbyte(i) | (src.getbyte(i + 1) << 8) | (src.getbyte(i + 2) << 16) | (src.getbyte(i + 3) << 24)
|
|
165
199
|
end
|
|
166
200
|
|
|
201
|
+
# Appends the extension of a literal or match length: 255-bytes followed by the remainder.
|
|
202
|
+
# @param out [String] binary output buffer, appended to
|
|
203
|
+
# @param len [Integer] length minus the 15 already stored in the token
|
|
204
|
+
# @return [String] +out+
|
|
167
205
|
def write_length(out, len)
|
|
168
206
|
if len >= 255
|
|
169
207
|
out << ("\xFF".b * (len / 255))
|
|
@@ -172,14 +210,27 @@ module Herringbone
|
|
|
172
210
|
out << len
|
|
173
211
|
end
|
|
174
212
|
|
|
213
|
+
# Appends the final, literals-only sequence that terminates every block.
|
|
214
|
+
# @param out [String] binary output buffer, appended to
|
|
215
|
+
# @param src [String] binary input
|
|
216
|
+
# @param anchor [Integer] offset of the first pending literal in +src+
|
|
217
|
+
# @param lit_len [Integer] number of trailing literal bytes (may be zero)
|
|
218
|
+
# @return [String, nil] +out+, or nil when there are no literals
|
|
175
219
|
def emit_last_literals(out, src, anchor, lit_len)
|
|
176
|
-
out << ((lit_len < 15 ? lit_len : 15) << 4)
|
|
220
|
+
out << (((lit_len < 15) ? lit_len : 15) << 4)
|
|
177
221
|
write_length(out, lit_len - 15) if lit_len >= 15
|
|
178
222
|
out << src.byteslice(anchor, lit_len) if lit_len > 0
|
|
179
223
|
end
|
|
180
224
|
|
|
181
225
|
# Decode one raw block from src[ip...iend], appending to out (which may already
|
|
182
226
|
# hold earlier data that matches can reference). out may not grow beyond limit.
|
|
227
|
+
# @param src [String] binary input
|
|
228
|
+
# @param ip [Integer] offset of the block in +src+
|
|
229
|
+
# @param iend [Integer] offset just past the block
|
|
230
|
+
# @param out [String] binary output buffer, appended to
|
|
231
|
+
# @param limit [Integer] maximum total byte size of +out+
|
|
232
|
+
# @return [Integer] input offset where decoding stopped
|
|
233
|
+
# @raise [Error] if the block is truncated, has an invalid offset or overflows +limit+
|
|
183
234
|
def decode_block(src, ip, iend, out, limit)
|
|
184
235
|
while ip < iend
|
|
185
236
|
token = src.getbyte(ip)
|
|
@@ -237,6 +288,9 @@ module Herringbone
|
|
|
237
288
|
end
|
|
238
289
|
|
|
239
290
|
# Arrow-compatible Hadoop frame parsing; returns nil if the data does not fit the framing.
|
|
291
|
+
# @param src [String] binary input
|
|
292
|
+
# @param uncompressed_size [Integer] exact decompressed size expected over all blocks
|
|
293
|
+
# @return [String, nil] decompressed bytes, or nil if +src+ is not valid Hadoop-framed LZ4
|
|
240
294
|
def try_hadoop(src, uncompressed_size)
|
|
241
295
|
n = src.bytesize
|
|
242
296
|
return nil if n < HADOOP_PREFIX
|
|
@@ -262,6 +316,15 @@ module Herringbone
|
|
|
262
316
|
out
|
|
263
317
|
end
|
|
264
318
|
|
|
319
|
+
# Decodes the descriptor and data blocks of one LZ4 frame (after its magic number),
|
|
320
|
+
# skipping content size and checksums.
|
|
321
|
+
# @param src [String] binary input
|
|
322
|
+
# @param ip [Integer] offset of the frame descriptor (just past the magic)
|
|
323
|
+
# @param n [Integer] byte size of +src+
|
|
324
|
+
# @param out [String] binary output buffer, appended to
|
|
325
|
+
# @param limit [Integer] maximum total byte size of +out+
|
|
326
|
+
# @return [Integer] offset just past the frame
|
|
327
|
+
# @raise [Error] on truncation, an unsupported version or dictionary, or output over +limit+
|
|
265
328
|
def decode_frame(src, ip, n, out, limit)
|
|
266
329
|
raise Error, "Truncated LZ4 frame descriptor" if ip + 3 > n
|
|
267
330
|
flg = src.getbyte(ip)
|
|
@@ -1,33 +1,103 @@
|
|
|
1
1
|
# frozen_string_literal: true
|
|
2
2
|
|
|
3
3
|
module Herringbone
|
|
4
|
+
# Pure-Ruby compression codecs for Parquet pages: Snappy, and LZ4 (raw blocks and the Hadoop framing).
|
|
5
|
+
# Compression dispatches to these; GZIP uses Zlib and ZSTD/BROTLI use external gems.
|
|
4
6
|
module Codecs
|
|
5
7
|
# Pure-Ruby implementation of the raw Snappy block format (as used by Parquet),
|
|
6
8
|
# see https://github.com/google/snappy/blob/main/format_description.txt
|
|
7
9
|
#
|
|
10
|
+
# When the `snappy` gem (a binding to Google's libsnappy) can be loaded, it is used instead:
|
|
11
|
+
# 12x faster decompression and 27x faster compression. It is optional and only a speedup: if
|
|
12
|
+
# it is missing, the pure-Ruby code below is used silently. Both produce raw Snappy blocks the
|
|
13
|
+
# other reads.
|
|
14
|
+
#
|
|
8
15
|
# 32-bit loads are done with four getbyte calls rather than unpack1(offset:) to stay
|
|
9
16
|
# compatible with Ruby 3.0 - the speed difference on MRI is marginal.
|
|
10
17
|
module Snappy
|
|
18
|
+
# Raised for corrupt or truncated Snappy input, and for input too large to compress.
|
|
11
19
|
class Error < StandardError; end
|
|
12
20
|
|
|
21
|
+
# The compressor works on independent 64 KiB fragments, as the reference implementation does.
|
|
13
22
|
BLOCK_SIZE = 1 << 16
|
|
23
|
+
# Size of the match-finder hash table, in bits (16384 entries).
|
|
14
24
|
HASH_BITS = 14
|
|
25
|
+
# Right shift that keeps the top HASH_BITS bits of the 32-bit multiplicative hash.
|
|
15
26
|
HASH_SHIFT = 32 - HASH_BITS
|
|
27
|
+
# Multiplier of the reference implementation's 4-byte hash.
|
|
16
28
|
HASH_MUL = 0x1e35a7bd
|
|
17
29
|
INPUT_MARGIN = 15 # bytes at the block end never searched for matches, as in the reference
|
|
30
|
+
# The length preamble is a 32-bit varint, so no block may decompress to more than this.
|
|
18
31
|
MAX_UNCOMPRESSED = (1 << 32) - 1
|
|
19
32
|
|
|
33
|
+
# Name of the optional native gem, passed to +require+.
|
|
34
|
+
NATIVE_GEM = "snappy"
|
|
35
|
+
|
|
20
36
|
module_function
|
|
21
37
|
|
|
22
38
|
# @param input [String] raw snappy block
|
|
23
39
|
# @return [String] decompressed bytes in ASCII-8BIT
|
|
40
|
+
# @raise [Error] if the block is corrupt or truncated
|
|
24
41
|
def decompress(input)
|
|
25
|
-
src = input.encoding == Encoding::BINARY ? input : input.b
|
|
42
|
+
src = (input.encoding == Encoding::BINARY) ? input : input.b
|
|
43
|
+
if (lib = native)
|
|
44
|
+
begin
|
|
45
|
+
return lib.inflate(src)
|
|
46
|
+
rescue lib::Error => e
|
|
47
|
+
raise Error, "corrupt snappy data (#{e.message})"
|
|
48
|
+
end
|
|
49
|
+
end
|
|
26
50
|
return decompress_io_buffer(src) if IOBufferSupport::AVAILABLE
|
|
27
51
|
decompress_string(src)
|
|
28
52
|
end
|
|
29
53
|
|
|
54
|
+
# The backend in use: :native (the snappy gem) or :ruby
|
|
55
|
+
# @return [Symbol] +:native+ or +:ruby+
|
|
56
|
+
def backend
|
|
57
|
+
native ? :native : :ruby
|
|
58
|
+
end
|
|
59
|
+
|
|
60
|
+
# For tests and benchmarks: :ruby forces pure Ruby, :native requires the snappy gem
|
|
61
|
+
# (UnsupportedError if it cannot be loaded), nil goes back to the default
|
|
62
|
+
# @param name [Symbol, nil] +:ruby+, +:native+ or nil
|
|
63
|
+
# @return [void]
|
|
64
|
+
# @raise [UnsupportedError] if +:native+ is requested and the gem cannot be loaded
|
|
65
|
+
# @raise [ArgumentError] for any other backend name
|
|
66
|
+
def backend=(name)
|
|
67
|
+
@native = case name
|
|
68
|
+
when :ruby then false
|
|
69
|
+
when :native
|
|
70
|
+
native_library || raise(UnsupportedError, "The \"#{NATIVE_GEM}\" gem could not be loaded")
|
|
71
|
+
when nil then nil
|
|
72
|
+
else raise ArgumentError, "Unknown Snappy backend #{name.inspect} (expected :ruby, :native or nil)"
|
|
73
|
+
end
|
|
74
|
+
end
|
|
75
|
+
|
|
76
|
+
# The native library to use, resolved (and memoized) on first call unless forced by backend=.
|
|
77
|
+
# @return [Module, nil] the +::Snappy+ module, or nil when the pure-Ruby code should run
|
|
78
|
+
def native
|
|
79
|
+
@native = native_library || false if @native.nil?
|
|
80
|
+
@native || nil
|
|
81
|
+
end
|
|
82
|
+
|
|
83
|
+
# Tries to load the snappy gem once and memoizes the result.
|
|
84
|
+
# @return [Module, false] the +::Snappy+ module, or false if it is missing or lacks inflate/deflate
|
|
85
|
+
def native_library
|
|
86
|
+
if @native_lib.nil?
|
|
87
|
+
@native_lib = begin
|
|
88
|
+
require NATIVE_GEM
|
|
89
|
+
(::Snappy.respond_to?(:inflate) && ::Snappy.respond_to?(:deflate)) ? ::Snappy : false
|
|
90
|
+
rescue LoadError
|
|
91
|
+
false
|
|
92
|
+
end
|
|
93
|
+
end
|
|
94
|
+
@native_lib
|
|
95
|
+
end
|
|
96
|
+
|
|
30
97
|
# Decompresses into a preallocated IO::Buffer: copies do not allocate intermediate Strings.
|
|
98
|
+
# @param src [String] raw snappy block in ASCII-8BIT
|
|
99
|
+
# @return [String] decompressed bytes in ASCII-8BIT
|
|
100
|
+
# @raise [Error] if the block is corrupt or truncated
|
|
31
101
|
def decompress_io_buffer(src)
|
|
32
102
|
n = src.bytesize
|
|
33
103
|
expected, pos = read_varint(src, n)
|
|
@@ -98,6 +168,10 @@ module Herringbone
|
|
|
98
168
|
inbuf&.free
|
|
99
169
|
end
|
|
100
170
|
|
|
171
|
+
# Decompresses by appending to a String; the fallback when IO::Buffer is not available.
|
|
172
|
+
# @param src [String] raw snappy block in ASCII-8BIT
|
|
173
|
+
# @return [String] decompressed bytes in ASCII-8BIT
|
|
174
|
+
# @raise [Error] if the block is corrupt or truncated
|
|
101
175
|
def decompress_string(src)
|
|
102
176
|
n = src.bytesize
|
|
103
177
|
expected, pos = read_varint(src, n)
|
|
@@ -162,10 +236,15 @@ module Herringbone
|
|
|
162
236
|
|
|
163
237
|
# @param input [String] bytes to compress
|
|
164
238
|
# @return [String] raw snappy block in ASCII-8BIT
|
|
239
|
+
# @raise [Error] if the input is larger than MAX_UNCOMPRESSED
|
|
165
240
|
def compress(input)
|
|
166
|
-
src = input.encoding == Encoding::BINARY ? input : input.b
|
|
241
|
+
src = (input.encoding == Encoding::BINARY) ? input : input.b
|
|
167
242
|
n = src.bytesize
|
|
243
|
+
# Checked for libsnappy too: it would truncate the 32-bit length preamble without complaint
|
|
168
244
|
raise Error, "input too large for snappy" if n > MAX_UNCOMPRESSED
|
|
245
|
+
if (lib = native)
|
|
246
|
+
return lib.deflate(src)
|
|
247
|
+
end
|
|
169
248
|
|
|
170
249
|
out = String.new(capacity: 32 + n + n / 6, encoding: Encoding::BINARY)
|
|
171
250
|
write_varint(out, n)
|
|
@@ -181,6 +260,11 @@ module Herringbone
|
|
|
181
260
|
out
|
|
182
261
|
end
|
|
183
262
|
|
|
263
|
+
# Reads the uncompressed-length preamble (a little-endian base-128 varint) at the start of +src+.
|
|
264
|
+
# @param src [String] raw snappy block
|
|
265
|
+
# @param n [Integer] byte size of +src+
|
|
266
|
+
# @return [Array(Integer, Integer)] the declared uncompressed length and the offset just past the varint
|
|
267
|
+
# @raise [Error] if the varint is truncated, longer than 5 bytes or exceeds MAX_UNCOMPRESSED
|
|
184
268
|
def read_varint(src, n)
|
|
185
269
|
value = 0
|
|
186
270
|
shift = 0
|
|
@@ -198,6 +282,10 @@ module Herringbone
|
|
|
198
282
|
[value, pos]
|
|
199
283
|
end
|
|
200
284
|
|
|
285
|
+
# Appends +value+ as a little-endian base-128 varint (the length preamble).
|
|
286
|
+
# @param out [String] binary output buffer, appended to
|
|
287
|
+
# @param value [Integer] non-negative length to encode
|
|
288
|
+
# @return [String] +out+
|
|
201
289
|
def write_varint(out, value)
|
|
202
290
|
while value >= 0x80
|
|
203
291
|
out << ((value & 0x7f) | 0x80)
|
|
@@ -210,6 +298,12 @@ module Herringbone
|
|
|
210
298
|
# positions relative to `base`; matches never cross the block boundary.
|
|
211
299
|
# The 4-byte little-endian word at every position of the block is unpacked up front
|
|
212
300
|
# (in C, via String#unpack) so hashing and match checks are single Array lookups.
|
|
301
|
+
# @param src [String] whole input in ASCII-8BIT
|
|
302
|
+
# @param base [Integer] offset of the fragment in +src+
|
|
303
|
+
# @param len [Integer] fragment length, at most BLOCK_SIZE
|
|
304
|
+
# @param out [String] binary output buffer, appended to
|
|
305
|
+
# @param table [Array<Integer>] zeroed hash table of 1 << HASH_BITS fragment-relative positions
|
|
306
|
+
# @return [void]
|
|
213
307
|
def compress_block(src, base, len, out, table)
|
|
214
308
|
ip_end = base + len
|
|
215
309
|
next_emit = base
|
|
@@ -280,12 +374,24 @@ module Herringbone
|
|
|
280
374
|
|
|
281
375
|
# words[k][j] is the 4-byte little-endian value at base + 4 * j + k, so the word at
|
|
282
376
|
# relative position i is words[i & 3][i >> 2]
|
|
377
|
+
# @param src [String] whole input in ASCII-8BIT
|
|
378
|
+
# @param base [Integer] offset of the fragment in +src+
|
|
379
|
+
# @param len [Integer] fragment length
|
|
380
|
+
# @return [Array<Array<Integer>>] four arrays of 32-bit words, one per byte phase
|
|
283
381
|
def block_words(src, base, len)
|
|
284
382
|
Array.new(4) { |k| src.byteslice(base + k, len - k).unpack("V*") }
|
|
285
383
|
end
|
|
286
384
|
|
|
287
385
|
# Like match_length, but compares 4 bytes at a time using the unpacked block words,
|
|
288
386
|
# which avoids allocating substrings for the (common) short matches.
|
|
387
|
+
# Falls back to match_length once a match reaches 64 bytes.
|
|
388
|
+
# @param words [Array<Array<Integer>>] fragment words from block_words
|
|
389
|
+
# @param src [String] whole input in ASCII-8BIT
|
|
390
|
+
# @param base [Integer] offset of the fragment in +src+
|
|
391
|
+
# @param s1 [Integer] absolute offset of the earlier occurrence
|
|
392
|
+
# @param s2 [Integer] absolute offset of the current position (s1 < s2)
|
|
393
|
+
# @param limit [Integer] absolute offset not to read past (the fragment end)
|
|
394
|
+
# @return [Integer] number of equal bytes starting at s1 and s2
|
|
289
395
|
def match_length_words(words, src, base, s1, s2, limit)
|
|
290
396
|
start = s2
|
|
291
397
|
r1 = s1 - base
|
|
@@ -307,6 +413,11 @@ module Herringbone
|
|
|
307
413
|
|
|
308
414
|
# Number of equal bytes at s1 and s2 (s1 < s2), not reading past limit. Gallops
|
|
309
415
|
# with byteslice comparisons (memcmp) to avoid per-byte loops on long matches.
|
|
416
|
+
# @param src [String] whole input in ASCII-8BIT
|
|
417
|
+
# @param s1 [Integer] absolute offset of the earlier occurrence
|
|
418
|
+
# @param s2 [Integer] absolute offset of the current position
|
|
419
|
+
# @param limit [Integer] absolute offset not to read past
|
|
420
|
+
# @return [Integer] number of equal bytes
|
|
310
421
|
def match_length(src, s1, s2, limit)
|
|
311
422
|
start = s2
|
|
312
423
|
return 0 if s2 >= limit || src.getbyte(s1) != src.getbyte(s2)
|
|
@@ -330,6 +441,13 @@ module Herringbone
|
|
|
330
441
|
s2 - start
|
|
331
442
|
end
|
|
332
443
|
|
|
444
|
+
# Appends a literal element: the tag (with a 1-4 byte length extension for long literals)
|
|
445
|
+
# followed by the bytes themselves. Does nothing for an empty literal.
|
|
446
|
+
# @param out [String] binary output buffer, appended to
|
|
447
|
+
# @param src [String] whole input in ASCII-8BIT
|
|
448
|
+
# @param pos [Integer] offset of the literal bytes in +src+
|
|
449
|
+
# @param len [Integer] number of literal bytes
|
|
450
|
+
# @return [String, nil] +out+, or nil when +len+ is zero
|
|
333
451
|
def emit_literal(out, src, pos, len)
|
|
334
452
|
return if len == 0
|
|
335
453
|
n = len - 1
|
|
@@ -348,6 +466,11 @@ module Herringbone
|
|
|
348
466
|
end
|
|
349
467
|
|
|
350
468
|
# Offsets are always < 64KB (matches stay within a block), so 4-byte offsets are never needed.
|
|
469
|
+
# Long matches are split into copies of at most 64 bytes, never leaving a remainder under 4.
|
|
470
|
+
# @param out [String] binary output buffer, appended to
|
|
471
|
+
# @param offset [Integer] backward distance to the match source, 1...65536
|
|
472
|
+
# @param len [Integer] match length, at least 4
|
|
473
|
+
# @return [String] +out+
|
|
351
474
|
def emit_copy(out, offset, len)
|
|
352
475
|
while len >= 68
|
|
353
476
|
emit_copy_upto64(out, offset, 64)
|
|
@@ -360,6 +483,12 @@ module Herringbone
|
|
|
360
483
|
emit_copy_upto64(out, offset, len)
|
|
361
484
|
end
|
|
362
485
|
|
|
486
|
+
# Appends one copy element: the 2-byte form (1-byte offset) when len is 4..11 and offset < 2048,
|
|
487
|
+
# otherwise the 3-byte form with a 2-byte offset.
|
|
488
|
+
# @param out [String] binary output buffer, appended to
|
|
489
|
+
# @param offset [Integer] backward distance to the match source, 1...65536
|
|
490
|
+
# @param len [Integer] copy length, 4..64
|
|
491
|
+
# @return [String] +out+
|
|
363
492
|
def emit_copy_upto64(out, offset, len)
|
|
364
493
|
if len < 12 && offset < 2048
|
|
365
494
|
out << (1 | ((len - 4) << 2) | ((offset >> 8) << 5)) << (offset & 0xff)
|
|
@@ -368,7 +497,7 @@ module Herringbone
|
|
|
368
497
|
end
|
|
369
498
|
end
|
|
370
499
|
|
|
371
|
-
private_class_method :decompress_io_buffer, :decompress_string, :read_varint, :write_varint, :compress_block, :block_words, :match_length_words,
|
|
500
|
+
private_class_method :native, :native_library, :decompress_io_buffer, :decompress_string, :read_varint, :write_varint, :compress_block, :block_words, :match_length_words,
|
|
372
501
|
:match_length, :emit_literal, :emit_copy, :emit_copy_upto64
|
|
373
502
|
end
|
|
374
503
|
end
|
|
@@ -6,8 +6,12 @@ require "stringio"
|
|
|
6
6
|
module Herringbone
|
|
7
7
|
# Raised when a file uses (or a writer asks for) a codec whose library is not installed
|
|
8
8
|
class MissingCodecError < UnsupportedError
|
|
9
|
+
# @return [String] codec name (+"ZSTD"+) and name of the gem providing it (+"zstd-ruby"+)
|
|
9
10
|
attr_reader :codec, :gem_name
|
|
10
11
|
|
|
12
|
+
# @param codec [String] codec name, as shown in the message
|
|
13
|
+
# @param gem_name [String] gem to add to the Gemfile
|
|
14
|
+
# @param load_error [LoadError] the error from requiring the gem, whose message is included
|
|
11
15
|
def initialize(codec, gem_name, load_error)
|
|
12
16
|
@codec = codec
|
|
13
17
|
@gem_name = gem_name
|
|
@@ -19,6 +23,7 @@ module Herringbone
|
|
|
19
23
|
# Dispatches page (de)compression by Parquet codec id. Snappy and LZ4 are pure Ruby and GZIP
|
|
20
24
|
# uses zlib, so those always work. ZSTD and Brotli come from optional gems (zstd-ruby, brotli),
|
|
21
25
|
# which are required on first use; if they are missing, MissingCodecError says what to add.
|
|
26
|
+
# (Snappy also uses the optional snappy gem when it is installed, see Codecs::Snappy.)
|
|
22
27
|
module Compression
|
|
23
28
|
module_function
|
|
24
29
|
|
|
@@ -32,6 +37,11 @@ module Herringbone
|
|
|
32
37
|
@library_mutex = Mutex.new
|
|
33
38
|
|
|
34
39
|
# The library module for a codec backed by an optional gem, requiring it on first use
|
|
40
|
+
#
|
|
41
|
+
# @param codec [Integer] a codec id that is a key of LIBRARIES
|
|
42
|
+
# @return [Module] the gem's module (+Zstd+ or +Brotli+)
|
|
43
|
+
# @raise [MissingCodecError] when the gem cannot be loaded
|
|
44
|
+
# @raise [KeyError] for a codec id not in LIBRARIES
|
|
35
45
|
def library(codec)
|
|
36
46
|
@libraries.fetch(codec) do
|
|
37
47
|
@library_mutex.synchronize do
|
|
@@ -48,11 +58,22 @@ module Herringbone
|
|
|
48
58
|
end
|
|
49
59
|
end
|
|
50
60
|
|
|
61
|
+
# Requires a codec gem; a separate method so tests can stub it to simulate a missing gem
|
|
62
|
+
#
|
|
63
|
+
# @param path [String] require path of the gem
|
|
64
|
+
# @return [Boolean] whether the library was newly loaded
|
|
65
|
+
# @raise [LoadError] when the gem is not installed
|
|
51
66
|
def require_library(path)
|
|
52
67
|
require path
|
|
53
68
|
end
|
|
54
69
|
|
|
55
70
|
# Raises MissingCodecError (or UnsupportedError) unless +codec+ can be used
|
|
71
|
+
#
|
|
72
|
+
# @param codec [Integer, Symbol, String] codec id or name (see NAMES)
|
|
73
|
+
# @return [Module, nil] the gem's module for a gem-backed codec, nil for a built-in one
|
|
74
|
+
# @raise [MissingCodecError] when the codec's gem cannot be loaded
|
|
75
|
+
# @raise [UnsupportedError] for a codec herringbone does not implement (LZO)
|
|
76
|
+
# @raise [ArgumentError] for an unknown codec name
|
|
56
77
|
def ensure_available!(codec)
|
|
57
78
|
codec = codec_id(codec)
|
|
58
79
|
return library(codec) if LIBRARIES.key?(codec)
|
|
@@ -60,6 +81,7 @@ module Herringbone
|
|
|
60
81
|
raise UnsupportedError, "#{Format::Codec::NAMES[codec] || codec} compression is not supported"
|
|
61
82
|
end
|
|
62
83
|
|
|
84
|
+
# Codec ids that work without optional gems
|
|
63
85
|
SUPPORTED = [
|
|
64
86
|
Format::Codec::UNCOMPRESSED, Format::Codec::SNAPPY, Format::Codec::GZIP,
|
|
65
87
|
Format::Codec::LZ4_RAW, Format::Codec::LZ4
|
|
@@ -71,8 +93,14 @@ module Herringbone
|
|
|
71
93
|
Format::Codec::LZ4_RAW => :lz4, Format::Codec::LZ4 => :lz4_hadoop, Format::Codec::ZSTD => :zstd,
|
|
72
94
|
Format::Codec::BROTLI => :brotli, Format::Codec::LZO => :lzo
|
|
73
95
|
}.freeze
|
|
96
|
+
# Codec name => codec id, the inverse of NAMES
|
|
74
97
|
CODECS_BY_NAME = NAMES.invert.freeze
|
|
75
98
|
|
|
99
|
+
# Codec id for a codec name; Integers are taken to be ids already and returned unchecked
|
|
100
|
+
#
|
|
101
|
+
# @param name [Integer, Symbol, String] codec id, or a name from NAMES (case-insensitive)
|
|
102
|
+
# @return [Integer] the Format::Codec id
|
|
103
|
+
# @raise [ArgumentError] for an unknown name
|
|
76
104
|
def codec_id(name)
|
|
77
105
|
return name if name.is_a?(Integer)
|
|
78
106
|
CODECS_BY_NAME.fetch(name.to_s.downcase.to_sym) do
|
|
@@ -80,6 +108,15 @@ module Herringbone
|
|
|
80
108
|
end
|
|
81
109
|
end
|
|
82
110
|
|
|
111
|
+
# Decompresses a page body, checking the result against the size the page header declares
|
|
112
|
+
#
|
|
113
|
+
# @param codec [Integer] Format::Codec id from the column chunk metadata
|
|
114
|
+
# @param data [String] compressed bytes
|
|
115
|
+
# @param uncompressed_size [Integer] expected decompressed size in bytes
|
|
116
|
+
# @return [String] decompressed bytes (binary)
|
|
117
|
+
# @raise [UnsupportedError] for a codec herringbone does not implement
|
|
118
|
+
# @raise [MissingCodecError] when the codec's gem cannot be loaded
|
|
119
|
+
# @raise [FormatError] when the decompressed size does not match +uncompressed_size+
|
|
83
120
|
def decompress(codec, data, uncompressed_size)
|
|
84
121
|
return "".b if uncompressed_size.zero? && data.empty?
|
|
85
122
|
out = case codec
|
|
@@ -100,6 +137,13 @@ module Herringbone
|
|
|
100
137
|
out
|
|
101
138
|
end
|
|
102
139
|
|
|
140
|
+
# Compresses a page body
|
|
141
|
+
#
|
|
142
|
+
# @param codec [Integer] Format::Codec id
|
|
143
|
+
# @param data [String] bytes to compress
|
|
144
|
+
# @return [String] compressed bytes (binary)
|
|
145
|
+
# @raise [UnsupportedError] for a codec herringbone does not implement
|
|
146
|
+
# @raise [MissingCodecError] when the codec's gem cannot be loaded
|
|
103
147
|
def compress(codec, data)
|
|
104
148
|
case codec
|
|
105
149
|
when Format::Codec::UNCOMPRESSED then data
|
|
@@ -115,6 +159,10 @@ module Herringbone
|
|
|
115
159
|
end
|
|
116
160
|
|
|
117
161
|
# Handles files whose gzip data consists of several concatenated members
|
|
162
|
+
#
|
|
163
|
+
# @param data [String] gzip data, one or more members
|
|
164
|
+
# @return [String] the members' decompressed bytes concatenated (binary)
|
|
165
|
+
# @raise [Zlib::GzipFile::Error] on corrupt gzip data
|
|
118
166
|
def gunzip(data)
|
|
119
167
|
io = StringIO.new(data)
|
|
120
168
|
out = String.new(encoding: Encoding::BINARY)
|