herringbone 0.2.0 → 0.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -5,17 +5,26 @@ module Herringbone
5
5
  # Pure-Ruby LZ4: raw block format (Parquet LZ4_RAW), Hadoop-framed blocks
6
6
  # (Parquet's deprecated LZ4) and a decoder for the LZ4 frame format.
7
7
  module LZ4
8
+ # Raised for corrupt, truncated or unsupported LZ4 input.
8
9
  class Error < StandardError; end
9
10
 
11
+ # Shortest match a sequence can encode; the token's match length is stored minus this.
10
12
  MIN_MATCH = 4
11
13
  LAST_LITERALS = 5 # the last 5 bytes of a block are always literals
12
14
  MFLIMIT = 12 # the last match must start at least 12 bytes before the end
15
+ # Largest backward distance a 2-byte match offset can express.
13
16
  MAX_OFFSET = 65_535
17
+ # Size of the compressor's hash table, in bits (16384 entries).
14
18
  HASH_LOG = 14
19
+ # Right shift that keeps the top HASH_LOG bits of the 32-bit multiplicative hash.
15
20
  HASH_SHIFT = 32 - HASH_LOG
21
+ # Controls how fast the compressor skips ahead through input without matches (as in the reference).
16
22
  SKIP_STRENGTH = 6
23
+ # Magic number at the start of an LZ4 frame (little-endian).
17
24
  FRAME_MAGIC = 0x184D2204
25
+ # Hadoop block header: big-endian 32-bit uncompressed size and compressed size.
18
26
  HADOOP_PREFIX = 8
27
+ # Shorthand for Encoding::BINARY.
19
28
  BINARY = Encoding::BINARY
20
29
  # String#unpack1 accepts offset: since Ruby 3.1
21
30
  UNPACK_OFFSET = begin
@@ -28,6 +37,10 @@ module Herringbone
28
37
  module_function
29
38
 
30
39
  # Decompress a raw LZ4 block that must expand to exactly uncompressed_size bytes.
40
+ # @param input [String] raw LZ4 block
41
+ # @param uncompressed_size [Integer] exact decompressed size, from the page header
42
+ # @return [String] decompressed bytes in ASCII-8BIT
43
+ # @raise [Error] if the block is corrupt or does not decode to +uncompressed_size+ bytes
31
44
  def decompress_block(input, uncompressed_size)
32
45
  src = binary(input)
33
46
  out = String.new(capacity: uncompressed_size, encoding: BINARY)
@@ -40,6 +53,10 @@ module Herringbone
40
53
 
41
54
  # Parquet LZ4 (codec 5). Tries Hadoop framing, then the LZ4 frame format,
42
55
  # then a bare raw block (Arrow falls back hadoop -> raw; some writers emitted frames).
56
+ # @param input [String] compressed page data
57
+ # @param uncompressed_size [Integer] exact decompressed size, from the page header
58
+ # @return [String] decompressed bytes in ASCII-8BIT
59
+ # @raise [Error] if none of the three layouts decodes
43
60
  def decompress_hadoop(input, uncompressed_size)
44
61
  src = binary(input)
45
62
  result = try_hadoop(src, uncompressed_size)
@@ -58,6 +75,10 @@ module Herringbone
58
75
 
59
76
  # Decode LZ4 frame format data (one or more frames, skippable frames ignored).
60
77
  # Checksums are skipped, not verified.
78
+ # @param input [String] one or more concatenated LZ4 frames
79
+ # @param max_size [Integer] upper bound on the decompressed size
80
+ # @return [String] decompressed bytes in ASCII-8BIT, possibly shorter than +max_size+
81
+ # @raise [Error] on bad magic, truncation, unsupported frame features or output over +max_size+
61
82
  def decompress_frame(input, max_size)
62
83
  src = binary(input)
63
84
  n = src.bytesize
@@ -79,6 +100,8 @@ module Herringbone
79
100
  end
80
101
 
81
102
  # Compress into a single raw LZ4 block.
103
+ # @param input [String] bytes to compress
104
+ # @return [String] raw LZ4 block in ASCII-8BIT
82
105
  def compress_block(input)
83
106
  src = binary(input)
84
107
  n = src.bytesize
@@ -121,7 +144,7 @@ module Herringbone
121
144
  # Emit the sequence: token, literal run, offset, extended match length
122
145
  lit_len = ip - anchor
123
146
  ml = len - MIN_MATCH
124
- out << (((lit_len < 15 ? lit_len : 15) << 4) | (ml < 15 ? ml : 15))
147
+ out << ((((lit_len < 15) ? lit_len : 15) << 4) | ((ml < 15) ? ml : 15))
125
148
  write_length(out, lit_len - 15) if lit_len >= 15
126
149
  out << src.byteslice(anchor, lit_len) if lit_len > 0
127
150
  offset = ip - ref
@@ -144,6 +167,8 @@ module Herringbone
144
167
  end
145
168
 
146
169
  # Single Hadoop-framed block: [BE uncompressed size][BE compressed size][raw block]
170
+ # @param input [String] bytes to compress
171
+ # @return [String] Hadoop-framed LZ4 data in ASCII-8BIT
147
172
  def compress_hadoop(input)
148
173
  src = binary(input)
149
174
  block = compress_block(src)
@@ -152,18 +177,31 @@ module Herringbone
152
177
 
153
178
  # -- internals --
154
179
 
180
+ # @param str [String] input in any encoding
181
+ # @return [String] +str+ itself if already binary, otherwise a binary copy
155
182
  def binary(str)
156
- str.encoding == BINARY ? str : str.b
183
+ (str.encoding == BINARY) ? str : str.b
157
184
  end
158
185
 
186
+ # @param src [String] binary input
187
+ # @param i [Integer] byte offset
188
+ # @return [Integer] unsigned little-endian 32-bit value at +i+
159
189
  def le32(src, i)
160
190
  src.byteslice(i, 4).unpack1("V")
161
191
  end
162
192
 
193
+ # Same as le32 but via getbyte, for Rubies without unpack1(offset:).
194
+ # @param src [String] binary input
195
+ # @param i [Integer] byte offset
196
+ # @return [Integer] unsigned little-endian 32-bit value at +i+
163
197
  def u32(src, i)
164
198
  src.getbyte(i) | (src.getbyte(i + 1) << 8) | (src.getbyte(i + 2) << 16) | (src.getbyte(i + 3) << 24)
165
199
  end
166
200
 
201
+ # Appends the extension of a literal or match length: 255-bytes followed by the remainder.
202
+ # @param out [String] binary output buffer, appended to
203
+ # @param len [Integer] length minus the 15 already stored in the token
204
+ # @return [String] +out+
167
205
  def write_length(out, len)
168
206
  if len >= 255
169
207
  out << ("\xFF".b * (len / 255))
@@ -172,14 +210,27 @@ module Herringbone
172
210
  out << len
173
211
  end
174
212
 
213
+ # Appends the final, literals-only sequence that terminates every block.
214
+ # @param out [String] binary output buffer, appended to
215
+ # @param src [String] binary input
216
+ # @param anchor [Integer] offset of the first pending literal in +src+
217
+ # @param lit_len [Integer] number of trailing literal bytes (may be zero)
218
+ # @return [String, nil] +out+, or nil when there are no literals
175
219
  def emit_last_literals(out, src, anchor, lit_len)
176
- out << ((lit_len < 15 ? lit_len : 15) << 4)
220
+ out << (((lit_len < 15) ? lit_len : 15) << 4)
177
221
  write_length(out, lit_len - 15) if lit_len >= 15
178
222
  out << src.byteslice(anchor, lit_len) if lit_len > 0
179
223
  end
180
224
 
181
225
  # Decode one raw block from src[ip...iend], appending to out (which may already
182
226
  # hold earlier data that matches can reference). out may not grow beyond limit.
227
+ # @param src [String] binary input
228
+ # @param ip [Integer] offset of the block in +src+
229
+ # @param iend [Integer] offset just past the block
230
+ # @param out [String] binary output buffer, appended to
231
+ # @param limit [Integer] maximum total byte size of +out+
232
+ # @return [Integer] input offset where decoding stopped
233
+ # @raise [Error] if the block is truncated, has an invalid offset or overflows +limit+
183
234
  def decode_block(src, ip, iend, out, limit)
184
235
  while ip < iend
185
236
  token = src.getbyte(ip)
@@ -237,6 +288,9 @@ module Herringbone
237
288
  end
238
289
 
239
290
  # Arrow-compatible Hadoop frame parsing; returns nil if the data does not fit the framing.
291
+ # @param src [String] binary input
292
+ # @param uncompressed_size [Integer] exact decompressed size expected over all blocks
293
+ # @return [String, nil] decompressed bytes, or nil if +src+ is not valid Hadoop-framed LZ4
240
294
  def try_hadoop(src, uncompressed_size)
241
295
  n = src.bytesize
242
296
  return nil if n < HADOOP_PREFIX
@@ -262,6 +316,15 @@ module Herringbone
262
316
  out
263
317
  end
264
318
 
319
+ # Decodes the descriptor and data blocks of one LZ4 frame (after its magic number),
320
+ # skipping content size and checksums.
321
+ # @param src [String] binary input
322
+ # @param ip [Integer] offset of the frame descriptor (just past the magic)
323
+ # @param n [Integer] byte size of +src+
324
+ # @param out [String] binary output buffer, appended to
325
+ # @param limit [Integer] maximum total byte size of +out+
326
+ # @return [Integer] offset just past the frame
327
+ # @raise [Error] on truncation, an unsupported version or dictionary, or output over +limit+
265
328
  def decode_frame(src, ip, n, out, limit)
266
329
  raise Error, "Truncated LZ4 frame descriptor" if ip + 3 > n
267
330
  flg = src.getbyte(ip)
@@ -1,33 +1,103 @@
1
1
  # frozen_string_literal: true
2
2
 
3
3
  module Herringbone
4
+ # Pure-Ruby compression codecs for Parquet pages: Snappy, and LZ4 (raw blocks and the Hadoop framing).
5
+ # Compression dispatches to these; GZIP uses Zlib and ZSTD/BROTLI use external gems.
4
6
  module Codecs
5
7
  # Pure-Ruby implementation of the raw Snappy block format (as used by Parquet),
6
8
  # see https://github.com/google/snappy/blob/main/format_description.txt
7
9
  #
10
+ # When the `snappy` gem (a binding to Google's libsnappy) can be loaded, it is used instead:
11
+ # 12x faster decompression and 27x faster compression. It is optional and only a speedup: if
12
+ # it is missing, the pure-Ruby code below is used silently. Both produce raw Snappy blocks the
13
+ # other reads.
14
+ #
8
15
  # 32-bit loads are done with four getbyte calls rather than unpack1(offset:) to stay
9
16
  # compatible with Ruby 3.0 - the speed difference on MRI is marginal.
10
17
  module Snappy
18
+ # Raised for corrupt or truncated Snappy input, and for input too large to compress.
11
19
  class Error < StandardError; end
12
20
 
21
+ # The compressor works on independent 64 KiB fragments, as the reference implementation does.
13
22
  BLOCK_SIZE = 1 << 16
23
+ # Size of the match-finder hash table, in bits (16384 entries).
14
24
  HASH_BITS = 14
25
+ # Right shift that keeps the top HASH_BITS bits of the 32-bit multiplicative hash.
15
26
  HASH_SHIFT = 32 - HASH_BITS
27
+ # Multiplier of the reference implementation's 4-byte hash.
16
28
  HASH_MUL = 0x1e35a7bd
17
29
  INPUT_MARGIN = 15 # bytes at the block end never searched for matches, as in the reference
30
+ # The length preamble is a 32-bit varint, so no block may decompress to more than this.
18
31
  MAX_UNCOMPRESSED = (1 << 32) - 1
19
32
 
33
+ # Name of the optional native gem, passed to +require+.
34
+ NATIVE_GEM = "snappy"
35
+
20
36
  module_function
21
37
 
22
38
  # @param input [String] raw snappy block
23
39
  # @return [String] decompressed bytes in ASCII-8BIT
40
+ # @raise [Error] if the block is corrupt or truncated
24
41
  def decompress(input)
25
- src = input.encoding == Encoding::BINARY ? input : input.b
42
+ src = (input.encoding == Encoding::BINARY) ? input : input.b
43
+ if (lib = native)
44
+ begin
45
+ return lib.inflate(src)
46
+ rescue lib::Error => e
47
+ raise Error, "corrupt snappy data (#{e.message})"
48
+ end
49
+ end
26
50
  return decompress_io_buffer(src) if IOBufferSupport::AVAILABLE
27
51
  decompress_string(src)
28
52
  end
29
53
 
54
+ # The backend in use: :native (the snappy gem) or :ruby
55
+ # @return [Symbol] +:native+ or +:ruby+
56
+ def backend
57
+ native ? :native : :ruby
58
+ end
59
+
60
+ # For tests and benchmarks: :ruby forces pure Ruby, :native requires the snappy gem
61
+ # (UnsupportedError if it cannot be loaded), nil goes back to the default
62
+ # @param name [Symbol, nil] +:ruby+, +:native+ or nil
63
+ # @return [void]
64
+ # @raise [UnsupportedError] if +:native+ is requested and the gem cannot be loaded
65
+ # @raise [ArgumentError] for any other backend name
66
+ def backend=(name)
67
+ @native = case name
68
+ when :ruby then false
69
+ when :native
70
+ native_library || raise(UnsupportedError, "The \"#{NATIVE_GEM}\" gem could not be loaded")
71
+ when nil then nil
72
+ else raise ArgumentError, "Unknown Snappy backend #{name.inspect} (expected :ruby, :native or nil)"
73
+ end
74
+ end
75
+
76
+ # The native library to use, resolved (and memoized) on first call unless forced by backend=.
77
+ # @return [Module, nil] the +::Snappy+ module, or nil when the pure-Ruby code should run
78
+ def native
79
+ @native = native_library || false if @native.nil?
80
+ @native || nil
81
+ end
82
+
83
+ # Tries to load the snappy gem once and memoizes the result.
84
+ # @return [Module, false] the +::Snappy+ module, or false if it is missing or lacks inflate/deflate
85
+ def native_library
86
+ if @native_lib.nil?
87
+ @native_lib = begin
88
+ require NATIVE_GEM
89
+ (::Snappy.respond_to?(:inflate) && ::Snappy.respond_to?(:deflate)) ? ::Snappy : false
90
+ rescue LoadError
91
+ false
92
+ end
93
+ end
94
+ @native_lib
95
+ end
96
+
30
97
  # Decompresses into a preallocated IO::Buffer: copies do not allocate intermediate Strings.
98
+ # @param src [String] raw snappy block in ASCII-8BIT
99
+ # @return [String] decompressed bytes in ASCII-8BIT
100
+ # @raise [Error] if the block is corrupt or truncated
31
101
  def decompress_io_buffer(src)
32
102
  n = src.bytesize
33
103
  expected, pos = read_varint(src, n)
@@ -98,6 +168,10 @@ module Herringbone
98
168
  inbuf&.free
99
169
  end
100
170
 
171
+ # Decompresses by appending to a String; the fallback when IO::Buffer is not available.
172
+ # @param src [String] raw snappy block in ASCII-8BIT
173
+ # @return [String] decompressed bytes in ASCII-8BIT
174
+ # @raise [Error] if the block is corrupt or truncated
101
175
  def decompress_string(src)
102
176
  n = src.bytesize
103
177
  expected, pos = read_varint(src, n)
@@ -162,10 +236,15 @@ module Herringbone
162
236
 
163
237
  # @param input [String] bytes to compress
164
238
  # @return [String] raw snappy block in ASCII-8BIT
239
+ # @raise [Error] if the input is larger than MAX_UNCOMPRESSED
165
240
  def compress(input)
166
- src = input.encoding == Encoding::BINARY ? input : input.b
241
+ src = (input.encoding == Encoding::BINARY) ? input : input.b
167
242
  n = src.bytesize
243
+ # Checked for libsnappy too: it would truncate the 32-bit length preamble without complaint
168
244
  raise Error, "input too large for snappy" if n > MAX_UNCOMPRESSED
245
+ if (lib = native)
246
+ return lib.deflate(src)
247
+ end
169
248
 
170
249
  out = String.new(capacity: 32 + n + n / 6, encoding: Encoding::BINARY)
171
250
  write_varint(out, n)
@@ -181,6 +260,11 @@ module Herringbone
181
260
  out
182
261
  end
183
262
 
263
+ # Reads the uncompressed-length preamble (a little-endian base-128 varint) at the start of +src+.
264
+ # @param src [String] raw snappy block
265
+ # @param n [Integer] byte size of +src+
266
+ # @return [Array(Integer, Integer)] the declared uncompressed length and the offset just past the varint
267
+ # @raise [Error] if the varint is truncated, longer than 5 bytes or exceeds MAX_UNCOMPRESSED
184
268
  def read_varint(src, n)
185
269
  value = 0
186
270
  shift = 0
@@ -198,6 +282,10 @@ module Herringbone
198
282
  [value, pos]
199
283
  end
200
284
 
285
+ # Appends +value+ as a little-endian base-128 varint (the length preamble).
286
+ # @param out [String] binary output buffer, appended to
287
+ # @param value [Integer] non-negative length to encode
288
+ # @return [String] +out+
201
289
  def write_varint(out, value)
202
290
  while value >= 0x80
203
291
  out << ((value & 0x7f) | 0x80)
@@ -210,6 +298,12 @@ module Herringbone
210
298
  # positions relative to `base`; matches never cross the block boundary.
211
299
  # The 4-byte little-endian word at every position of the block is unpacked up front
212
300
  # (in C, via String#unpack) so hashing and match checks are single Array lookups.
301
+ # @param src [String] whole input in ASCII-8BIT
302
+ # @param base [Integer] offset of the fragment in +src+
303
+ # @param len [Integer] fragment length, at most BLOCK_SIZE
304
+ # @param out [String] binary output buffer, appended to
305
+ # @param table [Array<Integer>] zeroed hash table of 1 << HASH_BITS fragment-relative positions
306
+ # @return [void]
213
307
  def compress_block(src, base, len, out, table)
214
308
  ip_end = base + len
215
309
  next_emit = base
@@ -280,12 +374,24 @@ module Herringbone
280
374
 
281
375
  # words[k][j] is the 4-byte little-endian value at base + 4 * j + k, so the word at
282
376
  # relative position i is words[i & 3][i >> 2]
377
+ # @param src [String] whole input in ASCII-8BIT
378
+ # @param base [Integer] offset of the fragment in +src+
379
+ # @param len [Integer] fragment length
380
+ # @return [Array<Array<Integer>>] four arrays of 32-bit words, one per byte phase
283
381
  def block_words(src, base, len)
284
382
  Array.new(4) { |k| src.byteslice(base + k, len - k).unpack("V*") }
285
383
  end
286
384
 
287
385
  # Like match_length, but compares 4 bytes at a time using the unpacked block words,
288
386
  # which avoids allocating substrings for the (common) short matches.
387
+ # Falls back to match_length once a match reaches 64 bytes.
388
+ # @param words [Array<Array<Integer>>] fragment words from block_words
389
+ # @param src [String] whole input in ASCII-8BIT
390
+ # @param base [Integer] offset of the fragment in +src+
391
+ # @param s1 [Integer] absolute offset of the earlier occurrence
392
+ # @param s2 [Integer] absolute offset of the current position (s1 < s2)
393
+ # @param limit [Integer] absolute offset not to read past (the fragment end)
394
+ # @return [Integer] number of equal bytes starting at s1 and s2
289
395
  def match_length_words(words, src, base, s1, s2, limit)
290
396
  start = s2
291
397
  r1 = s1 - base
@@ -307,6 +413,11 @@ module Herringbone
307
413
 
308
414
  # Number of equal bytes at s1 and s2 (s1 < s2), not reading past limit. Gallops
309
415
  # with byteslice comparisons (memcmp) to avoid per-byte loops on long matches.
416
+ # @param src [String] whole input in ASCII-8BIT
417
+ # @param s1 [Integer] absolute offset of the earlier occurrence
418
+ # @param s2 [Integer] absolute offset of the current position
419
+ # @param limit [Integer] absolute offset not to read past
420
+ # @return [Integer] number of equal bytes
310
421
  def match_length(src, s1, s2, limit)
311
422
  start = s2
312
423
  return 0 if s2 >= limit || src.getbyte(s1) != src.getbyte(s2)
@@ -330,6 +441,13 @@ module Herringbone
330
441
  s2 - start
331
442
  end
332
443
 
444
+ # Appends a literal element: the tag (with a 1-4 byte length extension for long literals)
445
+ # followed by the bytes themselves. Does nothing for an empty literal.
446
+ # @param out [String] binary output buffer, appended to
447
+ # @param src [String] whole input in ASCII-8BIT
448
+ # @param pos [Integer] offset of the literal bytes in +src+
449
+ # @param len [Integer] number of literal bytes
450
+ # @return [String, nil] +out+, or nil when +len+ is zero
333
451
  def emit_literal(out, src, pos, len)
334
452
  return if len == 0
335
453
  n = len - 1
@@ -348,6 +466,11 @@ module Herringbone
348
466
  end
349
467
 
350
468
  # Offsets are always < 64KB (matches stay within a block), so 4-byte offsets are never needed.
469
+ # Long matches are split into copies of at most 64 bytes, never leaving a remainder under 4.
470
+ # @param out [String] binary output buffer, appended to
471
+ # @param offset [Integer] backward distance to the match source, 1...65536
472
+ # @param len [Integer] match length, at least 4
473
+ # @return [String] +out+
351
474
  def emit_copy(out, offset, len)
352
475
  while len >= 68
353
476
  emit_copy_upto64(out, offset, 64)
@@ -360,6 +483,12 @@ module Herringbone
360
483
  emit_copy_upto64(out, offset, len)
361
484
  end
362
485
 
486
+ # Appends one copy element: the 2-byte form (1-byte offset) when len is 4..11 and offset < 2048,
487
+ # otherwise the 3-byte form with a 2-byte offset.
488
+ # @param out [String] binary output buffer, appended to
489
+ # @param offset [Integer] backward distance to the match source, 1...65536
490
+ # @param len [Integer] copy length, 4..64
491
+ # @return [String] +out+
363
492
  def emit_copy_upto64(out, offset, len)
364
493
  if len < 12 && offset < 2048
365
494
  out << (1 | ((len - 4) << 2) | ((offset >> 8) << 5)) << (offset & 0xff)
@@ -368,7 +497,7 @@ module Herringbone
368
497
  end
369
498
  end
370
499
 
371
- private_class_method :decompress_io_buffer, :decompress_string, :read_varint, :write_varint, :compress_block, :block_words, :match_length_words,
500
+ private_class_method :native, :native_library, :decompress_io_buffer, :decompress_string, :read_varint, :write_varint, :compress_block, :block_words, :match_length_words,
372
501
  :match_length, :emit_literal, :emit_copy, :emit_copy_upto64
373
502
  end
374
503
  end
@@ -6,8 +6,12 @@ require "stringio"
6
6
  module Herringbone
7
7
  # Raised when a file uses (or a writer asks for) a codec whose library is not installed
8
8
  class MissingCodecError < UnsupportedError
9
+ # @return [String] codec name (+"ZSTD"+) and name of the gem providing it (+"zstd-ruby"+)
9
10
  attr_reader :codec, :gem_name
10
11
 
12
+ # @param codec [String] codec name, as shown in the message
13
+ # @param gem_name [String] gem to add to the Gemfile
14
+ # @param load_error [LoadError] the error from requiring the gem, whose message is included
11
15
  def initialize(codec, gem_name, load_error)
12
16
  @codec = codec
13
17
  @gem_name = gem_name
@@ -19,6 +23,7 @@ module Herringbone
19
23
  # Dispatches page (de)compression by Parquet codec id. Snappy and LZ4 are pure Ruby and GZIP
20
24
  # uses zlib, so those always work. ZSTD and Brotli come from optional gems (zstd-ruby, brotli),
21
25
  # which are required on first use; if they are missing, MissingCodecError says what to add.
26
+ # (Snappy also uses the optional snappy gem when it is installed, see Codecs::Snappy.)
22
27
  module Compression
23
28
  module_function
24
29
 
@@ -32,6 +37,11 @@ module Herringbone
32
37
  @library_mutex = Mutex.new
33
38
 
34
39
  # The library module for a codec backed by an optional gem, requiring it on first use
40
+ #
41
+ # @param codec [Integer] a codec id that is a key of LIBRARIES
42
+ # @return [Module] the gem's module (+Zstd+ or +Brotli+)
43
+ # @raise [MissingCodecError] when the gem cannot be loaded
44
+ # @raise [KeyError] for a codec id not in LIBRARIES
35
45
  def library(codec)
36
46
  @libraries.fetch(codec) do
37
47
  @library_mutex.synchronize do
@@ -48,11 +58,22 @@ module Herringbone
48
58
  end
49
59
  end
50
60
 
61
+ # Requires a codec gem; a separate method so tests can stub it to simulate a missing gem
62
+ #
63
+ # @param path [String] require path of the gem
64
+ # @return [Boolean] whether the library was newly loaded
65
+ # @raise [LoadError] when the gem is not installed
51
66
  def require_library(path)
52
67
  require path
53
68
  end
54
69
 
55
70
  # Raises MissingCodecError (or UnsupportedError) unless +codec+ can be used
71
+ #
72
+ # @param codec [Integer, Symbol, String] codec id or name (see NAMES)
73
+ # @return [Module, nil] the gem's module for a gem-backed codec, nil for a built-in one
74
+ # @raise [MissingCodecError] when the codec's gem cannot be loaded
75
+ # @raise [UnsupportedError] for a codec herringbone does not implement (LZO)
76
+ # @raise [ArgumentError] for an unknown codec name
56
77
  def ensure_available!(codec)
57
78
  codec = codec_id(codec)
58
79
  return library(codec) if LIBRARIES.key?(codec)
@@ -60,6 +81,7 @@ module Herringbone
60
81
  raise UnsupportedError, "#{Format::Codec::NAMES[codec] || codec} compression is not supported"
61
82
  end
62
83
 
84
+ # Codec ids that work without optional gems
63
85
  SUPPORTED = [
64
86
  Format::Codec::UNCOMPRESSED, Format::Codec::SNAPPY, Format::Codec::GZIP,
65
87
  Format::Codec::LZ4_RAW, Format::Codec::LZ4
@@ -71,8 +93,14 @@ module Herringbone
71
93
  Format::Codec::LZ4_RAW => :lz4, Format::Codec::LZ4 => :lz4_hadoop, Format::Codec::ZSTD => :zstd,
72
94
  Format::Codec::BROTLI => :brotli, Format::Codec::LZO => :lzo
73
95
  }.freeze
96
+ # Codec name => codec id, the inverse of NAMES
74
97
  CODECS_BY_NAME = NAMES.invert.freeze
75
98
 
99
+ # Codec id for a codec name; Integers are taken to be ids already and returned unchecked
100
+ #
101
+ # @param name [Integer, Symbol, String] codec id, or a name from NAMES (case-insensitive)
102
+ # @return [Integer] the Format::Codec id
103
+ # @raise [ArgumentError] for an unknown name
76
104
  def codec_id(name)
77
105
  return name if name.is_a?(Integer)
78
106
  CODECS_BY_NAME.fetch(name.to_s.downcase.to_sym) do
@@ -80,6 +108,15 @@ module Herringbone
80
108
  end
81
109
  end
82
110
 
111
+ # Decompresses a page body, checking the result against the size the page header declares
112
+ #
113
+ # @param codec [Integer] Format::Codec id from the column chunk metadata
114
+ # @param data [String] compressed bytes
115
+ # @param uncompressed_size [Integer] expected decompressed size in bytes
116
+ # @return [String] decompressed bytes (binary)
117
+ # @raise [UnsupportedError] for a codec herringbone does not implement
118
+ # @raise [MissingCodecError] when the codec's gem cannot be loaded
119
+ # @raise [FormatError] when the decompressed size does not match +uncompressed_size+
83
120
  def decompress(codec, data, uncompressed_size)
84
121
  return "".b if uncompressed_size.zero? && data.empty?
85
122
  out = case codec
@@ -100,6 +137,13 @@ module Herringbone
100
137
  out
101
138
  end
102
139
 
140
+ # Compresses a page body
141
+ #
142
+ # @param codec [Integer] Format::Codec id
143
+ # @param data [String] bytes to compress
144
+ # @return [String] compressed bytes (binary)
145
+ # @raise [UnsupportedError] for a codec herringbone does not implement
146
+ # @raise [MissingCodecError] when the codec's gem cannot be loaded
103
147
  def compress(codec, data)
104
148
  case codec
105
149
  when Format::Codec::UNCOMPRESSED then data
@@ -115,6 +159,10 @@ module Herringbone
115
159
  end
116
160
 
117
161
  # Handles files whose gzip data consists of several concatenated members
162
+ #
163
+ # @param data [String] gzip data, one or more members
164
+ # @return [String] the members' decompressed bytes concatenated (binary)
165
+ # @raise [Zlib::GzipFile::Error] on corrupt gzip data
118
166
  def gunzip(data)
119
167
  io = StringIO.new(data)
120
168
  out = String.new(encoding: Encoding::BINARY)