herringbone 0.1.0 → 0.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,33 +1,103 @@
1
1
  # frozen_string_literal: true
2
2
 
3
3
  module Herringbone
4
+ # Pure-Ruby compression codecs for Parquet pages: Snappy, and LZ4 (raw blocks and the Hadoop framing).
5
+ # Compression dispatches to these; GZIP uses Zlib and ZSTD/BROTLI use external gems.
4
6
  module Codecs
5
7
  # Pure-Ruby implementation of the raw Snappy block format (as used by Parquet),
6
8
  # see https://github.com/google/snappy/blob/main/format_description.txt
7
9
  #
10
+ # When the `snappy` gem (a binding to Google's libsnappy) can be loaded, it is used instead:
11
+ # 12x faster decompression and 27x faster compression. It is optional and only a speedup: if
12
+ # it is missing, the pure-Ruby code below is used silently. Both produce raw Snappy blocks the
13
+ # other reads.
14
+ #
8
15
  # 32-bit loads are done with four getbyte calls rather than unpack1(offset:) to stay
9
16
  # compatible with Ruby 3.0 - the speed difference on MRI is marginal.
10
17
  module Snappy
18
+ # Raised for corrupt or truncated Snappy input, and for input too large to compress.
11
19
  class Error < StandardError; end
12
20
 
21
+ # The compressor works on independent 64 KiB fragments, as the reference implementation does.
13
22
  BLOCK_SIZE = 1 << 16
23
+ # Size of the match-finder hash table, in bits (16384 entries).
14
24
  HASH_BITS = 14
25
+ # Right shift that keeps the top HASH_BITS bits of the 32-bit multiplicative hash.
15
26
  HASH_SHIFT = 32 - HASH_BITS
27
+ # Multiplier of the reference implementation's 4-byte hash.
16
28
  HASH_MUL = 0x1e35a7bd
17
29
  INPUT_MARGIN = 15 # bytes at the block end never searched for matches, as in the reference
30
+ # The length preamble is a 32-bit varint, so no block may decompress to more than this.
18
31
  MAX_UNCOMPRESSED = (1 << 32) - 1
19
32
 
33
+ # Name of the optional native gem, passed to +require+.
34
+ NATIVE_GEM = "snappy"
35
+
20
36
  module_function
21
37
 
22
38
  # @param input [String] raw snappy block
23
39
  # @return [String] decompressed bytes in ASCII-8BIT
40
+ # @raise [Error] if the block is corrupt or truncated
24
41
  def decompress(input)
25
- src = input.encoding == Encoding::BINARY ? input : input.b
42
+ src = (input.encoding == Encoding::BINARY) ? input : input.b
43
+ if (lib = native)
44
+ begin
45
+ return lib.inflate(src)
46
+ rescue lib::Error => e
47
+ raise Error, "corrupt snappy data (#{e.message})"
48
+ end
49
+ end
26
50
  return decompress_io_buffer(src) if IOBufferSupport::AVAILABLE
27
51
  decompress_string(src)
28
52
  end
29
53
 
54
+ # The backend in use: :native (the snappy gem) or :ruby
55
+ # @return [Symbol] +:native+ or +:ruby+
56
+ def backend
57
+ native ? :native : :ruby
58
+ end
59
+
60
+ # For tests and benchmarks: :ruby forces pure Ruby, :native requires the snappy gem
61
+ # (UnsupportedError if it cannot be loaded), nil goes back to the default
62
+ # @param name [Symbol, nil] +:ruby+, +:native+ or nil
63
+ # @return [void]
64
+ # @raise [UnsupportedError] if +:native+ is requested and the gem cannot be loaded
65
+ # @raise [ArgumentError] for any other backend name
66
+ def backend=(name)
67
+ @native = case name
68
+ when :ruby then false
69
+ when :native
70
+ native_library || raise(UnsupportedError, "The \"#{NATIVE_GEM}\" gem could not be loaded")
71
+ when nil then nil
72
+ else raise ArgumentError, "Unknown Snappy backend #{name.inspect} (expected :ruby, :native or nil)"
73
+ end
74
+ end
75
+
76
+ # The native library to use, resolved (and memoized) on first call unless forced by backend=.
77
+ # @return [Module, nil] the +::Snappy+ module, or nil when the pure-Ruby code should run
78
+ def native
79
+ @native = native_library || false if @native.nil?
80
+ @native || nil
81
+ end
82
+
83
+ # Tries to load the snappy gem once and memoizes the result.
84
+ # @return [Module, false] the +::Snappy+ module, or false if it is missing or lacks inflate/deflate
85
+ def native_library
86
+ if @native_lib.nil?
87
+ @native_lib = begin
88
+ require NATIVE_GEM
89
+ (::Snappy.respond_to?(:inflate) && ::Snappy.respond_to?(:deflate)) ? ::Snappy : false
90
+ rescue LoadError
91
+ false
92
+ end
93
+ end
94
+ @native_lib
95
+ end
96
+
30
97
  # Decompresses into a preallocated IO::Buffer: copies do not allocate intermediate Strings.
98
+ # @param src [String] raw snappy block in ASCII-8BIT
99
+ # @return [String] decompressed bytes in ASCII-8BIT
100
+ # @raise [Error] if the block is corrupt or truncated
31
101
  def decompress_io_buffer(src)
32
102
  n = src.bytesize
33
103
  expected, pos = read_varint(src, n)
@@ -98,6 +168,10 @@ module Herringbone
98
168
  inbuf&.free
99
169
  end
100
170
 
171
+ # Decompresses by appending to a String; the fallback when IO::Buffer is not available.
172
+ # @param src [String] raw snappy block in ASCII-8BIT
173
+ # @return [String] decompressed bytes in ASCII-8BIT
174
+ # @raise [Error] if the block is corrupt or truncated
101
175
  def decompress_string(src)
102
176
  n = src.bytesize
103
177
  expected, pos = read_varint(src, n)
@@ -162,10 +236,15 @@ module Herringbone
162
236
 
163
237
  # @param input [String] bytes to compress
164
238
  # @return [String] raw snappy block in ASCII-8BIT
239
+ # @raise [Error] if the input is larger than MAX_UNCOMPRESSED
165
240
  def compress(input)
166
- src = input.encoding == Encoding::BINARY ? input : input.b
241
+ src = (input.encoding == Encoding::BINARY) ? input : input.b
167
242
  n = src.bytesize
243
+ # Checked for libsnappy too: it would truncate the 32-bit length preamble without complaint
168
244
  raise Error, "input too large for snappy" if n > MAX_UNCOMPRESSED
245
+ if (lib = native)
246
+ return lib.deflate(src)
247
+ end
169
248
 
170
249
  out = String.new(capacity: 32 + n + n / 6, encoding: Encoding::BINARY)
171
250
  write_varint(out, n)
@@ -181,6 +260,11 @@ module Herringbone
181
260
  out
182
261
  end
183
262
 
263
+ # Reads the uncompressed-length preamble (a little-endian base-128 varint) at the start of +src+.
264
+ # @param src [String] raw snappy block
265
+ # @param n [Integer] byte size of +src+
266
+ # @return [Array(Integer, Integer)] the declared uncompressed length and the offset just past the varint
267
+ # @raise [Error] if the varint is truncated, longer than 5 bytes or exceeds MAX_UNCOMPRESSED
184
268
  def read_varint(src, n)
185
269
  value = 0
186
270
  shift = 0
@@ -198,6 +282,10 @@ module Herringbone
198
282
  [value, pos]
199
283
  end
200
284
 
285
+ # Appends +value+ as a little-endian base-128 varint (the length preamble).
286
+ # @param out [String] binary output buffer, appended to
287
+ # @param value [Integer] non-negative length to encode
288
+ # @return [String] +out+
201
289
  def write_varint(out, value)
202
290
  while value >= 0x80
203
291
  out << ((value & 0x7f) | 0x80)
@@ -210,6 +298,12 @@ module Herringbone
210
298
  # positions relative to `base`; matches never cross the block boundary.
211
299
  # The 4-byte little-endian word at every position of the block is unpacked up front
212
300
  # (in C, via String#unpack) so hashing and match checks are single Array lookups.
301
+ # @param src [String] whole input in ASCII-8BIT
302
+ # @param base [Integer] offset of the fragment in +src+
303
+ # @param len [Integer] fragment length, at most BLOCK_SIZE
304
+ # @param out [String] binary output buffer, appended to
305
+ # @param table [Array<Integer>] zeroed hash table of 1 << HASH_BITS fragment-relative positions
306
+ # @return [void]
213
307
  def compress_block(src, base, len, out, table)
214
308
  ip_end = base + len
215
309
  next_emit = base
@@ -280,12 +374,24 @@ module Herringbone
280
374
 
281
375
  # words[k][j] is the 4-byte little-endian value at base + 4 * j + k, so the word at
282
376
  # relative position i is words[i & 3][i >> 2]
377
+ # @param src [String] whole input in ASCII-8BIT
378
+ # @param base [Integer] offset of the fragment in +src+
379
+ # @param len [Integer] fragment length
380
+ # @return [Array<Array<Integer>>] four arrays of 32-bit words, one per byte phase
283
381
  def block_words(src, base, len)
284
382
  Array.new(4) { |k| src.byteslice(base + k, len - k).unpack("V*") }
285
383
  end
286
384
 
287
385
  # Like match_length, but compares 4 bytes at a time using the unpacked block words,
288
386
  # which avoids allocating substrings for the (common) short matches.
387
+ # Falls back to match_length once a match reaches 64 bytes.
388
+ # @param words [Array<Array<Integer>>] fragment words from block_words
389
+ # @param src [String] whole input in ASCII-8BIT
390
+ # @param base [Integer] offset of the fragment in +src+
391
+ # @param s1 [Integer] absolute offset of the earlier occurrence
392
+ # @param s2 [Integer] absolute offset of the current position (s1 < s2)
393
+ # @param limit [Integer] absolute offset not to read past (the fragment end)
394
+ # @return [Integer] number of equal bytes starting at s1 and s2
289
395
  def match_length_words(words, src, base, s1, s2, limit)
290
396
  start = s2
291
397
  r1 = s1 - base
@@ -307,6 +413,11 @@ module Herringbone
307
413
 
308
414
  # Number of equal bytes at s1 and s2 (s1 < s2), not reading past limit. Gallops
309
415
  # with byteslice comparisons (memcmp) to avoid per-byte loops on long matches.
416
+ # @param src [String] whole input in ASCII-8BIT
417
+ # @param s1 [Integer] absolute offset of the earlier occurrence
418
+ # @param s2 [Integer] absolute offset of the current position
419
+ # @param limit [Integer] absolute offset not to read past
420
+ # @return [Integer] number of equal bytes
310
421
  def match_length(src, s1, s2, limit)
311
422
  start = s2
312
423
  return 0 if s2 >= limit || src.getbyte(s1) != src.getbyte(s2)
@@ -330,6 +441,13 @@ module Herringbone
330
441
  s2 - start
331
442
  end
332
443
 
444
+ # Appends a literal element: the tag (with a 1-4 byte length extension for long literals)
445
+ # followed by the bytes themselves. Does nothing for an empty literal.
446
+ # @param out [String] binary output buffer, appended to
447
+ # @param src [String] whole input in ASCII-8BIT
448
+ # @param pos [Integer] offset of the literal bytes in +src+
449
+ # @param len [Integer] number of literal bytes
450
+ # @return [String, nil] +out+, or nil when +len+ is zero
333
451
  def emit_literal(out, src, pos, len)
334
452
  return if len == 0
335
453
  n = len - 1
@@ -348,6 +466,11 @@ module Herringbone
348
466
  end
349
467
 
350
468
  # Offsets are always < 64KB (matches stay within a block), so 4-byte offsets are never needed.
469
+ # Long matches are split into copies of at most 64 bytes, never leaving a remainder under 4.
470
+ # @param out [String] binary output buffer, appended to
471
+ # @param offset [Integer] backward distance to the match source, 1...65536
472
+ # @param len [Integer] match length, at least 4
473
+ # @return [String] +out+
351
474
  def emit_copy(out, offset, len)
352
475
  while len >= 68
353
476
  emit_copy_upto64(out, offset, 64)
@@ -360,6 +483,12 @@ module Herringbone
360
483
  emit_copy_upto64(out, offset, len)
361
484
  end
362
485
 
486
+ # Appends one copy element: the 2-byte form (1-byte offset) when len is 4..11 and offset < 2048,
487
+ # otherwise the 3-byte form with a 2-byte offset.
488
+ # @param out [String] binary output buffer, appended to
489
+ # @param offset [Integer] backward distance to the match source, 1...65536
490
+ # @param len [Integer] copy length, 4..64
491
+ # @return [String] +out+
363
492
  def emit_copy_upto64(out, offset, len)
364
493
  if len < 12 && offset < 2048
365
494
  out << (1 | ((len - 4) << 2) | ((offset >> 8) << 5)) << (offset & 0xff)
@@ -368,7 +497,7 @@ module Herringbone
368
497
  end
369
498
  end
370
499
 
371
- private_class_method :decompress_io_buffer, :decompress_string, :read_varint, :write_varint, :compress_block, :block_words, :match_length_words,
500
+ private_class_method :native, :native_library, :decompress_io_buffer, :decompress_string, :read_varint, :write_varint, :compress_block, :block_words, :match_length_words,
372
501
  :match_length, :emit_literal, :emit_copy, :emit_copy_upto64
373
502
  end
374
503
  end
@@ -2,27 +2,121 @@
2
2
 
3
3
  require "zlib"
4
4
  require "stringio"
5
- require "zstd-ruby"
6
- require "brotli"
7
5
 
8
6
  module Herringbone
9
- # Dispatches page (de)compression by Parquet codec id. Snappy and LZ4 are pure Ruby,
10
- # GZIP uses zlib, ZSTD and Brotli use the zstd-ruby and brotli gems.
7
+ # Raised when a file uses (or a writer asks for) a codec whose library is not installed
8
+ class MissingCodecError < UnsupportedError
9
+ # @return [String] codec name (+"ZSTD"+) and name of the gem providing it (+"zstd-ruby"+)
10
+ attr_reader :codec, :gem_name
11
+
12
+ # @param codec [String] codec name, as shown in the message
13
+ # @param gem_name [String] gem to add to the Gemfile
14
+ # @param load_error [LoadError] the error from requiring the gem, whose message is included
15
+ def initialize(codec, gem_name, load_error)
16
+ @codec = codec
17
+ @gem_name = gem_name
18
+ super("#{codec} compression needs the \"#{gem_name}\" gem, which could not be loaded " \
19
+ "(#{load_error.message}). Add `gem \"#{gem_name}\"` to your Gemfile to use #{codec}.")
20
+ end
21
+ end
22
+
23
+ # Dispatches page (de)compression by Parquet codec id. Snappy and LZ4 are pure Ruby and GZIP
24
+ # uses zlib, so those always work. ZSTD and Brotli come from optional gems (zstd-ruby, brotli),
25
+ # which are required on first use; if they are missing, MissingCodecError says what to add.
26
+ # (Snappy also uses the optional snappy gem when it is installed, see Codecs::Snappy.)
11
27
  module Compression
12
28
  module_function
13
29
 
14
- CODECS_BY_NAME = {
15
- none: Format::Codec::UNCOMPRESSED, uncompressed: Format::Codec::UNCOMPRESSED,
16
- snappy: Format::Codec::SNAPPY, gzip: Format::Codec::GZIP, brotli: Format::Codec::BROTLI,
17
- lz4: Format::Codec::LZ4_RAW, lz4_raw: Format::Codec::LZ4_RAW, lz4_hadoop: Format::Codec::LZ4,
18
- zstd: Format::Codec::ZSTD
30
+ # Codecs backed by optional native gems: codec id => [name, gem, require path, constant]
31
+ LIBRARIES = {
32
+ Format::Codec::ZSTD => ["ZSTD", "zstd-ruby", "zstd-ruby", :Zstd],
33
+ Format::Codec::BROTLI => ["Brotli", "brotli", "brotli", :Brotli]
34
+ }.freeze
35
+
36
+ @libraries = {}
37
+ @library_mutex = Mutex.new
38
+
39
+ # The library module for a codec backed by an optional gem, requiring it on first use
40
+ #
41
+ # @param codec [Integer] a codec id that is a key of LIBRARIES
42
+ # @return [Module] the gem's module (+Zstd+ or +Brotli+)
43
+ # @raise [MissingCodecError] when the gem cannot be loaded
44
+ # @raise [KeyError] for a codec id not in LIBRARIES
45
+ def library(codec)
46
+ @libraries.fetch(codec) do
47
+ @library_mutex.synchronize do
48
+ @libraries.fetch(codec) do
49
+ name, gem_name, path, const = LIBRARIES.fetch(codec)
50
+ begin
51
+ require_library(path)
52
+ rescue LoadError => e
53
+ raise MissingCodecError.new(name, gem_name, e)
54
+ end
55
+ @libraries[codec] = Object.const_get(const)
56
+ end
57
+ end
58
+ end
59
+ end
60
+
61
+ # Requires a codec gem; a separate method so tests can stub it to simulate a missing gem
62
+ #
63
+ # @param path [String] require path of the gem
64
+ # @return [Boolean] whether the library was newly loaded
65
+ # @raise [LoadError] when the gem is not installed
66
+ def require_library(path)
67
+ require path
68
+ end
69
+
70
+ # Raises MissingCodecError (or UnsupportedError) unless +codec+ can be used
71
+ #
72
+ # @param codec [Integer, Symbol, String] codec id or name (see NAMES)
73
+ # @return [Module, nil] the gem's module for a gem-backed codec, nil for a built-in one
74
+ # @raise [MissingCodecError] when the codec's gem cannot be loaded
75
+ # @raise [UnsupportedError] for a codec herringbone does not implement (LZO)
76
+ # @raise [ArgumentError] for an unknown codec name
77
+ def ensure_available!(codec)
78
+ codec = codec_id(codec)
79
+ return library(codec) if LIBRARIES.key?(codec)
80
+ return if SUPPORTED.include?(codec)
81
+ raise UnsupportedError, "#{Format::Codec::NAMES[codec] || codec} compression is not supported"
82
+ end
83
+
84
+ # Codec ids that work without optional gems
85
+ SUPPORTED = [
86
+ Format::Codec::UNCOMPRESSED, Format::Codec::SNAPPY, Format::Codec::GZIP,
87
+ Format::Codec::LZ4_RAW, Format::Codec::LZ4
88
+ ].freeze
89
+
90
+ # Codec id => name, as given to the writer's compression: option and listed by Herringbone.codecs
91
+ NAMES = {
92
+ Format::Codec::UNCOMPRESSED => :none, Format::Codec::SNAPPY => :snappy, Format::Codec::GZIP => :gzip,
93
+ Format::Codec::LZ4_RAW => :lz4, Format::Codec::LZ4 => :lz4_hadoop, Format::Codec::ZSTD => :zstd,
94
+ Format::Codec::BROTLI => :brotli, Format::Codec::LZO => :lzo
19
95
  }.freeze
96
+ # Codec name => codec id, the inverse of NAMES
97
+ CODECS_BY_NAME = NAMES.invert.freeze
20
98
 
99
+ # Codec id for a codec name; Integers are taken to be ids already and returned unchecked
100
+ #
101
+ # @param name [Integer, Symbol, String] codec id, or a name from NAMES (case-insensitive)
102
+ # @return [Integer] the Format::Codec id
103
+ # @raise [ArgumentError] for an unknown name
21
104
  def codec_id(name)
22
105
  return name if name.is_a?(Integer)
23
- CODECS_BY_NAME.fetch(name.to_s.downcase.to_sym) { raise ArgumentError, "Unknown compression codec #{name.inspect}" }
106
+ CODECS_BY_NAME.fetch(name.to_s.downcase.to_sym) do
107
+ raise ArgumentError, "Unknown compression codec #{name.inspect}, expected one of #{NAMES.values.map(&:inspect).join(", ")}"
108
+ end
24
109
  end
25
110
 
111
+ # Decompresses a page body, checking the result against the size the page header declares
112
+ #
113
+ # @param codec [Integer] Format::Codec id from the column chunk metadata
114
+ # @param data [String] compressed bytes
115
+ # @param uncompressed_size [Integer] expected decompressed size in bytes
116
+ # @return [String] decompressed bytes (binary)
117
+ # @raise [UnsupportedError] for a codec herringbone does not implement
118
+ # @raise [MissingCodecError] when the codec's gem cannot be loaded
119
+ # @raise [FormatError] when the decompressed size does not match +uncompressed_size+
26
120
  def decompress(codec, data, uncompressed_size)
27
121
  return "".b if uncompressed_size.zero? && data.empty?
28
122
  out = case codec
@@ -31,18 +125,25 @@ module Herringbone
31
125
  when Format::Codec::GZIP then gunzip(data)
32
126
  when Format::Codec::LZ4_RAW then Codecs::LZ4.decompress_block(data, uncompressed_size)
33
127
  when Format::Codec::LZ4 then Codecs::LZ4.decompress_hadoop(data, uncompressed_size)
34
- when Format::Codec::ZSTD then ::Zstd.decompress(data)
35
- when Format::Codec::BROTLI then ::Brotli.inflate(data)
128
+ when Format::Codec::ZSTD then library(codec).decompress(data)
129
+ when Format::Codec::BROTLI then library(codec).inflate(data)
36
130
  else
37
131
  raise UnsupportedError, "Unsupported compression codec #{Format::Codec::NAMES[codec] || codec}"
38
132
  end
39
133
  out = out.b
40
134
  if out.bytesize != uncompressed_size
41
- raise DecodeError, "Decompressed #{out.bytesize} bytes, expected #{uncompressed_size}"
135
+ raise FormatError, "Decompressed #{out.bytesize} bytes, expected #{uncompressed_size}"
42
136
  end
43
137
  out
44
138
  end
45
139
 
140
+ # Compresses a page body
141
+ #
142
+ # @param codec [Integer] Format::Codec id
143
+ # @param data [String] bytes to compress
144
+ # @return [String] compressed bytes (binary)
145
+ # @raise [UnsupportedError] for a codec herringbone does not implement
146
+ # @raise [MissingCodecError] when the codec's gem cannot be loaded
46
147
  def compress(codec, data)
47
148
  case codec
48
149
  when Format::Codec::UNCOMPRESSED then data
@@ -50,18 +151,22 @@ module Herringbone
50
151
  when Format::Codec::GZIP then Zlib.gzip(data)
51
152
  when Format::Codec::LZ4_RAW then Codecs::LZ4.compress_block(data)
52
153
  when Format::Codec::LZ4 then Codecs::LZ4.compress_hadoop(data)
53
- when Format::Codec::ZSTD then ::Zstd.compress(data)
54
- when Format::Codec::BROTLI then ::Brotli.deflate(data)
154
+ when Format::Codec::ZSTD then library(codec).compress(data)
155
+ when Format::Codec::BROTLI then library(codec).deflate(data)
55
156
  else
56
157
  raise UnsupportedError, "Unsupported compression codec #{Format::Codec::NAMES[codec] || codec}"
57
158
  end.b
58
159
  end
59
160
 
60
161
  # Handles files whose gzip data consists of several concatenated members
162
+ #
163
+ # @param data [String] gzip data, one or more members
164
+ # @return [String] the members' decompressed bytes concatenated (binary)
165
+ # @raise [Zlib::GzipFile::Error] on corrupt gzip data
61
166
  def gunzip(data)
62
167
  io = StringIO.new(data)
63
168
  out = String.new(encoding: Encoding::BINARY)
64
- loop do
169
+ while true
65
170
  gz = Zlib::GzipReader.new(io)
66
171
  out << gz.read
67
172
  unused = gz.unused
@@ -6,56 +6,84 @@ module Herringbone
6
6
  module Delta
7
7
  module_function
8
8
 
9
+ # Values per block written by the encoder (the size parquet-mr and Arrow use).
9
10
  BLOCK_SIZE = 128
11
+ # Miniblocks per block written by the encoder, so 32 values per miniblock.
10
12
  MINIBLOCKS = 4
11
13
 
14
+ # @param n [Integer] zigzag-encoded unsigned integer
15
+ # @return [Integer] the signed integer it represents
12
16
  def zigzag_decode(n) = (n >> 1) ^ -(n & 1)
17
+
18
+ # @param n [Integer] signed integer
19
+ # @return [Integer] zigzag-encoded form (0, -1, 1, -2 map to 0, 1, 2, 3)
13
20
  def zigzag_encode(n) = n.negative? ? ((-n) << 1) - 1 : n << 1
14
21
 
15
22
  # Wraps an integer into the signed range of +bits+ bits
23
+ # @param v [Integer] integer to wrap
24
+ # @param bits [Integer] target width, 32 or 64
25
+ # @return [Integer] +v+ modulo 2^bits, as a two's complement signed integer
16
26
  def wrap(v, bits)
17
27
  half = 1 << (bits - 1)
18
28
  ((v + half) & ((1 << bits) - 1)) - half
19
29
  end
20
30
 
21
31
  # Decodes DELTA_BINARY_PACKED integers. +bits+ is 32 or 64 (for wraparound).
22
- # Returns [values, new_pos]. If +count+ is nil, the total from the header is used.
32
+ # Returns [values, new_pos]. If +count+ is nil, the total from the header is used. A smaller
33
+ # +count+ decodes only that many values, but the returned offset is still the end of the
34
+ # whole encoded block, so data following it can be read from there.
35
+ # @param data [String] binary page data
36
+ # @param pos [Integer] byte offset of the block header
37
+ # @param bits [Integer] integer width that deltas wrap around in, 32 or 64
38
+ # @param count [Integer, nil] maximum number of values to decode
39
+ # @return [Array(Array<Integer>, Integer)] the decoded values and the offset just past the encoded block
40
+ # @raise [FormatError] on an invalid header or miniblock bit width, or truncated data
23
41
  def decode_binary_packed(data, pos, bits = 64, count = nil)
24
42
  block_size, pos = RLE.read_uleb(data, pos)
25
43
  miniblocks, pos = RLE.read_uleb(data, pos)
26
44
  total, pos = RLE.read_uleb(data, pos)
27
45
  first, pos = RLE.read_uleb(data, pos)
28
- raise DecodeError, "Invalid DELTA_BINARY_PACKED header" if miniblocks.zero? || block_size % miniblocks != 0
46
+ raise FormatError, "Invalid DELTA_BINARY_PACKED header" if miniblocks.zero? || block_size % miniblocks != 0
29
47
  per_mini = block_size / miniblocks
30
- raise DecodeError, "Invalid miniblock size #{per_mini}" if per_mini % 8 != 0
31
- total = count if count && count < total
48
+ raise FormatError, "Invalid miniblock size #{per_mini}" if per_mini % 8 != 0
49
+ want = (count && count < total) ? count : total
32
50
  values = []
33
51
  return [values, pos] if total.zero?
34
52
  last = zigzag_decode(first)
35
- values << last
53
+ values << last if want.positive?
36
54
  half = 1 << (bits - 1)
37
55
  mask = (1 << bits) - 1
38
- while values.size < total
56
+ left = total - 1 # deltas still encoded, decoded or skipped
57
+ while left.positive?
39
58
  min_delta, pos = RLE.read_uleb(data, pos)
40
59
  min_delta = zigzag_decode(min_delta)
41
60
  widths = data.byteslice(pos, miniblocks).unpack("C*")
42
61
  pos += miniblocks
43
62
  widths.each do |w|
44
- break if values.size >= total
45
- raise DecodeError, "Invalid delta bit width #{w}" if w > bits
46
- deltas = RLE.unpack_bits(data, pos, per_mini, w)
47
- pos += per_mini * w / 8
48
- take = total - values.size
49
- deltas = deltas.first(take) if take < per_mini
50
- deltas.each do |d|
51
- last = ((last + min_delta + d + half) & mask) - half
52
- values << last
63
+ # Miniblocks past the last value have a width byte but no body
64
+ break unless left.positive?
65
+ raise FormatError, "Invalid delta bit width #{w}" if w > bits
66
+ if values.size < want
67
+ deltas = RLE.unpack_bits(data, pos, per_mini, w)
68
+ take = want - values.size
69
+ deltas = deltas.first(take) if take < per_mini
70
+ deltas.each do |d|
71
+ last = ((last + min_delta + d + half) & mask) - half
72
+ values << last
73
+ end
53
74
  end
75
+ pos += per_mini * w / 8
76
+ left -= per_mini
54
77
  end
55
78
  end
56
79
  [values, pos]
57
80
  end
58
81
 
82
+ # Encodes integers with DELTA_BINARY_PACKED, using BLOCK_SIZE values per block in MINIBLOCKS
83
+ # miniblocks. Deltas wrap at +bits+ so INT32 columns never need more than 32-bit widths.
84
+ # @param values [Array<Integer>] signed integers that fit in +bits+ bits
85
+ # @param bits [Integer] physical integer width, 32 or 64
86
+ # @return [String] encoded bytes in ASCII-8BIT
59
87
  def encode_binary_packed(values, bits = 64)
60
88
  out = String.new(encoding: Encoding::BINARY)
61
89
  per_mini = BLOCK_SIZE / MINIBLOCKS
@@ -82,35 +110,54 @@ module Herringbone
82
110
  out
83
111
  end
84
112
 
113
+ # Decodes DELTA_LENGTH_BYTE_ARRAY: DELTA_BINARY_PACKED lengths followed by the concatenated bytes.
114
+ # @param data [String] binary page data
115
+ # @param pos [Integer] byte offset of the lengths block
116
+ # @param count [Integer] number of values to decode
117
+ # @return [Array(Array<String>, Integer)] binary slices of +data+ and the offset just past them
118
+ # @raise [FormatError] if there are too few lengths or a value runs past the end of +data+
85
119
  def decode_length_byte_array(data, pos, count)
86
120
  lengths, pos = decode_binary_packed(data, pos, 32, count)
87
- raise DecodeError, "DELTA_LENGTH_BYTE_ARRAY has #{lengths.size} lengths, need #{count}" if lengths.size < count
121
+ raise FormatError, "DELTA_LENGTH_BYTE_ARRAY has #{lengths.size} lengths, need #{count}" if lengths.size < count
88
122
  out = Array.new(count)
89
123
  size = data.bytesize
90
124
  lengths.each_with_index do |len, i|
91
- raise DecodeError, "DELTA_LENGTH_BYTE_ARRAY value overruns page" if len.negative? || pos + len > size
125
+ raise FormatError, "DELTA_LENGTH_BYTE_ARRAY value overruns page" if len.negative? || pos + len > size
92
126
  out[i] = data.byteslice(pos, len)
93
127
  pos += len
94
128
  end
95
129
  [out, pos]
96
130
  end
97
131
 
132
+ # Encodes byte strings with DELTA_LENGTH_BYTE_ARRAY.
133
+ # @param values [Array<String>] byte strings, in any encoding
134
+ # @return [String] encoded bytes in ASCII-8BIT
98
135
  def encode_length_byte_array(values)
99
136
  encode_binary_packed(values.map(&:bytesize), 32) << values.pack("a*" * values.size)
100
137
  end
101
138
 
139
+ # Decodes DELTA_BYTE_ARRAY (incremental encoding): DELTA_BINARY_PACKED prefix lengths, then
140
+ # the suffixes as DELTA_LENGTH_BYTE_ARRAY. Each value is the previous value's prefix plus its suffix.
141
+ # @param data [String] binary page data
142
+ # @param pos [Integer] byte offset of the prefix lengths block
143
+ # @param count [Integer] number of values to decode
144
+ # @return [Array(Array<String>, Integer)] binary values and the offset just past them
145
+ # @raise [FormatError] if a prefix is negative or longer than the previous value, or the data is malformed
102
146
  def decode_byte_array(data, pos, count)
103
147
  prefixes, pos = decode_binary_packed(data, pos, 32, count)
104
148
  suffixes, pos = decode_length_byte_array(data, pos, count)
105
149
  prev = "".b
106
150
  out = Array.new(count) do |i|
107
151
  prefix = prefixes[i]
108
- raise DecodeError, "DELTA_BYTE_ARRAY prefix longer than previous value" if prefix > prev.bytesize
152
+ raise FormatError, "DELTA_BYTE_ARRAY prefix length #{prefix} out of range" if prefix > prev.bytesize || prefix.negative?
109
153
  prev = prefix.zero? ? suffixes[i] : prev.byteslice(0, prefix) + suffixes[i]
110
154
  end
111
155
  [out, pos]
112
156
  end
113
157
 
158
+ # Encodes byte strings with DELTA_BYTE_ARRAY, sharing the common byte prefix with the previous value.
159
+ # @param values [Array<String>] byte strings, in any encoding
160
+ # @return [String] encoded bytes in ASCII-8BIT
114
161
  def encode_byte_array(values)
115
162
  prev = "".b
116
163
  prefixes = []
@@ -132,13 +179,23 @@ module Herringbone
132
179
  module_function
133
180
 
134
181
  # Returns the value bytes re-interleaved into PLAIN layout
182
+ # @param data [String] binary page data
183
+ # @param pos [Integer] byte offset of the first stream
184
+ # @param count [Integer] number of values
185
+ # @param width [Integer] byte width of one value
186
+ # @return [Array(String, Integer)] PLAIN-layout bytes and the offset just past the streams
187
+ # @raise [FormatError] if fewer than count * width bytes remain
135
188
  def decode(data, pos, count, width)
136
189
  nbytes = count * width
137
- raise DecodeError, "Truncated BYTE_STREAM_SPLIT data" if pos + nbytes > data.bytesize
190
+ raise FormatError, "Truncated BYTE_STREAM_SPLIT data" if pos + nbytes > data.bytesize
138
191
  streams = Array.new(width) { |k| data.byteslice(pos + k * count, count).unpack("C*") }
139
192
  [streams[0].zip(*streams[1..]).flatten.pack("C*"), pos + nbytes]
140
193
  end
141
194
 
195
+ # Splits PLAIN-layout fixed-width values into +width+ byte streams. Inverse of decode.
196
+ # @param plain [String] PLAIN-encoded values, a multiple of +width+ bytes
197
+ # @param width [Integer] byte width of one value
198
+ # @return [String] the concatenated streams in ASCII-8BIT
142
199
  def encode(plain, width)
143
200
  count = plain.bytesize / width
144
201
  bytes = plain.unpack("C*")