herringbone 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,75 @@
1
+ # frozen_string_literal: true
2
+
3
+ require "zlib"
4
+ require "stringio"
5
+ require "zstd-ruby"
6
+ require "brotli"
7
+
8
+ module Herringbone
9
+ # Dispatches page (de)compression by Parquet codec id. Snappy and LZ4 are pure Ruby,
10
+ # GZIP uses zlib, ZSTD and Brotli use the zstd-ruby and brotli gems.
11
+ module Compression
12
+ module_function
13
+
14
+ CODECS_BY_NAME = {
15
+ none: Format::Codec::UNCOMPRESSED, uncompressed: Format::Codec::UNCOMPRESSED,
16
+ snappy: Format::Codec::SNAPPY, gzip: Format::Codec::GZIP, brotli: Format::Codec::BROTLI,
17
+ lz4: Format::Codec::LZ4_RAW, lz4_raw: Format::Codec::LZ4_RAW, lz4_hadoop: Format::Codec::LZ4,
18
+ zstd: Format::Codec::ZSTD
19
+ }.freeze
20
+
21
+ def codec_id(name)
22
+ return name if name.is_a?(Integer)
23
+ CODECS_BY_NAME.fetch(name.to_s.downcase.to_sym) { raise ArgumentError, "Unknown compression codec #{name.inspect}" }
24
+ end
25
+
26
+ def decompress(codec, data, uncompressed_size)
27
+ return "".b if uncompressed_size.zero? && data.empty?
28
+ out = case codec
29
+ when Format::Codec::UNCOMPRESSED then data
30
+ when Format::Codec::SNAPPY then Codecs::Snappy.decompress(data)
31
+ when Format::Codec::GZIP then gunzip(data)
32
+ when Format::Codec::LZ4_RAW then Codecs::LZ4.decompress_block(data, uncompressed_size)
33
+ when Format::Codec::LZ4 then Codecs::LZ4.decompress_hadoop(data, uncompressed_size)
34
+ when Format::Codec::ZSTD then ::Zstd.decompress(data)
35
+ when Format::Codec::BROTLI then ::Brotli.inflate(data)
36
+ else
37
+ raise UnsupportedError, "Unsupported compression codec #{Format::Codec::NAMES[codec] || codec}"
38
+ end
39
+ out = out.b
40
+ if out.bytesize != uncompressed_size
41
+ raise DecodeError, "Decompressed #{out.bytesize} bytes, expected #{uncompressed_size}"
42
+ end
43
+ out
44
+ end
45
+
46
+ def compress(codec, data)
47
+ case codec
48
+ when Format::Codec::UNCOMPRESSED then data
49
+ when Format::Codec::SNAPPY then Codecs::Snappy.compress(data)
50
+ when Format::Codec::GZIP then Zlib.gzip(data)
51
+ when Format::Codec::LZ4_RAW then Codecs::LZ4.compress_block(data)
52
+ when Format::Codec::LZ4 then Codecs::LZ4.compress_hadoop(data)
53
+ when Format::Codec::ZSTD then ::Zstd.compress(data)
54
+ when Format::Codec::BROTLI then ::Brotli.deflate(data)
55
+ else
56
+ raise UnsupportedError, "Unsupported compression codec #{Format::Codec::NAMES[codec] || codec}"
57
+ end.b
58
+ end
59
+
60
+ # Handles files whose gzip data consists of several concatenated members
61
+ def gunzip(data)
62
+ io = StringIO.new(data)
63
+ out = String.new(encoding: Encoding::BINARY)
64
+ loop do
65
+ gz = Zlib::GzipReader.new(io)
66
+ out << gz.read
67
+ unused = gz.unused
68
+ gz.finish
69
+ break if unused.nil? || unused.empty?
70
+ io.pos -= unused.bytesize
71
+ end
72
+ out
73
+ end
74
+ end
75
+ end
@@ -0,0 +1,153 @@
1
+ # frozen_string_literal: true
2
+
3
+ module Herringbone
4
+ module Encodings
5
+ # DELTA_BINARY_PACKED, DELTA_LENGTH_BYTE_ARRAY and DELTA_BYTE_ARRAY
6
+ module Delta
7
+ module_function
8
+
9
+ BLOCK_SIZE = 128
10
+ MINIBLOCKS = 4
11
+
12
+ def zigzag_decode(n) = (n >> 1) ^ -(n & 1)
13
+ def zigzag_encode(n) = n.negative? ? ((-n) << 1) - 1 : n << 1
14
+
15
+ # Wraps an integer into the signed range of +bits+ bits
16
+ def wrap(v, bits)
17
+ half = 1 << (bits - 1)
18
+ ((v + half) & ((1 << bits) - 1)) - half
19
+ end
20
+
21
+ # Decodes DELTA_BINARY_PACKED integers. +bits+ is 32 or 64 (for wraparound).
22
+ # Returns [values, new_pos]. If +count+ is nil, the total from the header is used.
23
+ def decode_binary_packed(data, pos, bits = 64, count = nil)
24
+ block_size, pos = RLE.read_uleb(data, pos)
25
+ miniblocks, pos = RLE.read_uleb(data, pos)
26
+ total, pos = RLE.read_uleb(data, pos)
27
+ first, pos = RLE.read_uleb(data, pos)
28
+ raise DecodeError, "Invalid DELTA_BINARY_PACKED header" if miniblocks.zero? || block_size % miniblocks != 0
29
+ per_mini = block_size / miniblocks
30
+ raise DecodeError, "Invalid miniblock size #{per_mini}" if per_mini % 8 != 0
31
+ total = count if count && count < total
32
+ values = []
33
+ return [values, pos] if total.zero?
34
+ last = zigzag_decode(first)
35
+ values << last
36
+ half = 1 << (bits - 1)
37
+ mask = (1 << bits) - 1
38
+ while values.size < total
39
+ min_delta, pos = RLE.read_uleb(data, pos)
40
+ min_delta = zigzag_decode(min_delta)
41
+ widths = data.byteslice(pos, miniblocks).unpack("C*")
42
+ pos += miniblocks
43
+ widths.each do |w|
44
+ break if values.size >= total
45
+ raise DecodeError, "Invalid delta bit width #{w}" if w > bits
46
+ deltas = RLE.unpack_bits(data, pos, per_mini, w)
47
+ pos += per_mini * w / 8
48
+ take = total - values.size
49
+ deltas = deltas.first(take) if take < per_mini
50
+ deltas.each do |d|
51
+ last = ((last + min_delta + d + half) & mask) - half
52
+ values << last
53
+ end
54
+ end
55
+ end
56
+ [values, pos]
57
+ end
58
+
59
+ def encode_binary_packed(values, bits = 64)
60
+ out = String.new(encoding: Encoding::BINARY)
61
+ per_mini = BLOCK_SIZE / MINIBLOCKS
62
+ RLE.write_uleb(out, BLOCK_SIZE)
63
+ RLE.write_uleb(out, MINIBLOCKS)
64
+ RLE.write_uleb(out, values.size)
65
+ RLE.write_uleb(out, zigzag_encode(values.empty? ? 0 : values[0]))
66
+ return out if values.size <= 1
67
+
68
+ mask = (1 << bits) - 1
69
+ deltas = Array.new(values.size - 1) { |i| wrap(values[i + 1] - values[i], bits) }
70
+ deltas.each_slice(BLOCK_SIZE) do |block|
71
+ min = block.min
72
+ RLE.write_uleb(out, zigzag_encode(min))
73
+ adjusted = block.map { |d| (d - min) & mask }
74
+ minis = adjusted.each_slice(per_mini).to_a
75
+ widths = Array.new(MINIBLOCKS) { |m| minis[m] ? minis[m].max.bit_length : 0 }
76
+ out << widths.pack("C*")
77
+ minis.each_with_index do |mini, m|
78
+ mini += Array.new(per_mini - mini.size, 0) if mini.size < per_mini
79
+ out << RLE.pack_bits(mini, widths[m])
80
+ end
81
+ end
82
+ out
83
+ end
84
+
85
+ def decode_length_byte_array(data, pos, count)
86
+ lengths, pos = decode_binary_packed(data, pos, 32, count)
87
+ raise DecodeError, "DELTA_LENGTH_BYTE_ARRAY has #{lengths.size} lengths, need #{count}" if lengths.size < count
88
+ out = Array.new(count)
89
+ size = data.bytesize
90
+ lengths.each_with_index do |len, i|
91
+ raise DecodeError, "DELTA_LENGTH_BYTE_ARRAY value overruns page" if len.negative? || pos + len > size
92
+ out[i] = data.byteslice(pos, len)
93
+ pos += len
94
+ end
95
+ [out, pos]
96
+ end
97
+
98
+ def encode_length_byte_array(values)
99
+ encode_binary_packed(values.map(&:bytesize), 32) << values.pack("a*" * values.size)
100
+ end
101
+
102
+ def decode_byte_array(data, pos, count)
103
+ prefixes, pos = decode_binary_packed(data, pos, 32, count)
104
+ suffixes, pos = decode_length_byte_array(data, pos, count)
105
+ prev = "".b
106
+ out = Array.new(count) do |i|
107
+ prefix = prefixes[i]
108
+ raise DecodeError, "DELTA_BYTE_ARRAY prefix longer than previous value" if prefix > prev.bytesize
109
+ prev = prefix.zero? ? suffixes[i] : prev.byteslice(0, prefix) + suffixes[i]
110
+ end
111
+ [out, pos]
112
+ end
113
+
114
+ def encode_byte_array(values)
115
+ prev = "".b
116
+ prefixes = []
117
+ suffixes = []
118
+ values.each do |v|
119
+ max = [prev.bytesize, v.bytesize].min
120
+ n = 0
121
+ n += 1 while n < max && prev.getbyte(n) == v.getbyte(n)
122
+ prefixes << n
123
+ suffixes << v.byteslice(n, v.bytesize - n)
124
+ prev = v
125
+ end
126
+ encode_binary_packed(prefixes, 32) << encode_length_byte_array(suffixes)
127
+ end
128
+ end
129
+
130
+ # BYTE_STREAM_SPLIT: byte k of every value is stored in stream k.
131
+ module ByteStreamSplit
132
+ module_function
133
+
134
+ # Returns the value bytes re-interleaved into PLAIN layout
135
+ def decode(data, pos, count, width)
136
+ nbytes = count * width
137
+ raise DecodeError, "Truncated BYTE_STREAM_SPLIT data" if pos + nbytes > data.bytesize
138
+ streams = Array.new(width) { |k| data.byteslice(pos + k * count, count).unpack("C*") }
139
+ [streams[0].zip(*streams[1..]).flatten.pack("C*"), pos + nbytes]
140
+ end
141
+
142
+ def encode(plain, width)
143
+ count = plain.bytesize / width
144
+ bytes = plain.unpack("C*")
145
+ out = String.new(capacity: plain.bytesize, encoding: Encoding::BINARY)
146
+ width.times do |k|
147
+ out << Array.new(count) { |i| bytes[i * width + k] }.pack("C*")
148
+ end
149
+ out
150
+ end
151
+ end
152
+ end
153
+ end
@@ -0,0 +1,87 @@
1
+ # frozen_string_literal: true
2
+
3
+ module Herringbone
4
+ module Encodings
5
+ # PLAIN encoding for all physical types. Values are returned in their physical
6
+ # Ruby form (Integer, Float, true/false, binary String); logical conversion happens elsewhere.
7
+ module Plain
8
+ module_function
9
+
10
+ FORMATS = {
11
+ Format::Type::INT32 => ["l<", 4],
12
+ Format::Type::INT64 => ["q<", 8],
13
+ Format::Type::FLOAT => ["e", 4],
14
+ Format::Type::DOUBLE => ["E", 8]
15
+ }.freeze
16
+
17
+ # Decodes +count+ values of +type+ from +data+ starting at +pos+.
18
+ # Returns [values, new_pos].
19
+ def decode(data, pos, count, type, type_length = nil)
20
+ case type
21
+ when Format::Type::BOOLEAN
22
+ nbytes = (count + 7) / 8
23
+ bits = data.byteslice(pos, nbytes).unpack1("b*")
24
+ raise DecodeError, "Truncated BOOLEAN data" if bits.bytesize < count
25
+ [Array.new(count) { |i| bits.getbyte(i) == 49 }, pos + nbytes]
26
+ when Format::Type::INT32, Format::Type::INT64, Format::Type::FLOAT, Format::Type::DOUBLE
27
+ fmt, width = FORMATS[type]
28
+ nbytes = count * width
29
+ raise DecodeError, "Truncated PLAIN data" if pos + nbytes > data.bytesize
30
+ [data.byteslice(pos, nbytes).unpack("#{fmt}#{count}"), pos + nbytes]
31
+ when Format::Type::INT96
32
+ nbytes = count * 12
33
+ raise DecodeError, "Truncated INT96 data" if pos + nbytes > data.bytesize
34
+ vals = data.byteslice(pos, nbytes).unpack("Q<L<" * count).each_slice(2).map { |nanos, day| [nanos, day] }
35
+ [vals, pos + nbytes]
36
+ when Format::Type::BYTE_ARRAY
37
+ decode_byte_arrays(data, pos, count)
38
+ when Format::Type::FIXED_LEN_BYTE_ARRAY
39
+ nbytes = count * type_length
40
+ raise DecodeError, "Truncated FIXED_LEN_BYTE_ARRAY data" if pos + nbytes > data.bytesize
41
+ [Array.new(count) { |i| data.byteslice(pos + i * type_length, type_length) }, pos + nbytes]
42
+ else
43
+ raise DecodeError, "Unknown physical type #{type}"
44
+ end
45
+ end
46
+
47
+ def decode_byte_arrays(data, pos, count)
48
+ out = Array.new(count)
49
+ size = data.bytesize
50
+ i = 0
51
+ while i < count
52
+ raise DecodeError, "Truncated BYTE_ARRAY data" if pos + 4 > size
53
+ len = data.getbyte(pos) | (data.getbyte(pos + 1) << 8) | (data.getbyte(pos + 2) << 16) | (data.getbyte(pos + 3) << 24)
54
+ pos += 4
55
+ raise DecodeError, "BYTE_ARRAY value overruns page" if pos + len > size
56
+ out[i] = data.byteslice(pos, len)
57
+ pos += len
58
+ i += 1
59
+ end
60
+ [out, pos]
61
+ end
62
+
63
+ def encode(values, type, type_length = nil)
64
+ case type
65
+ when Format::Type::BOOLEAN
66
+ [values.map { |v| v ? "1" : "0" }.join].pack("b*")
67
+ when Format::Type::INT32, Format::Type::INT64, Format::Type::FLOAT, Format::Type::DOUBLE
68
+ values.pack("#{FORMATS[type][0]}*")
69
+ when Format::Type::INT96
70
+ values.flat_map { |nanos, day| [nanos, day] }.pack("Q<L<" * values.size)
71
+ when Format::Type::BYTE_ARRAY
72
+ # pack("a*") copies raw bytes whatever the string's encoding, without an intermediate copy
73
+ args = []
74
+ values.each { |v| args << v.bytesize << v }
75
+ args.pack("Va*" * values.size)
76
+ when Format::Type::FIXED_LEN_BYTE_ARRAY
77
+ values.each do |v|
78
+ raise EncodeError, "Expected #{type_length} bytes, got #{v.bytesize}" if v.bytesize != type_length
79
+ end
80
+ values.pack("a*" * values.size)
81
+ else
82
+ raise EncodeError, "Unknown physical type #{type}"
83
+ end
84
+ end
85
+ end
86
+ end
87
+ end
@@ -0,0 +1,203 @@
1
+ # frozen_string_literal: true
2
+
3
+ module Herringbone
4
+ module Encodings
5
+ # Bit packing (LSB-first, as used by Parquet) and the RLE / bit-packed hybrid encoding
6
+ # used for repetition/definition levels, dictionary indices and RLE booleans.
7
+ module RLE
8
+ module_function
9
+
10
+ # Unpacks +count+ values of +width+ bits each, starting at byte +offset+ of +data+.
11
+ # Missing trailing bytes are treated as zeroes.
12
+ def unpack_bits(data, offset, count, width)
13
+ return Array.new(count, 0) if width.zero?
14
+ nbytes = (count * width + 7) / 8
15
+ chunk = data.byteslice(offset, nbytes) || "".b
16
+ if width == 8
17
+ vals = chunk.unpack("C*")
18
+ vals.fill(0, vals.size, count - vals.size) if vals.size < count
19
+ return vals
20
+ end
21
+ return unpack_wide(chunk, count, width) if width > 32
22
+
23
+ # Pad to a whole number of 32-bit words plus one spare word, so reads never go out of range
24
+ pad = (-chunk.bytesize % 4) + 4
25
+ words = (pad.zero? ? chunk : chunk + ("\0" * pad)).unpack("V*")
26
+ mask = (1 << width) - 1
27
+ out = Array.new(count)
28
+ bitpos = 0
29
+ i = 0
30
+ while i < count
31
+ wi = bitpos >> 5
32
+ off = bitpos & 31
33
+ v = words[wi] >> off
34
+ spill = off + width - 32
35
+ v |= (words[wi + 1] & ((1 << spill) - 1)) << (32 - off) if spill.positive?
36
+ out[i] = v & mask
37
+ bitpos += width
38
+ i += 1
39
+ end
40
+ out
41
+ end
42
+
43
+ # Slow path for widths above 32 bits (used by DELTA_BINARY_PACKED with 64-bit values)
44
+ def unpack_wide(chunk, count, width)
45
+ mask = (1 << width) - 1
46
+ out = Array.new(count)
47
+ group_bytes = width # 8 values * width bits = width bytes
48
+ ngroups = (count + 7) / 8
49
+ i = 0
50
+ ngroups.times do |g|
51
+ bytes = chunk.byteslice(g * group_bytes, group_bytes) || "".b
52
+ bytes += "\0" * (group_bytes - bytes.bytesize) if bytes.bytesize < group_bytes
53
+ big = bytes.reverse.unpack1("H*").to_i(16)
54
+ 8.times do |j|
55
+ break if i >= count
56
+ out[i] = (big >> (j * width)) & mask
57
+ i += 1
58
+ end
59
+ end
60
+ out
61
+ end
62
+
63
+ # Packs +values+ with +width+ bits each; the value count is padded up to a multiple of 8.
64
+ def pack_bits(values, width)
65
+ return "".b if width.zero? || values.empty?
66
+ n = values.size
67
+ padded = (n + 7) & ~7
68
+ if width == 8
69
+ s = values.pack("C*")
70
+ s << ("\0" * (padded - n)) if padded > n
71
+ return s
72
+ end
73
+ out = String.new(capacity: padded * width / 8, encoding: Encoding::BINARY)
74
+ acc = 0
75
+ nbits = 0
76
+ i = 0
77
+ while i < padded
78
+ acc |= (values[i] || 0) << nbits
79
+ nbits += width
80
+ while nbits >= 32
81
+ out << [acc & 0xFFFFFFFF].pack("V")
82
+ acc >>= 32
83
+ nbits -= 32
84
+ end
85
+ i += 1
86
+ end
87
+ # Remaining bits: padded*width is a multiple of 8, so nbits is a whole number of bytes
88
+ while nbits.positive?
89
+ out << (acc & 0xFF)
90
+ acc >>= 8
91
+ nbits -= 8
92
+ end
93
+ out
94
+ end
95
+
96
+ def read_uleb(data, pos)
97
+ result = 0
98
+ shift = 0
99
+ while true
100
+ b = data.getbyte(pos)
101
+ raise DecodeError, "Truncated varint" unless b
102
+ pos += 1
103
+ result |= (b & 0x7F) << shift
104
+ return [result, pos] if b < 0x80
105
+ shift += 7
106
+ end
107
+ end
108
+
109
+ def write_uleb(out, n)
110
+ while n >= 0x80
111
+ out << ((n & 0x7F) | 0x80)
112
+ n >>= 7
113
+ end
114
+ out << n
115
+ end
116
+
117
+ # Decodes the RLE/bit-packed hybrid from +data+ between +pos+ and +limit+,
118
+ # returning exactly +count+ values (missing values are an error).
119
+ def decode_hybrid(data, pos, limit, width, count)
120
+ out = []
121
+ value_bytes = (width + 7) / 8
122
+ while out.size < count
123
+ raise DecodeError, "RLE data exhausted (#{out.size}/#{count} values)" if pos >= limit
124
+ header, pos = read_uleb(data, pos)
125
+ if header & 1 == 1
126
+ groups = header >> 1
127
+ n = groups * 8
128
+ vals = unpack_bits(data, pos, n, width)
129
+ pos += groups * width
130
+ remaining = count - out.size
131
+ vals = vals.first(remaining) if vals.size > remaining
132
+ out.concat(vals)
133
+ else
134
+ run = header >> 1
135
+ v = 0
136
+ value_bytes.times { |k| v |= data.getbyte(pos + k).to_i << (8 * k) }
137
+ pos += value_bytes
138
+ remaining = count - out.size
139
+ run = remaining if run > remaining
140
+ out.fill(v, out.size, run)
141
+ end
142
+ end
143
+ out
144
+ end
145
+
146
+ # Encodes +values+ with the RLE/bit-packed hybrid. Repeated runs of 8+ equal values
147
+ # become RLE runs, everything else goes into bit-packed groups of 8.
148
+ def encode_hybrid(values, width)
149
+ out = String.new(encoding: Encoding::BINARY)
150
+ n = values.size
151
+ value_bytes = (width + 7) / 8
152
+ return out if n.zero?
153
+ min, max = values.minmax
154
+ if min == max
155
+ # A single run, the common case for levels of columns without nulls
156
+ write_uleb(out, n << 1)
157
+ value_bytes.times { |k| out << ((min >> (8 * k)) & 0xFF) }
158
+ return out
159
+ end
160
+ literal_start = nil
161
+ i = 0
162
+ while i < n
163
+ v = values[i]
164
+ # Measure how long the run of v starting at i is
165
+ j = i + 1
166
+ j += 1 while j < n && values[j] == v
167
+ run = j - i
168
+ if run >= 8 && (literal_start.nil? || (i - literal_start) % 8 == 0)
169
+ flush_literals(out, values, literal_start, i, width) if literal_start
170
+ literal_start = nil
171
+ write_uleb(out, run << 1)
172
+ value_bytes.times { |k| out << ((v >> (8 * k)) & 0xFF) }
173
+ i = j
174
+ else
175
+ literal_start ||= i
176
+ # Consume values so that the literal count stays aligned to groups of 8 where possible
177
+ i += 1
178
+ end
179
+ end
180
+ flush_literals(out, values, literal_start, n, width) if literal_start
181
+ out
182
+ end
183
+
184
+ def flush_literals(out, values, from, to, width)
185
+ count = to - from
186
+ groups = (count + 7) / 8
187
+ write_uleb(out, (groups << 1) | 1)
188
+ out << pack_bits(values[from, count], width)
189
+ end
190
+
191
+ def bit_width(max_value)
192
+ max_value.to_i.bit_length
193
+ end
194
+
195
+ # Legacy BIT_PACKED level encoding (deprecated): MSB-first bit order, no header.
196
+ def decode_legacy_bit_packed(data, pos, width, count)
197
+ nbytes = (count * width + 7) / 8
198
+ bits = data.byteslice(pos, nbytes).unpack1("B*")
199
+ Array.new(count) { |i| bits[i * width, width].to_i(2) }
200
+ end
201
+ end
202
+ end
203
+ end