herringbone 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +7 -0
- data/LICENSE +21 -0
- data/README.md +209 -0
- data/bin/herringbone +54 -0
- data/lib/herringbone/active_record.rb +131 -0
- data/lib/herringbone/byte_values.rb +144 -0
- data/lib/herringbone/codecs/lz4.rb +305 -0
- data/lib/herringbone/codecs/snappy.rb +375 -0
- data/lib/herringbone/compression.rb +75 -0
- data/lib/herringbone/encodings/delta.rb +153 -0
- data/lib/herringbone/encodings/plain.rb +87 -0
- data/lib/herringbone/encodings/rle.rb +203 -0
- data/lib/herringbone/format.rb +309 -0
- data/lib/herringbone/io_buffer_support.rb +25 -0
- data/lib/herringbone/reader.rb +469 -0
- data/lib/herringbone/schema.rb +551 -0
- data/lib/herringbone/thrift.rb +334 -0
- data/lib/herringbone/types.rb +454 -0
- data/lib/herringbone/version.rb +5 -0
- data/lib/herringbone/writer.rb +634 -0
- data/lib/herringbone.rb +45 -0
- metadata +104 -0
|
@@ -0,0 +1,75 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require "zlib"
|
|
4
|
+
require "stringio"
|
|
5
|
+
require "zstd-ruby"
|
|
6
|
+
require "brotli"
|
|
7
|
+
|
|
8
|
+
module Herringbone
|
|
9
|
+
# Dispatches page (de)compression by Parquet codec id. Snappy and LZ4 are pure Ruby,
|
|
10
|
+
# GZIP uses zlib, ZSTD and Brotli use the zstd-ruby and brotli gems.
|
|
11
|
+
module Compression
|
|
12
|
+
module_function
|
|
13
|
+
|
|
14
|
+
CODECS_BY_NAME = {
|
|
15
|
+
none: Format::Codec::UNCOMPRESSED, uncompressed: Format::Codec::UNCOMPRESSED,
|
|
16
|
+
snappy: Format::Codec::SNAPPY, gzip: Format::Codec::GZIP, brotli: Format::Codec::BROTLI,
|
|
17
|
+
lz4: Format::Codec::LZ4_RAW, lz4_raw: Format::Codec::LZ4_RAW, lz4_hadoop: Format::Codec::LZ4,
|
|
18
|
+
zstd: Format::Codec::ZSTD
|
|
19
|
+
}.freeze
|
|
20
|
+
|
|
21
|
+
def codec_id(name)
|
|
22
|
+
return name if name.is_a?(Integer)
|
|
23
|
+
CODECS_BY_NAME.fetch(name.to_s.downcase.to_sym) { raise ArgumentError, "Unknown compression codec #{name.inspect}" }
|
|
24
|
+
end
|
|
25
|
+
|
|
26
|
+
def decompress(codec, data, uncompressed_size)
|
|
27
|
+
return "".b if uncompressed_size.zero? && data.empty?
|
|
28
|
+
out = case codec
|
|
29
|
+
when Format::Codec::UNCOMPRESSED then data
|
|
30
|
+
when Format::Codec::SNAPPY then Codecs::Snappy.decompress(data)
|
|
31
|
+
when Format::Codec::GZIP then gunzip(data)
|
|
32
|
+
when Format::Codec::LZ4_RAW then Codecs::LZ4.decompress_block(data, uncompressed_size)
|
|
33
|
+
when Format::Codec::LZ4 then Codecs::LZ4.decompress_hadoop(data, uncompressed_size)
|
|
34
|
+
when Format::Codec::ZSTD then ::Zstd.decompress(data)
|
|
35
|
+
when Format::Codec::BROTLI then ::Brotli.inflate(data)
|
|
36
|
+
else
|
|
37
|
+
raise UnsupportedError, "Unsupported compression codec #{Format::Codec::NAMES[codec] || codec}"
|
|
38
|
+
end
|
|
39
|
+
out = out.b
|
|
40
|
+
if out.bytesize != uncompressed_size
|
|
41
|
+
raise DecodeError, "Decompressed #{out.bytesize} bytes, expected #{uncompressed_size}"
|
|
42
|
+
end
|
|
43
|
+
out
|
|
44
|
+
end
|
|
45
|
+
|
|
46
|
+
def compress(codec, data)
|
|
47
|
+
case codec
|
|
48
|
+
when Format::Codec::UNCOMPRESSED then data
|
|
49
|
+
when Format::Codec::SNAPPY then Codecs::Snappy.compress(data)
|
|
50
|
+
when Format::Codec::GZIP then Zlib.gzip(data)
|
|
51
|
+
when Format::Codec::LZ4_RAW then Codecs::LZ4.compress_block(data)
|
|
52
|
+
when Format::Codec::LZ4 then Codecs::LZ4.compress_hadoop(data)
|
|
53
|
+
when Format::Codec::ZSTD then ::Zstd.compress(data)
|
|
54
|
+
when Format::Codec::BROTLI then ::Brotli.deflate(data)
|
|
55
|
+
else
|
|
56
|
+
raise UnsupportedError, "Unsupported compression codec #{Format::Codec::NAMES[codec] || codec}"
|
|
57
|
+
end.b
|
|
58
|
+
end
|
|
59
|
+
|
|
60
|
+
# Handles files whose gzip data consists of several concatenated members
|
|
61
|
+
def gunzip(data)
|
|
62
|
+
io = StringIO.new(data)
|
|
63
|
+
out = String.new(encoding: Encoding::BINARY)
|
|
64
|
+
loop do
|
|
65
|
+
gz = Zlib::GzipReader.new(io)
|
|
66
|
+
out << gz.read
|
|
67
|
+
unused = gz.unused
|
|
68
|
+
gz.finish
|
|
69
|
+
break if unused.nil? || unused.empty?
|
|
70
|
+
io.pos -= unused.bytesize
|
|
71
|
+
end
|
|
72
|
+
out
|
|
73
|
+
end
|
|
74
|
+
end
|
|
75
|
+
end
|
|
@@ -0,0 +1,153 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Herringbone
|
|
4
|
+
module Encodings
|
|
5
|
+
# DELTA_BINARY_PACKED, DELTA_LENGTH_BYTE_ARRAY and DELTA_BYTE_ARRAY
|
|
6
|
+
module Delta
|
|
7
|
+
module_function
|
|
8
|
+
|
|
9
|
+
BLOCK_SIZE = 128
|
|
10
|
+
MINIBLOCKS = 4
|
|
11
|
+
|
|
12
|
+
def zigzag_decode(n) = (n >> 1) ^ -(n & 1)
|
|
13
|
+
def zigzag_encode(n) = n.negative? ? ((-n) << 1) - 1 : n << 1
|
|
14
|
+
|
|
15
|
+
# Wraps an integer into the signed range of +bits+ bits
|
|
16
|
+
def wrap(v, bits)
|
|
17
|
+
half = 1 << (bits - 1)
|
|
18
|
+
((v + half) & ((1 << bits) - 1)) - half
|
|
19
|
+
end
|
|
20
|
+
|
|
21
|
+
# Decodes DELTA_BINARY_PACKED integers. +bits+ is 32 or 64 (for wraparound).
|
|
22
|
+
# Returns [values, new_pos]. If +count+ is nil, the total from the header is used.
|
|
23
|
+
def decode_binary_packed(data, pos, bits = 64, count = nil)
|
|
24
|
+
block_size, pos = RLE.read_uleb(data, pos)
|
|
25
|
+
miniblocks, pos = RLE.read_uleb(data, pos)
|
|
26
|
+
total, pos = RLE.read_uleb(data, pos)
|
|
27
|
+
first, pos = RLE.read_uleb(data, pos)
|
|
28
|
+
raise DecodeError, "Invalid DELTA_BINARY_PACKED header" if miniblocks.zero? || block_size % miniblocks != 0
|
|
29
|
+
per_mini = block_size / miniblocks
|
|
30
|
+
raise DecodeError, "Invalid miniblock size #{per_mini}" if per_mini % 8 != 0
|
|
31
|
+
total = count if count && count < total
|
|
32
|
+
values = []
|
|
33
|
+
return [values, pos] if total.zero?
|
|
34
|
+
last = zigzag_decode(first)
|
|
35
|
+
values << last
|
|
36
|
+
half = 1 << (bits - 1)
|
|
37
|
+
mask = (1 << bits) - 1
|
|
38
|
+
while values.size < total
|
|
39
|
+
min_delta, pos = RLE.read_uleb(data, pos)
|
|
40
|
+
min_delta = zigzag_decode(min_delta)
|
|
41
|
+
widths = data.byteslice(pos, miniblocks).unpack("C*")
|
|
42
|
+
pos += miniblocks
|
|
43
|
+
widths.each do |w|
|
|
44
|
+
break if values.size >= total
|
|
45
|
+
raise DecodeError, "Invalid delta bit width #{w}" if w > bits
|
|
46
|
+
deltas = RLE.unpack_bits(data, pos, per_mini, w)
|
|
47
|
+
pos += per_mini * w / 8
|
|
48
|
+
take = total - values.size
|
|
49
|
+
deltas = deltas.first(take) if take < per_mini
|
|
50
|
+
deltas.each do |d|
|
|
51
|
+
last = ((last + min_delta + d + half) & mask) - half
|
|
52
|
+
values << last
|
|
53
|
+
end
|
|
54
|
+
end
|
|
55
|
+
end
|
|
56
|
+
[values, pos]
|
|
57
|
+
end
|
|
58
|
+
|
|
59
|
+
def encode_binary_packed(values, bits = 64)
|
|
60
|
+
out = String.new(encoding: Encoding::BINARY)
|
|
61
|
+
per_mini = BLOCK_SIZE / MINIBLOCKS
|
|
62
|
+
RLE.write_uleb(out, BLOCK_SIZE)
|
|
63
|
+
RLE.write_uleb(out, MINIBLOCKS)
|
|
64
|
+
RLE.write_uleb(out, values.size)
|
|
65
|
+
RLE.write_uleb(out, zigzag_encode(values.empty? ? 0 : values[0]))
|
|
66
|
+
return out if values.size <= 1
|
|
67
|
+
|
|
68
|
+
mask = (1 << bits) - 1
|
|
69
|
+
deltas = Array.new(values.size - 1) { |i| wrap(values[i + 1] - values[i], bits) }
|
|
70
|
+
deltas.each_slice(BLOCK_SIZE) do |block|
|
|
71
|
+
min = block.min
|
|
72
|
+
RLE.write_uleb(out, zigzag_encode(min))
|
|
73
|
+
adjusted = block.map { |d| (d - min) & mask }
|
|
74
|
+
minis = adjusted.each_slice(per_mini).to_a
|
|
75
|
+
widths = Array.new(MINIBLOCKS) { |m| minis[m] ? minis[m].max.bit_length : 0 }
|
|
76
|
+
out << widths.pack("C*")
|
|
77
|
+
minis.each_with_index do |mini, m|
|
|
78
|
+
mini += Array.new(per_mini - mini.size, 0) if mini.size < per_mini
|
|
79
|
+
out << RLE.pack_bits(mini, widths[m])
|
|
80
|
+
end
|
|
81
|
+
end
|
|
82
|
+
out
|
|
83
|
+
end
|
|
84
|
+
|
|
85
|
+
def decode_length_byte_array(data, pos, count)
|
|
86
|
+
lengths, pos = decode_binary_packed(data, pos, 32, count)
|
|
87
|
+
raise DecodeError, "DELTA_LENGTH_BYTE_ARRAY has #{lengths.size} lengths, need #{count}" if lengths.size < count
|
|
88
|
+
out = Array.new(count)
|
|
89
|
+
size = data.bytesize
|
|
90
|
+
lengths.each_with_index do |len, i|
|
|
91
|
+
raise DecodeError, "DELTA_LENGTH_BYTE_ARRAY value overruns page" if len.negative? || pos + len > size
|
|
92
|
+
out[i] = data.byteslice(pos, len)
|
|
93
|
+
pos += len
|
|
94
|
+
end
|
|
95
|
+
[out, pos]
|
|
96
|
+
end
|
|
97
|
+
|
|
98
|
+
def encode_length_byte_array(values)
|
|
99
|
+
encode_binary_packed(values.map(&:bytesize), 32) << values.pack("a*" * values.size)
|
|
100
|
+
end
|
|
101
|
+
|
|
102
|
+
def decode_byte_array(data, pos, count)
|
|
103
|
+
prefixes, pos = decode_binary_packed(data, pos, 32, count)
|
|
104
|
+
suffixes, pos = decode_length_byte_array(data, pos, count)
|
|
105
|
+
prev = "".b
|
|
106
|
+
out = Array.new(count) do |i|
|
|
107
|
+
prefix = prefixes[i]
|
|
108
|
+
raise DecodeError, "DELTA_BYTE_ARRAY prefix longer than previous value" if prefix > prev.bytesize
|
|
109
|
+
prev = prefix.zero? ? suffixes[i] : prev.byteslice(0, prefix) + suffixes[i]
|
|
110
|
+
end
|
|
111
|
+
[out, pos]
|
|
112
|
+
end
|
|
113
|
+
|
|
114
|
+
def encode_byte_array(values)
|
|
115
|
+
prev = "".b
|
|
116
|
+
prefixes = []
|
|
117
|
+
suffixes = []
|
|
118
|
+
values.each do |v|
|
|
119
|
+
max = [prev.bytesize, v.bytesize].min
|
|
120
|
+
n = 0
|
|
121
|
+
n += 1 while n < max && prev.getbyte(n) == v.getbyte(n)
|
|
122
|
+
prefixes << n
|
|
123
|
+
suffixes << v.byteslice(n, v.bytesize - n)
|
|
124
|
+
prev = v
|
|
125
|
+
end
|
|
126
|
+
encode_binary_packed(prefixes, 32) << encode_length_byte_array(suffixes)
|
|
127
|
+
end
|
|
128
|
+
end
|
|
129
|
+
|
|
130
|
+
# BYTE_STREAM_SPLIT: byte k of every value is stored in stream k.
|
|
131
|
+
module ByteStreamSplit
|
|
132
|
+
module_function
|
|
133
|
+
|
|
134
|
+
# Returns the value bytes re-interleaved into PLAIN layout
|
|
135
|
+
def decode(data, pos, count, width)
|
|
136
|
+
nbytes = count * width
|
|
137
|
+
raise DecodeError, "Truncated BYTE_STREAM_SPLIT data" if pos + nbytes > data.bytesize
|
|
138
|
+
streams = Array.new(width) { |k| data.byteslice(pos + k * count, count).unpack("C*") }
|
|
139
|
+
[streams[0].zip(*streams[1..]).flatten.pack("C*"), pos + nbytes]
|
|
140
|
+
end
|
|
141
|
+
|
|
142
|
+
def encode(plain, width)
|
|
143
|
+
count = plain.bytesize / width
|
|
144
|
+
bytes = plain.unpack("C*")
|
|
145
|
+
out = String.new(capacity: plain.bytesize, encoding: Encoding::BINARY)
|
|
146
|
+
width.times do |k|
|
|
147
|
+
out << Array.new(count) { |i| bytes[i * width + k] }.pack("C*")
|
|
148
|
+
end
|
|
149
|
+
out
|
|
150
|
+
end
|
|
151
|
+
end
|
|
152
|
+
end
|
|
153
|
+
end
|
|
@@ -0,0 +1,87 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Herringbone
|
|
4
|
+
module Encodings
|
|
5
|
+
# PLAIN encoding for all physical types. Values are returned in their physical
|
|
6
|
+
# Ruby form (Integer, Float, true/false, binary String); logical conversion happens elsewhere.
|
|
7
|
+
module Plain
|
|
8
|
+
module_function
|
|
9
|
+
|
|
10
|
+
FORMATS = {
|
|
11
|
+
Format::Type::INT32 => ["l<", 4],
|
|
12
|
+
Format::Type::INT64 => ["q<", 8],
|
|
13
|
+
Format::Type::FLOAT => ["e", 4],
|
|
14
|
+
Format::Type::DOUBLE => ["E", 8]
|
|
15
|
+
}.freeze
|
|
16
|
+
|
|
17
|
+
# Decodes +count+ values of +type+ from +data+ starting at +pos+.
|
|
18
|
+
# Returns [values, new_pos].
|
|
19
|
+
def decode(data, pos, count, type, type_length = nil)
|
|
20
|
+
case type
|
|
21
|
+
when Format::Type::BOOLEAN
|
|
22
|
+
nbytes = (count + 7) / 8
|
|
23
|
+
bits = data.byteslice(pos, nbytes).unpack1("b*")
|
|
24
|
+
raise DecodeError, "Truncated BOOLEAN data" if bits.bytesize < count
|
|
25
|
+
[Array.new(count) { |i| bits.getbyte(i) == 49 }, pos + nbytes]
|
|
26
|
+
when Format::Type::INT32, Format::Type::INT64, Format::Type::FLOAT, Format::Type::DOUBLE
|
|
27
|
+
fmt, width = FORMATS[type]
|
|
28
|
+
nbytes = count * width
|
|
29
|
+
raise DecodeError, "Truncated PLAIN data" if pos + nbytes > data.bytesize
|
|
30
|
+
[data.byteslice(pos, nbytes).unpack("#{fmt}#{count}"), pos + nbytes]
|
|
31
|
+
when Format::Type::INT96
|
|
32
|
+
nbytes = count * 12
|
|
33
|
+
raise DecodeError, "Truncated INT96 data" if pos + nbytes > data.bytesize
|
|
34
|
+
vals = data.byteslice(pos, nbytes).unpack("Q<L<" * count).each_slice(2).map { |nanos, day| [nanos, day] }
|
|
35
|
+
[vals, pos + nbytes]
|
|
36
|
+
when Format::Type::BYTE_ARRAY
|
|
37
|
+
decode_byte_arrays(data, pos, count)
|
|
38
|
+
when Format::Type::FIXED_LEN_BYTE_ARRAY
|
|
39
|
+
nbytes = count * type_length
|
|
40
|
+
raise DecodeError, "Truncated FIXED_LEN_BYTE_ARRAY data" if pos + nbytes > data.bytesize
|
|
41
|
+
[Array.new(count) { |i| data.byteslice(pos + i * type_length, type_length) }, pos + nbytes]
|
|
42
|
+
else
|
|
43
|
+
raise DecodeError, "Unknown physical type #{type}"
|
|
44
|
+
end
|
|
45
|
+
end
|
|
46
|
+
|
|
47
|
+
def decode_byte_arrays(data, pos, count)
|
|
48
|
+
out = Array.new(count)
|
|
49
|
+
size = data.bytesize
|
|
50
|
+
i = 0
|
|
51
|
+
while i < count
|
|
52
|
+
raise DecodeError, "Truncated BYTE_ARRAY data" if pos + 4 > size
|
|
53
|
+
len = data.getbyte(pos) | (data.getbyte(pos + 1) << 8) | (data.getbyte(pos + 2) << 16) | (data.getbyte(pos + 3) << 24)
|
|
54
|
+
pos += 4
|
|
55
|
+
raise DecodeError, "BYTE_ARRAY value overruns page" if pos + len > size
|
|
56
|
+
out[i] = data.byteslice(pos, len)
|
|
57
|
+
pos += len
|
|
58
|
+
i += 1
|
|
59
|
+
end
|
|
60
|
+
[out, pos]
|
|
61
|
+
end
|
|
62
|
+
|
|
63
|
+
def encode(values, type, type_length = nil)
|
|
64
|
+
case type
|
|
65
|
+
when Format::Type::BOOLEAN
|
|
66
|
+
[values.map { |v| v ? "1" : "0" }.join].pack("b*")
|
|
67
|
+
when Format::Type::INT32, Format::Type::INT64, Format::Type::FLOAT, Format::Type::DOUBLE
|
|
68
|
+
values.pack("#{FORMATS[type][0]}*")
|
|
69
|
+
when Format::Type::INT96
|
|
70
|
+
values.flat_map { |nanos, day| [nanos, day] }.pack("Q<L<" * values.size)
|
|
71
|
+
when Format::Type::BYTE_ARRAY
|
|
72
|
+
# pack("a*") copies raw bytes whatever the string's encoding, without an intermediate copy
|
|
73
|
+
args = []
|
|
74
|
+
values.each { |v| args << v.bytesize << v }
|
|
75
|
+
args.pack("Va*" * values.size)
|
|
76
|
+
when Format::Type::FIXED_LEN_BYTE_ARRAY
|
|
77
|
+
values.each do |v|
|
|
78
|
+
raise EncodeError, "Expected #{type_length} bytes, got #{v.bytesize}" if v.bytesize != type_length
|
|
79
|
+
end
|
|
80
|
+
values.pack("a*" * values.size)
|
|
81
|
+
else
|
|
82
|
+
raise EncodeError, "Unknown physical type #{type}"
|
|
83
|
+
end
|
|
84
|
+
end
|
|
85
|
+
end
|
|
86
|
+
end
|
|
87
|
+
end
|
|
@@ -0,0 +1,203 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Herringbone
|
|
4
|
+
module Encodings
|
|
5
|
+
# Bit packing (LSB-first, as used by Parquet) and the RLE / bit-packed hybrid encoding
|
|
6
|
+
# used for repetition/definition levels, dictionary indices and RLE booleans.
|
|
7
|
+
module RLE
|
|
8
|
+
module_function
|
|
9
|
+
|
|
10
|
+
# Unpacks +count+ values of +width+ bits each, starting at byte +offset+ of +data+.
|
|
11
|
+
# Missing trailing bytes are treated as zeroes.
|
|
12
|
+
def unpack_bits(data, offset, count, width)
|
|
13
|
+
return Array.new(count, 0) if width.zero?
|
|
14
|
+
nbytes = (count * width + 7) / 8
|
|
15
|
+
chunk = data.byteslice(offset, nbytes) || "".b
|
|
16
|
+
if width == 8
|
|
17
|
+
vals = chunk.unpack("C*")
|
|
18
|
+
vals.fill(0, vals.size, count - vals.size) if vals.size < count
|
|
19
|
+
return vals
|
|
20
|
+
end
|
|
21
|
+
return unpack_wide(chunk, count, width) if width > 32
|
|
22
|
+
|
|
23
|
+
# Pad to a whole number of 32-bit words plus one spare word, so reads never go out of range
|
|
24
|
+
pad = (-chunk.bytesize % 4) + 4
|
|
25
|
+
words = (pad.zero? ? chunk : chunk + ("\0" * pad)).unpack("V*")
|
|
26
|
+
mask = (1 << width) - 1
|
|
27
|
+
out = Array.new(count)
|
|
28
|
+
bitpos = 0
|
|
29
|
+
i = 0
|
|
30
|
+
while i < count
|
|
31
|
+
wi = bitpos >> 5
|
|
32
|
+
off = bitpos & 31
|
|
33
|
+
v = words[wi] >> off
|
|
34
|
+
spill = off + width - 32
|
|
35
|
+
v |= (words[wi + 1] & ((1 << spill) - 1)) << (32 - off) if spill.positive?
|
|
36
|
+
out[i] = v & mask
|
|
37
|
+
bitpos += width
|
|
38
|
+
i += 1
|
|
39
|
+
end
|
|
40
|
+
out
|
|
41
|
+
end
|
|
42
|
+
|
|
43
|
+
# Slow path for widths above 32 bits (used by DELTA_BINARY_PACKED with 64-bit values)
|
|
44
|
+
def unpack_wide(chunk, count, width)
|
|
45
|
+
mask = (1 << width) - 1
|
|
46
|
+
out = Array.new(count)
|
|
47
|
+
group_bytes = width # 8 values * width bits = width bytes
|
|
48
|
+
ngroups = (count + 7) / 8
|
|
49
|
+
i = 0
|
|
50
|
+
ngroups.times do |g|
|
|
51
|
+
bytes = chunk.byteslice(g * group_bytes, group_bytes) || "".b
|
|
52
|
+
bytes += "\0" * (group_bytes - bytes.bytesize) if bytes.bytesize < group_bytes
|
|
53
|
+
big = bytes.reverse.unpack1("H*").to_i(16)
|
|
54
|
+
8.times do |j|
|
|
55
|
+
break if i >= count
|
|
56
|
+
out[i] = (big >> (j * width)) & mask
|
|
57
|
+
i += 1
|
|
58
|
+
end
|
|
59
|
+
end
|
|
60
|
+
out
|
|
61
|
+
end
|
|
62
|
+
|
|
63
|
+
# Packs +values+ with +width+ bits each; the value count is padded up to a multiple of 8.
|
|
64
|
+
def pack_bits(values, width)
|
|
65
|
+
return "".b if width.zero? || values.empty?
|
|
66
|
+
n = values.size
|
|
67
|
+
padded = (n + 7) & ~7
|
|
68
|
+
if width == 8
|
|
69
|
+
s = values.pack("C*")
|
|
70
|
+
s << ("\0" * (padded - n)) if padded > n
|
|
71
|
+
return s
|
|
72
|
+
end
|
|
73
|
+
out = String.new(capacity: padded * width / 8, encoding: Encoding::BINARY)
|
|
74
|
+
acc = 0
|
|
75
|
+
nbits = 0
|
|
76
|
+
i = 0
|
|
77
|
+
while i < padded
|
|
78
|
+
acc |= (values[i] || 0) << nbits
|
|
79
|
+
nbits += width
|
|
80
|
+
while nbits >= 32
|
|
81
|
+
out << [acc & 0xFFFFFFFF].pack("V")
|
|
82
|
+
acc >>= 32
|
|
83
|
+
nbits -= 32
|
|
84
|
+
end
|
|
85
|
+
i += 1
|
|
86
|
+
end
|
|
87
|
+
# Remaining bits: padded*width is a multiple of 8, so nbits is a whole number of bytes
|
|
88
|
+
while nbits.positive?
|
|
89
|
+
out << (acc & 0xFF)
|
|
90
|
+
acc >>= 8
|
|
91
|
+
nbits -= 8
|
|
92
|
+
end
|
|
93
|
+
out
|
|
94
|
+
end
|
|
95
|
+
|
|
96
|
+
def read_uleb(data, pos)
|
|
97
|
+
result = 0
|
|
98
|
+
shift = 0
|
|
99
|
+
while true
|
|
100
|
+
b = data.getbyte(pos)
|
|
101
|
+
raise DecodeError, "Truncated varint" unless b
|
|
102
|
+
pos += 1
|
|
103
|
+
result |= (b & 0x7F) << shift
|
|
104
|
+
return [result, pos] if b < 0x80
|
|
105
|
+
shift += 7
|
|
106
|
+
end
|
|
107
|
+
end
|
|
108
|
+
|
|
109
|
+
def write_uleb(out, n)
|
|
110
|
+
while n >= 0x80
|
|
111
|
+
out << ((n & 0x7F) | 0x80)
|
|
112
|
+
n >>= 7
|
|
113
|
+
end
|
|
114
|
+
out << n
|
|
115
|
+
end
|
|
116
|
+
|
|
117
|
+
# Decodes the RLE/bit-packed hybrid from +data+ between +pos+ and +limit+,
|
|
118
|
+
# returning exactly +count+ values (missing values are an error).
|
|
119
|
+
def decode_hybrid(data, pos, limit, width, count)
|
|
120
|
+
out = []
|
|
121
|
+
value_bytes = (width + 7) / 8
|
|
122
|
+
while out.size < count
|
|
123
|
+
raise DecodeError, "RLE data exhausted (#{out.size}/#{count} values)" if pos >= limit
|
|
124
|
+
header, pos = read_uleb(data, pos)
|
|
125
|
+
if header & 1 == 1
|
|
126
|
+
groups = header >> 1
|
|
127
|
+
n = groups * 8
|
|
128
|
+
vals = unpack_bits(data, pos, n, width)
|
|
129
|
+
pos += groups * width
|
|
130
|
+
remaining = count - out.size
|
|
131
|
+
vals = vals.first(remaining) if vals.size > remaining
|
|
132
|
+
out.concat(vals)
|
|
133
|
+
else
|
|
134
|
+
run = header >> 1
|
|
135
|
+
v = 0
|
|
136
|
+
value_bytes.times { |k| v |= data.getbyte(pos + k).to_i << (8 * k) }
|
|
137
|
+
pos += value_bytes
|
|
138
|
+
remaining = count - out.size
|
|
139
|
+
run = remaining if run > remaining
|
|
140
|
+
out.fill(v, out.size, run)
|
|
141
|
+
end
|
|
142
|
+
end
|
|
143
|
+
out
|
|
144
|
+
end
|
|
145
|
+
|
|
146
|
+
# Encodes +values+ with the RLE/bit-packed hybrid. Repeated runs of 8+ equal values
|
|
147
|
+
# become RLE runs, everything else goes into bit-packed groups of 8.
|
|
148
|
+
def encode_hybrid(values, width)
|
|
149
|
+
out = String.new(encoding: Encoding::BINARY)
|
|
150
|
+
n = values.size
|
|
151
|
+
value_bytes = (width + 7) / 8
|
|
152
|
+
return out if n.zero?
|
|
153
|
+
min, max = values.minmax
|
|
154
|
+
if min == max
|
|
155
|
+
# A single run, the common case for levels of columns without nulls
|
|
156
|
+
write_uleb(out, n << 1)
|
|
157
|
+
value_bytes.times { |k| out << ((min >> (8 * k)) & 0xFF) }
|
|
158
|
+
return out
|
|
159
|
+
end
|
|
160
|
+
literal_start = nil
|
|
161
|
+
i = 0
|
|
162
|
+
while i < n
|
|
163
|
+
v = values[i]
|
|
164
|
+
# Measure how long the run of v starting at i is
|
|
165
|
+
j = i + 1
|
|
166
|
+
j += 1 while j < n && values[j] == v
|
|
167
|
+
run = j - i
|
|
168
|
+
if run >= 8 && (literal_start.nil? || (i - literal_start) % 8 == 0)
|
|
169
|
+
flush_literals(out, values, literal_start, i, width) if literal_start
|
|
170
|
+
literal_start = nil
|
|
171
|
+
write_uleb(out, run << 1)
|
|
172
|
+
value_bytes.times { |k| out << ((v >> (8 * k)) & 0xFF) }
|
|
173
|
+
i = j
|
|
174
|
+
else
|
|
175
|
+
literal_start ||= i
|
|
176
|
+
# Consume values so that the literal count stays aligned to groups of 8 where possible
|
|
177
|
+
i += 1
|
|
178
|
+
end
|
|
179
|
+
end
|
|
180
|
+
flush_literals(out, values, literal_start, n, width) if literal_start
|
|
181
|
+
out
|
|
182
|
+
end
|
|
183
|
+
|
|
184
|
+
def flush_literals(out, values, from, to, width)
|
|
185
|
+
count = to - from
|
|
186
|
+
groups = (count + 7) / 8
|
|
187
|
+
write_uleb(out, (groups << 1) | 1)
|
|
188
|
+
out << pack_bits(values[from, count], width)
|
|
189
|
+
end
|
|
190
|
+
|
|
191
|
+
def bit_width(max_value)
|
|
192
|
+
max_value.to_i.bit_length
|
|
193
|
+
end
|
|
194
|
+
|
|
195
|
+
# Legacy BIT_PACKED level encoding (deprecated): MSB-first bit order, no header.
|
|
196
|
+
def decode_legacy_bit_packed(data, pos, width, count)
|
|
197
|
+
nbytes = (count * width + 7) / 8
|
|
198
|
+
bits = data.byteslice(pos, nbytes).unpack1("B*")
|
|
199
|
+
Array.new(count) { |i| bits[i * width, width].to_i(2) }
|
|
200
|
+
end
|
|
201
|
+
end
|
|
202
|
+
end
|
|
203
|
+
end
|