herringbone 0.1.0 → 0.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/README.md +146 -74
- data/bin/herringbone +38 -26
- data/lib/herringbone/active_record.rb +10 -13
- data/lib/herringbone/bloom_filter.rb +270 -0
- data/lib/herringbone/compression.rb +73 -16
- data/lib/herringbone/encodings/delta.rb +7 -7
- data/lib/herringbone/encodings/plain.rb +7 -7
- data/lib/herringbone/encodings/rle.rb +2 -2
- data/lib/herringbone/format.rb +53 -0
- data/lib/herringbone/inspector.rb +1388 -0
- data/lib/herringbone/reader/column_chunk_reader.rb +296 -0
- data/lib/herringbone/reader/column_cursor.rb +181 -0
- data/lib/herringbone/reader/page_stream.rb +339 -0
- data/lib/herringbone/reader/scan.rb +256 -0
- data/lib/herringbone/reader.rb +294 -272
- data/lib/herringbone/schema.rb +21 -76
- data/lib/herringbone/types.rb +4 -4
- data/lib/herringbone/version.rb +1 -1
- data/lib/herringbone/visualizer.rb +1096 -0
- data/lib/herringbone/writer.rb +258 -90
- data/lib/herringbone/xxhash.rb +319 -0
- data/lib/herringbone.rb +36 -9
- metadata +12 -32
|
@@ -0,0 +1,270 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Herringbone
|
|
4
|
+
# A Parquet Split Block Bloom Filter (parquet-format BloomFilter.md).
|
|
5
|
+
#
|
|
6
|
+
# The bitset is made of 32-byte blocks of eight 32-bit words. A value is hashed with XXH64 over
|
|
7
|
+
# the PLAIN encoding of its physical value (without the length prefix for BYTE_ARRAY); the upper
|
|
8
|
+
# 32 bits of the hash pick a block, and the lower 32 bits set one bit in each of its words.
|
|
9
|
+
#
|
|
10
|
+
# A filter answers "definitely not in the column chunk" or "maybe": +might_contain?+ never
|
|
11
|
+
# returns false for a value that was inserted. Values are given as Ruby values and converted
|
|
12
|
+
# like the writer converts them (the column's encoder), so a Date, Time, BigDecimal or UUID
|
|
13
|
+
# String hashes the same bytes as the stored value. Nulls are never in a bloom filter.
|
|
14
|
+
# The writer builds them (bloom_filters: option) and reads with where: consult them.
|
|
15
|
+
class BloomFilter
|
|
16
|
+
SALT = [0x47b6137b, 0x44974d91, 0x8824ad5b, 0xa2b7289d, 0x705495c7, 0x2df1424b, 0x9efc4947, 0x5c6bfb31].freeze
|
|
17
|
+
S0, S1, S2, S3, S4, S5, S6, S7 = SALT
|
|
18
|
+
# Low 16 bits of the salts: (lo * salt) mod 2**32 is computed as
|
|
19
|
+
# (lo & 0xFFFF) * salt + ((lo >> 16) * (salt & 0xFFFF) << 16), which stays a Fixnum
|
|
20
|
+
L0, L1, L2, L3, L4, L5, L6, L7 = SALT.map { |s| s & 0xFFFF }
|
|
21
|
+
BLOCK_BYTES = 32
|
|
22
|
+
MIN_BYTES = 32
|
|
23
|
+
MAX_BYTES = 128 * 1024 * 1024
|
|
24
|
+
# Default cap for filters sized by the writer (as in parquet-mr)
|
|
25
|
+
DEFAULT_MAX_BYTES = 1024 * 1024
|
|
26
|
+
DEFAULT_FPP = 0.01
|
|
27
|
+
M32 = 0xFFFF_FFFF
|
|
28
|
+
M64 = 0xFFFF_FFFF_FFFF_FFFF
|
|
29
|
+
T = Format::Type
|
|
30
|
+
# Physical types a bloom filter can be built for (the spec does not define BOOLEAN hashing)
|
|
31
|
+
TYPES = [T::INT32, T::INT64, T::INT96, T::FLOAT, T::DOUBLE, T::BYTE_ARRAY, T::FIXED_LEN_BYTE_ARRAY].freeze
|
|
32
|
+
|
|
33
|
+
# Bitset size in bytes for +ndv+ distinct values at false positive probability +fpp+, per the
|
|
34
|
+
# spec's formula (m = -8 * ndv / ln(1 - fpp ** (1/8)) bits), rounded up to a power of two
|
|
35
|
+
# and clamped to MIN_BYTES..max_bytes
|
|
36
|
+
def self.optimal_num_bytes(ndv, fpp = DEFAULT_FPP, max_bytes: DEFAULT_MAX_BYTES)
|
|
37
|
+
fpp = Float(fpp)
|
|
38
|
+
raise ArgumentError, "fpp must be between 0 and 1, got #{fpp}" unless fpp > 0 && fpp < 1
|
|
39
|
+
max_bytes = [[Integer(max_bytes), MAX_BYTES].min, MIN_BYTES].max
|
|
40
|
+
max_bytes = 1 << (max_bytes.bit_length - 1) # a power of two
|
|
41
|
+
ndv = [Integer(ndv), 1].max
|
|
42
|
+
bits = -8.0 * ndv / Math.log(1 - fpp**(1.0 / 8))
|
|
43
|
+
bytes = (bits / 8).ceil
|
|
44
|
+
return max_bytes if bytes >= max_bytes
|
|
45
|
+
return MIN_BYTES if bytes <= MIN_BYTES
|
|
46
|
+
1 << (bytes - 1).bit_length
|
|
47
|
+
end
|
|
48
|
+
|
|
49
|
+
# XXH64 of the PLAIN encoding of a physical value of +type+ (as the column encoders produce it)
|
|
50
|
+
def self.hash_physical(value, type)
|
|
51
|
+
case type
|
|
52
|
+
when T::INT32 then XXHash.xxh64_u32(value)
|
|
53
|
+
when T::INT64 then XXHash.xxh64_u64(value)
|
|
54
|
+
when T::FLOAT then XXHash.xxh64_u32([value].pack("e").unpack1("L<"))
|
|
55
|
+
when T::DOUBLE then XXHash.xxh64_u64([value].pack("E").unpack1("Q<"))
|
|
56
|
+
when T::INT96 then XXHash.xxh64(value.pack("Q<L<"))
|
|
57
|
+
when T::BYTE_ARRAY, T::FIXED_LEN_BYTE_ARRAY then XXHash.xxh64(value)
|
|
58
|
+
else raise UnsupportedError, "Bloom filters are not defined for #{T::NAMES.fetch(type, type)} values"
|
|
59
|
+
end
|
|
60
|
+
end
|
|
61
|
+
|
|
62
|
+
# Hashes of many physical values of +type+, converting them in bulk. With +distinct+, each
|
|
63
|
+
# distinct physical value is hashed once (floats are compared by their bytes, so -0.0 and 0.0,
|
|
64
|
+
# or NaNs with different payloads, stay apart as they hash differently).
|
|
65
|
+
def self.hash_physical_all(values, type, distinct: false)
|
|
66
|
+
case type
|
|
67
|
+
when T::INT32 then XXHash.xxh64_u32_all(distinct ? values.uniq : values)
|
|
68
|
+
when T::INT64 then XXHash.xxh64_u64_all(distinct ? values.uniq : values)
|
|
69
|
+
when T::FLOAT
|
|
70
|
+
words = values.pack("e*").unpack("L<*")
|
|
71
|
+
XXHash.xxh64_u32_all(distinct ? words.uniq! || words : words)
|
|
72
|
+
when T::DOUBLE
|
|
73
|
+
lanes = values.pack("E*").unpack("Q<*")
|
|
74
|
+
XXHash.xxh64_u64_all(distinct ? lanes.uniq! || lanes : lanes)
|
|
75
|
+
when T::INT96 then XXHash.xxh64_all((distinct ? values.uniq : values).map { |v| v.pack("Q<L<") })
|
|
76
|
+
when T::BYTE_ARRAY, T::FIXED_LEN_BYTE_ARRAY then XXHash.xxh64_all(distinct ? values.uniq : values)
|
|
77
|
+
else raise UnsupportedError, "Bloom filters are not defined for #{T::NAMES.fetch(type, type)} values"
|
|
78
|
+
end
|
|
79
|
+
end
|
|
80
|
+
|
|
81
|
+
# Reads a filter (header and bitset) from +buf+ at +pos+. Returns nil for algorithms, hashes
|
|
82
|
+
# or compressions this implementation does not know.
|
|
83
|
+
def self.decode(buf, pos = 0, column: nil)
|
|
84
|
+
reader = Thrift::Reader.new(buf, pos)
|
|
85
|
+
header = reader.read_struct(Format::BloomFilterHeader)
|
|
86
|
+
return nil unless supported_header?(header)
|
|
87
|
+
bitset = buf.byteslice(reader.pos, header.num_bytes)
|
|
88
|
+
raise FormatError, "Truncated bloom filter" if bitset.nil? || bitset.bytesize != header.num_bytes
|
|
89
|
+
new(bitset: bitset, column: column)
|
|
90
|
+
end
|
|
91
|
+
|
|
92
|
+
def self.supported_header?(header)
|
|
93
|
+
n = header.num_bytes
|
|
94
|
+
n.is_a?(Integer) && n >= BLOCK_BYTES && (n % BLOCK_BYTES).zero? && n <= MAX_BYTES &&
|
|
95
|
+
header.algorithm&.block && header.hash_function&.xxhash &&
|
|
96
|
+
(header.compression.nil? || header.compression.uncompressed)
|
|
97
|
+
end
|
|
98
|
+
|
|
99
|
+
attr_reader :column
|
|
100
|
+
|
|
101
|
+
# A filter of +num_bytes+ (a multiple of 32, normally a power of two), or one using an existing
|
|
102
|
+
# +bitset+ String. With a +column+ (a Schema::Column), values are Ruby values converted with
|
|
103
|
+
# the column's encoder; without one, values must be Strings and their bytes are hashed.
|
|
104
|
+
def initialize(num_bytes = nil, bitset: nil, column: nil)
|
|
105
|
+
if bitset
|
|
106
|
+
@words = bitset.unpack("V*")
|
|
107
|
+
else
|
|
108
|
+
num_bytes = Integer(num_bytes || MIN_BYTES)
|
|
109
|
+
unless num_bytes >= BLOCK_BYTES && (num_bytes % BLOCK_BYTES).zero? && num_bytes <= MAX_BYTES
|
|
110
|
+
raise ArgumentError, "Bloom filter size must be a multiple of 32 bytes up to 128MB, got #{num_bytes}"
|
|
111
|
+
end
|
|
112
|
+
@words = Array.new(num_bytes / 4, 0)
|
|
113
|
+
end
|
|
114
|
+
@num_blocks = @words.size / 8
|
|
115
|
+
raise ArgumentError, "Bloom filter bitset must be a multiple of 32 bytes" if @num_blocks.zero? || @words.size % 8 != 0
|
|
116
|
+
@column = column
|
|
117
|
+
if column && !TYPES.include?(column.type)
|
|
118
|
+
raise UnsupportedError, "Bloom filters are not supported for #{T::NAMES[column.type]} column #{column.dotted_path}"
|
|
119
|
+
end
|
|
120
|
+
end
|
|
121
|
+
|
|
122
|
+
def num_bytes = @words.size * 4
|
|
123
|
+
|
|
124
|
+
def insert(value)
|
|
125
|
+
insert_hash(hash_of(value))
|
|
126
|
+
self
|
|
127
|
+
end
|
|
128
|
+
|
|
129
|
+
def might_contain?(value)
|
|
130
|
+
might_contain_hash?(hash_of(value))
|
|
131
|
+
end
|
|
132
|
+
|
|
133
|
+
# XXH64 hash the filter uses for a Ruby +value+
|
|
134
|
+
def hash_of(value)
|
|
135
|
+
raise ArgumentError, "Nulls are not recorded in bloom filters" if value.nil?
|
|
136
|
+
if @column
|
|
137
|
+
begin
|
|
138
|
+
physical = @column.encoder.call(value)
|
|
139
|
+
rescue ArgumentError, TypeError, NoMethodError, RangeError, EncodeError => e
|
|
140
|
+
raise ArgumentError, "#{value.inspect} is not a valid value for #{@column.dotted_path}: #{e.message}"
|
|
141
|
+
end
|
|
142
|
+
self.class.hash_physical(physical, @column.type)
|
|
143
|
+
else
|
|
144
|
+
raise ArgumentError, "Without a column, bloom filter values must be Strings, got #{value.class}" unless value.is_a?(String)
|
|
145
|
+
XXHash.xxh64(value)
|
|
146
|
+
end
|
|
147
|
+
end
|
|
148
|
+
|
|
149
|
+
def insert_hash(h)
|
|
150
|
+
i = (((h >> 32) * @num_blocks) >> 32) << 3
|
|
151
|
+
x0 = h & 0xFFFF
|
|
152
|
+
x1 = (h >> 16) & 0xFFFF
|
|
153
|
+
w = @words
|
|
154
|
+
w[i] |= 1 << (((x0 * S0 + (x1 * L0 << 16)) & M32) >> 27)
|
|
155
|
+
w[i + 1] |= 1 << (((x0 * S1 + (x1 * L1 << 16)) & M32) >> 27)
|
|
156
|
+
w[i + 2] |= 1 << (((x0 * S2 + (x1 * L2 << 16)) & M32) >> 27)
|
|
157
|
+
w[i + 3] |= 1 << (((x0 * S3 + (x1 * L3 << 16)) & M32) >> 27)
|
|
158
|
+
w[i + 4] |= 1 << (((x0 * S4 + (x1 * L4 << 16)) & M32) >> 27)
|
|
159
|
+
w[i + 5] |= 1 << (((x0 * S5 + (x1 * L5 << 16)) & M32) >> 27)
|
|
160
|
+
w[i + 6] |= 1 << (((x0 * S6 + (x1 * L6 << 16)) & M32) >> 27)
|
|
161
|
+
w[i + 7] |= 1 << (((x0 * S7 + (x1 * L7 << 16)) & M32) >> 27)
|
|
162
|
+
self
|
|
163
|
+
end
|
|
164
|
+
|
|
165
|
+
BITS = Array.new(32) { |i| 1 << i }.freeze
|
|
166
|
+
|
|
167
|
+
# Inserts many hashes at once: faster than insert_hash, as the hashes (mostly Bignums) are
|
|
168
|
+
# split into 32-bit halves in bulk and the rest is Fixnum arithmetic
|
|
169
|
+
def insert_hashes(hashes)
|
|
170
|
+
halves = hashes.pack("Q<*").unpack("V*")
|
|
171
|
+
w = @words
|
|
172
|
+
num_blocks = @num_blocks
|
|
173
|
+
bit = BITS
|
|
174
|
+
j = 0
|
|
175
|
+
n = halves.size
|
|
176
|
+
while j < n
|
|
177
|
+
lo = halves[j]
|
|
178
|
+
i = halves[j + 1] * num_blocks / 4_294_967_296 * 8
|
|
179
|
+
x0 = lo & 0xFFFF
|
|
180
|
+
x1 = lo / 65_536
|
|
181
|
+
w[i] |= bit[(x0 * S0 + x1 * L0 * 65536) / 134_217_728 & 31]
|
|
182
|
+
w[i + 1] |= bit[(x0 * S1 + x1 * L1 * 65536) / 134_217_728 & 31]
|
|
183
|
+
w[i + 2] |= bit[(x0 * S2 + x1 * L2 * 65536) / 134_217_728 & 31]
|
|
184
|
+
w[i + 3] |= bit[(x0 * S3 + x1 * L3 * 65536) / 134_217_728 & 31]
|
|
185
|
+
w[i + 4] |= bit[(x0 * S4 + x1 * L4 * 65536) / 134_217_728 & 31]
|
|
186
|
+
w[i + 5] |= bit[(x0 * S5 + x1 * L5 * 65536) / 134_217_728 & 31]
|
|
187
|
+
w[i + 6] |= bit[(x0 * S6 + x1 * L6 * 65536) / 134_217_728 & 31]
|
|
188
|
+
w[i + 7] |= bit[(x0 * S7 + x1 * L7 * 65536) / 134_217_728 & 31]
|
|
189
|
+
j += 2
|
|
190
|
+
end
|
|
191
|
+
self
|
|
192
|
+
end
|
|
193
|
+
|
|
194
|
+
def might_contain_hash?(h)
|
|
195
|
+
i = (((h >> 32) * @num_blocks) >> 32) << 3
|
|
196
|
+
x0 = h & 0xFFFF
|
|
197
|
+
x1 = (h >> 16) & 0xFFFF
|
|
198
|
+
w = @words
|
|
199
|
+
w[i][(((x0 * S0 + (x1 * L0 << 16)) & M32) >> 27)] == 1 &&
|
|
200
|
+
w[i + 1][(((x0 * S1 + (x1 * L1 << 16)) & M32) >> 27)] == 1 &&
|
|
201
|
+
w[i + 2][(((x0 * S2 + (x1 * L2 << 16)) & M32) >> 27)] == 1 &&
|
|
202
|
+
w[i + 3][(((x0 * S3 + (x1 * L3 << 16)) & M32) >> 27)] == 1 &&
|
|
203
|
+
w[i + 4][(((x0 * S4 + (x1 * L4 << 16)) & M32) >> 27)] == 1 &&
|
|
204
|
+
w[i + 5][(((x0 * S5 + (x1 * L5 << 16)) & M32) >> 27)] == 1 &&
|
|
205
|
+
w[i + 6][(((x0 * S6 + (x1 * L6 << 16)) & M32) >> 27)] == 1 &&
|
|
206
|
+
w[i + 7][(((x0 * S7 + (x1 * L7 << 16)) & M32) >> 27)] == 1
|
|
207
|
+
end
|
|
208
|
+
|
|
209
|
+
# The raw bitset (little-endian 32-bit words)
|
|
210
|
+
def bitset = @words.pack("V*")
|
|
211
|
+
|
|
212
|
+
# Header and bitset, as stored in a Parquet file
|
|
213
|
+
def encode
|
|
214
|
+
header.encode << bitset
|
|
215
|
+
end
|
|
216
|
+
|
|
217
|
+
def inspect
|
|
218
|
+
"#<#{self.class.name} #{num_bytes} bytes#{" for #{@column.dotted_path}" if @column}>"
|
|
219
|
+
end
|
|
220
|
+
|
|
221
|
+
private
|
|
222
|
+
|
|
223
|
+
def header
|
|
224
|
+
Format::BloomFilterHeader.new(
|
|
225
|
+
num_bytes: num_bytes,
|
|
226
|
+
algorithm: Format::BloomFilterAlgorithm.new(block: Format::SplitBlockAlgorithm.new),
|
|
227
|
+
hash_function: Format::BloomFilterHash.new(xxhash: Format::XxHash.new),
|
|
228
|
+
compression: Format::BloomFilterCompression.new(uncompressed: Format::BloomFilterUncompressed.new)
|
|
229
|
+
)
|
|
230
|
+
end
|
|
231
|
+
end
|
|
232
|
+
|
|
233
|
+
class Reader
|
|
234
|
+
# Internal (used by reads with where:): the bloom filter of a column chunk. +column+ is a
|
|
235
|
+
# dotted path ("a.b"), an Array path or a Schema::Column. Returns a BloomFilter, or nil when
|
|
236
|
+
# the chunk has none (or one of an unknown kind).
|
|
237
|
+
def bloom_filter(row_group_index, column)
|
|
238
|
+
col = bloom_filter_column(column)
|
|
239
|
+
rg = row_groups.fetch(row_group_index) { raise IndexError, "No row group #{row_group_index}" }
|
|
240
|
+
meta = rg.columns.fetch(col.index).meta_data
|
|
241
|
+
offset = meta&.bloom_filter_offset
|
|
242
|
+
return nil unless offset
|
|
243
|
+
length = meta.bloom_filter_length
|
|
244
|
+
@io.seek(offset)
|
|
245
|
+
if length
|
|
246
|
+
buf = @io.read(length)
|
|
247
|
+
raise FormatError, "Truncated bloom filter" if buf.nil? || buf.bytesize < length
|
|
248
|
+
return BloomFilter.decode(buf, column: col)
|
|
249
|
+
end
|
|
250
|
+
# Without a length (older writers), read the header first, then the bitset it announces
|
|
251
|
+
head = @io.read(256) || "".b
|
|
252
|
+
reader = Thrift::Reader.new(head)
|
|
253
|
+
header = reader.read_struct(Format::BloomFilterHeader)
|
|
254
|
+
return nil unless BloomFilter.supported_header?(header)
|
|
255
|
+
@io.seek(offset + reader.pos)
|
|
256
|
+
bitset = @io.read(header.num_bytes)
|
|
257
|
+
raise FormatError, "Truncated bloom filter" if bitset.nil? || bitset.bytesize != header.num_bytes
|
|
258
|
+
BloomFilter.new(bitset: bitset, column: col)
|
|
259
|
+
end
|
|
260
|
+
|
|
261
|
+
private
|
|
262
|
+
|
|
263
|
+
def bloom_filter_column(column)
|
|
264
|
+
return column if column.is_a?(Schema::Column)
|
|
265
|
+
col = schema.column(column.is_a?(Array) ? column.map(&:to_s) : column.to_s)
|
|
266
|
+
raise ArgumentError, "No column #{column.inspect}" unless col
|
|
267
|
+
col
|
|
268
|
+
end
|
|
269
|
+
end
|
|
270
|
+
end
|
|
@@ -2,25 +2,82 @@
|
|
|
2
2
|
|
|
3
3
|
require "zlib"
|
|
4
4
|
require "stringio"
|
|
5
|
-
require "zstd-ruby"
|
|
6
|
-
require "brotli"
|
|
7
5
|
|
|
8
6
|
module Herringbone
|
|
9
|
-
#
|
|
10
|
-
|
|
7
|
+
# Raised when a file uses (or a writer asks for) a codec whose library is not installed
|
|
8
|
+
class MissingCodecError < UnsupportedError
|
|
9
|
+
attr_reader :codec, :gem_name
|
|
10
|
+
|
|
11
|
+
def initialize(codec, gem_name, load_error)
|
|
12
|
+
@codec = codec
|
|
13
|
+
@gem_name = gem_name
|
|
14
|
+
super("#{codec} compression needs the \"#{gem_name}\" gem, which could not be loaded " \
|
|
15
|
+
"(#{load_error.message}). Add `gem \"#{gem_name}\"` to your Gemfile to use #{codec}.")
|
|
16
|
+
end
|
|
17
|
+
end
|
|
18
|
+
|
|
19
|
+
# Dispatches page (de)compression by Parquet codec id. Snappy and LZ4 are pure Ruby and GZIP
|
|
20
|
+
# uses zlib, so those always work. ZSTD and Brotli come from optional gems (zstd-ruby, brotli),
|
|
21
|
+
# which are required on first use; if they are missing, MissingCodecError says what to add.
|
|
11
22
|
module Compression
|
|
12
23
|
module_function
|
|
13
24
|
|
|
14
|
-
|
|
15
|
-
|
|
16
|
-
|
|
17
|
-
|
|
18
|
-
|
|
25
|
+
# Codecs backed by optional native gems: codec id => [name, gem, require path, constant]
|
|
26
|
+
LIBRARIES = {
|
|
27
|
+
Format::Codec::ZSTD => ["ZSTD", "zstd-ruby", "zstd-ruby", :Zstd],
|
|
28
|
+
Format::Codec::BROTLI => ["Brotli", "brotli", "brotli", :Brotli]
|
|
29
|
+
}.freeze
|
|
30
|
+
|
|
31
|
+
@libraries = {}
|
|
32
|
+
@library_mutex = Mutex.new
|
|
33
|
+
|
|
34
|
+
# The library module for a codec backed by an optional gem, requiring it on first use
|
|
35
|
+
def library(codec)
|
|
36
|
+
@libraries.fetch(codec) do
|
|
37
|
+
@library_mutex.synchronize do
|
|
38
|
+
@libraries.fetch(codec) do
|
|
39
|
+
name, gem_name, path, const = LIBRARIES.fetch(codec)
|
|
40
|
+
begin
|
|
41
|
+
require_library(path)
|
|
42
|
+
rescue LoadError => e
|
|
43
|
+
raise MissingCodecError.new(name, gem_name, e)
|
|
44
|
+
end
|
|
45
|
+
@libraries[codec] = Object.const_get(const)
|
|
46
|
+
end
|
|
47
|
+
end
|
|
48
|
+
end
|
|
49
|
+
end
|
|
50
|
+
|
|
51
|
+
def require_library(path)
|
|
52
|
+
require path
|
|
53
|
+
end
|
|
54
|
+
|
|
55
|
+
# Raises MissingCodecError (or UnsupportedError) unless +codec+ can be used
|
|
56
|
+
def ensure_available!(codec)
|
|
57
|
+
codec = codec_id(codec)
|
|
58
|
+
return library(codec) if LIBRARIES.key?(codec)
|
|
59
|
+
return if SUPPORTED.include?(codec)
|
|
60
|
+
raise UnsupportedError, "#{Format::Codec::NAMES[codec] || codec} compression is not supported"
|
|
61
|
+
end
|
|
62
|
+
|
|
63
|
+
SUPPORTED = [
|
|
64
|
+
Format::Codec::UNCOMPRESSED, Format::Codec::SNAPPY, Format::Codec::GZIP,
|
|
65
|
+
Format::Codec::LZ4_RAW, Format::Codec::LZ4
|
|
66
|
+
].freeze
|
|
67
|
+
|
|
68
|
+
# Codec id => name, as given to the writer's compression: option and listed by Herringbone.codecs
|
|
69
|
+
NAMES = {
|
|
70
|
+
Format::Codec::UNCOMPRESSED => :none, Format::Codec::SNAPPY => :snappy, Format::Codec::GZIP => :gzip,
|
|
71
|
+
Format::Codec::LZ4_RAW => :lz4, Format::Codec::LZ4 => :lz4_hadoop, Format::Codec::ZSTD => :zstd,
|
|
72
|
+
Format::Codec::BROTLI => :brotli, Format::Codec::LZO => :lzo
|
|
19
73
|
}.freeze
|
|
74
|
+
CODECS_BY_NAME = NAMES.invert.freeze
|
|
20
75
|
|
|
21
76
|
def codec_id(name)
|
|
22
77
|
return name if name.is_a?(Integer)
|
|
23
|
-
CODECS_BY_NAME.fetch(name.to_s.downcase.to_sym)
|
|
78
|
+
CODECS_BY_NAME.fetch(name.to_s.downcase.to_sym) do
|
|
79
|
+
raise ArgumentError, "Unknown compression codec #{name.inspect}, expected one of #{NAMES.values.map(&:inspect).join(", ")}"
|
|
80
|
+
end
|
|
24
81
|
end
|
|
25
82
|
|
|
26
83
|
def decompress(codec, data, uncompressed_size)
|
|
@@ -31,14 +88,14 @@ module Herringbone
|
|
|
31
88
|
when Format::Codec::GZIP then gunzip(data)
|
|
32
89
|
when Format::Codec::LZ4_RAW then Codecs::LZ4.decompress_block(data, uncompressed_size)
|
|
33
90
|
when Format::Codec::LZ4 then Codecs::LZ4.decompress_hadoop(data, uncompressed_size)
|
|
34
|
-
when Format::Codec::ZSTD then
|
|
35
|
-
when Format::Codec::BROTLI then
|
|
91
|
+
when Format::Codec::ZSTD then library(codec).decompress(data)
|
|
92
|
+
when Format::Codec::BROTLI then library(codec).inflate(data)
|
|
36
93
|
else
|
|
37
94
|
raise UnsupportedError, "Unsupported compression codec #{Format::Codec::NAMES[codec] || codec}"
|
|
38
95
|
end
|
|
39
96
|
out = out.b
|
|
40
97
|
if out.bytesize != uncompressed_size
|
|
41
|
-
raise
|
|
98
|
+
raise FormatError, "Decompressed #{out.bytesize} bytes, expected #{uncompressed_size}"
|
|
42
99
|
end
|
|
43
100
|
out
|
|
44
101
|
end
|
|
@@ -50,8 +107,8 @@ module Herringbone
|
|
|
50
107
|
when Format::Codec::GZIP then Zlib.gzip(data)
|
|
51
108
|
when Format::Codec::LZ4_RAW then Codecs::LZ4.compress_block(data)
|
|
52
109
|
when Format::Codec::LZ4 then Codecs::LZ4.compress_hadoop(data)
|
|
53
|
-
when Format::Codec::ZSTD then
|
|
54
|
-
when Format::Codec::BROTLI then
|
|
110
|
+
when Format::Codec::ZSTD then library(codec).compress(data)
|
|
111
|
+
when Format::Codec::BROTLI then library(codec).deflate(data)
|
|
55
112
|
else
|
|
56
113
|
raise UnsupportedError, "Unsupported compression codec #{Format::Codec::NAMES[codec] || codec}"
|
|
57
114
|
end.b
|
|
@@ -61,7 +118,7 @@ module Herringbone
|
|
|
61
118
|
def gunzip(data)
|
|
62
119
|
io = StringIO.new(data)
|
|
63
120
|
out = String.new(encoding: Encoding::BINARY)
|
|
64
|
-
|
|
121
|
+
while true
|
|
65
122
|
gz = Zlib::GzipReader.new(io)
|
|
66
123
|
out << gz.read
|
|
67
124
|
unused = gz.unused
|
|
@@ -25,9 +25,9 @@ module Herringbone
|
|
|
25
25
|
miniblocks, pos = RLE.read_uleb(data, pos)
|
|
26
26
|
total, pos = RLE.read_uleb(data, pos)
|
|
27
27
|
first, pos = RLE.read_uleb(data, pos)
|
|
28
|
-
raise
|
|
28
|
+
raise FormatError, "Invalid DELTA_BINARY_PACKED header" if miniblocks.zero? || block_size % miniblocks != 0
|
|
29
29
|
per_mini = block_size / miniblocks
|
|
30
|
-
raise
|
|
30
|
+
raise FormatError, "Invalid miniblock size #{per_mini}" if per_mini % 8 != 0
|
|
31
31
|
total = count if count && count < total
|
|
32
32
|
values = []
|
|
33
33
|
return [values, pos] if total.zero?
|
|
@@ -42,7 +42,7 @@ module Herringbone
|
|
|
42
42
|
pos += miniblocks
|
|
43
43
|
widths.each do |w|
|
|
44
44
|
break if values.size >= total
|
|
45
|
-
raise
|
|
45
|
+
raise FormatError, "Invalid delta bit width #{w}" if w > bits
|
|
46
46
|
deltas = RLE.unpack_bits(data, pos, per_mini, w)
|
|
47
47
|
pos += per_mini * w / 8
|
|
48
48
|
take = total - values.size
|
|
@@ -84,11 +84,11 @@ module Herringbone
|
|
|
84
84
|
|
|
85
85
|
def decode_length_byte_array(data, pos, count)
|
|
86
86
|
lengths, pos = decode_binary_packed(data, pos, 32, count)
|
|
87
|
-
raise
|
|
87
|
+
raise FormatError, "DELTA_LENGTH_BYTE_ARRAY has #{lengths.size} lengths, need #{count}" if lengths.size < count
|
|
88
88
|
out = Array.new(count)
|
|
89
89
|
size = data.bytesize
|
|
90
90
|
lengths.each_with_index do |len, i|
|
|
91
|
-
raise
|
|
91
|
+
raise FormatError, "DELTA_LENGTH_BYTE_ARRAY value overruns page" if len.negative? || pos + len > size
|
|
92
92
|
out[i] = data.byteslice(pos, len)
|
|
93
93
|
pos += len
|
|
94
94
|
end
|
|
@@ -105,7 +105,7 @@ module Herringbone
|
|
|
105
105
|
prev = "".b
|
|
106
106
|
out = Array.new(count) do |i|
|
|
107
107
|
prefix = prefixes[i]
|
|
108
|
-
raise
|
|
108
|
+
raise FormatError, "DELTA_BYTE_ARRAY prefix longer than previous value" if prefix > prev.bytesize
|
|
109
109
|
prev = prefix.zero? ? suffixes[i] : prev.byteslice(0, prefix) + suffixes[i]
|
|
110
110
|
end
|
|
111
111
|
[out, pos]
|
|
@@ -134,7 +134,7 @@ module Herringbone
|
|
|
134
134
|
# Returns the value bytes re-interleaved into PLAIN layout
|
|
135
135
|
def decode(data, pos, count, width)
|
|
136
136
|
nbytes = count * width
|
|
137
|
-
raise
|
|
137
|
+
raise FormatError, "Truncated BYTE_STREAM_SPLIT data" if pos + nbytes > data.bytesize
|
|
138
138
|
streams = Array.new(width) { |k| data.byteslice(pos + k * count, count).unpack("C*") }
|
|
139
139
|
[streams[0].zip(*streams[1..]).flatten.pack("C*"), pos + nbytes]
|
|
140
140
|
end
|
|
@@ -21,26 +21,26 @@ module Herringbone
|
|
|
21
21
|
when Format::Type::BOOLEAN
|
|
22
22
|
nbytes = (count + 7) / 8
|
|
23
23
|
bits = data.byteslice(pos, nbytes).unpack1("b*")
|
|
24
|
-
raise
|
|
24
|
+
raise FormatError, "Truncated BOOLEAN data" if bits.bytesize < count
|
|
25
25
|
[Array.new(count) { |i| bits.getbyte(i) == 49 }, pos + nbytes]
|
|
26
26
|
when Format::Type::INT32, Format::Type::INT64, Format::Type::FLOAT, Format::Type::DOUBLE
|
|
27
27
|
fmt, width = FORMATS[type]
|
|
28
28
|
nbytes = count * width
|
|
29
|
-
raise
|
|
29
|
+
raise FormatError, "Truncated PLAIN data" if pos + nbytes > data.bytesize
|
|
30
30
|
[data.byteslice(pos, nbytes).unpack("#{fmt}#{count}"), pos + nbytes]
|
|
31
31
|
when Format::Type::INT96
|
|
32
32
|
nbytes = count * 12
|
|
33
|
-
raise
|
|
33
|
+
raise FormatError, "Truncated INT96 data" if pos + nbytes > data.bytesize
|
|
34
34
|
vals = data.byteslice(pos, nbytes).unpack("Q<L<" * count).each_slice(2).map { |nanos, day| [nanos, day] }
|
|
35
35
|
[vals, pos + nbytes]
|
|
36
36
|
when Format::Type::BYTE_ARRAY
|
|
37
37
|
decode_byte_arrays(data, pos, count)
|
|
38
38
|
when Format::Type::FIXED_LEN_BYTE_ARRAY
|
|
39
39
|
nbytes = count * type_length
|
|
40
|
-
raise
|
|
40
|
+
raise FormatError, "Truncated FIXED_LEN_BYTE_ARRAY data" if pos + nbytes > data.bytesize
|
|
41
41
|
[Array.new(count) { |i| data.byteslice(pos + i * type_length, type_length) }, pos + nbytes]
|
|
42
42
|
else
|
|
43
|
-
raise
|
|
43
|
+
raise FormatError, "Unknown physical type #{type}"
|
|
44
44
|
end
|
|
45
45
|
end
|
|
46
46
|
|
|
@@ -49,10 +49,10 @@ module Herringbone
|
|
|
49
49
|
size = data.bytesize
|
|
50
50
|
i = 0
|
|
51
51
|
while i < count
|
|
52
|
-
raise
|
|
52
|
+
raise FormatError, "Truncated BYTE_ARRAY data" if pos + 4 > size
|
|
53
53
|
len = data.getbyte(pos) | (data.getbyte(pos + 1) << 8) | (data.getbyte(pos + 2) << 16) | (data.getbyte(pos + 3) << 24)
|
|
54
54
|
pos += 4
|
|
55
|
-
raise
|
|
55
|
+
raise FormatError, "BYTE_ARRAY value overruns page" if pos + len > size
|
|
56
56
|
out[i] = data.byteslice(pos, len)
|
|
57
57
|
pos += len
|
|
58
58
|
i += 1
|
|
@@ -98,7 +98,7 @@ module Herringbone
|
|
|
98
98
|
shift = 0
|
|
99
99
|
while true
|
|
100
100
|
b = data.getbyte(pos)
|
|
101
|
-
raise
|
|
101
|
+
raise FormatError, "Truncated varint" unless b
|
|
102
102
|
pos += 1
|
|
103
103
|
result |= (b & 0x7F) << shift
|
|
104
104
|
return [result, pos] if b < 0x80
|
|
@@ -120,7 +120,7 @@ module Herringbone
|
|
|
120
120
|
out = []
|
|
121
121
|
value_bytes = (width + 7) / 8
|
|
122
122
|
while out.size < count
|
|
123
|
-
raise
|
|
123
|
+
raise FormatError, "RLE data exhausted (#{out.size}/#{count} values)" if pos >= limit
|
|
124
124
|
header, pos = read_uleb(data, pos)
|
|
125
125
|
if header & 1 == 1
|
|
126
126
|
groups = header >> 1
|
data/lib/herringbone/format.rb
CHANGED
|
@@ -292,6 +292,35 @@ module Herringbone
|
|
|
292
292
|
field 7, :ordinal, :i16
|
|
293
293
|
end
|
|
294
294
|
|
|
295
|
+
module BoundaryOrder
|
|
296
|
+
UNORDERED = 0
|
|
297
|
+
ASCENDING = 1
|
|
298
|
+
DESCENDING = 2
|
|
299
|
+
NAMES = constants.to_h { |c| [const_get(c), c] }.freeze
|
|
300
|
+
end
|
|
301
|
+
|
|
302
|
+
# Page index structures (stored between the row groups and the footer)
|
|
303
|
+
class PageLocation < S
|
|
304
|
+
field 1, :offset, :i64
|
|
305
|
+
field 2, :compressed_page_size, :i32
|
|
306
|
+
field 3, :first_row_index, :i64
|
|
307
|
+
end
|
|
308
|
+
|
|
309
|
+
class OffsetIndex < S
|
|
310
|
+
field 1, :page_locations, [:list, PageLocation]
|
|
311
|
+
field 2, :unencoded_byte_array_data_bytes, [:list, :i64]
|
|
312
|
+
end
|
|
313
|
+
|
|
314
|
+
class ColumnIndex < S
|
|
315
|
+
field 1, :null_pages, [:list, :bool]
|
|
316
|
+
field 2, :min_values, [:list, :binary]
|
|
317
|
+
field 3, :max_values, [:list, :binary]
|
|
318
|
+
field 4, :boundary_order, :i32
|
|
319
|
+
field 5, :null_counts, [:list, :i64]
|
|
320
|
+
field 6, :repetition_level_histograms, [:list, :i64]
|
|
321
|
+
field 7, :definition_level_histograms, [:list, :i64]
|
|
322
|
+
end
|
|
323
|
+
|
|
295
324
|
class ColumnOrder < S
|
|
296
325
|
field 1, :type_order, TypeDefinedOrder
|
|
297
326
|
end
|
|
@@ -305,5 +334,29 @@ module Herringbone
|
|
|
305
334
|
field 6, :created_by, :string
|
|
306
335
|
field 7, :column_orders, [:list, ColumnOrder]
|
|
307
336
|
end
|
|
337
|
+
|
|
338
|
+
# Bloom filters (BloomFilter.md). Each union has a single member, an empty struct.
|
|
339
|
+
class SplitBlockAlgorithm < S; end
|
|
340
|
+
class XxHash < S; end
|
|
341
|
+
class BloomFilterUncompressed < S; end
|
|
342
|
+
|
|
343
|
+
class BloomFilterAlgorithm < S
|
|
344
|
+
field 1, :block, SplitBlockAlgorithm
|
|
345
|
+
end
|
|
346
|
+
|
|
347
|
+
class BloomFilterHash < S
|
|
348
|
+
field 1, :xxhash, XxHash
|
|
349
|
+
end
|
|
350
|
+
|
|
351
|
+
class BloomFilterCompression < S
|
|
352
|
+
field 1, :uncompressed, BloomFilterUncompressed
|
|
353
|
+
end
|
|
354
|
+
|
|
355
|
+
class BloomFilterHeader < S
|
|
356
|
+
field 1, :num_bytes, :i32
|
|
357
|
+
field 2, :algorithm, BloomFilterAlgorithm
|
|
358
|
+
field 3, :hash_function, BloomFilterHash # "hash" in parquet.thrift; renamed to keep Object#hash
|
|
359
|
+
field 4, :compression, BloomFilterCompression
|
|
360
|
+
end
|
|
308
361
|
end
|
|
309
362
|
end
|