herringbone 0.1.0 → 0.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/README.md +198 -74
- data/bin/herringbone +38 -26
- data/lib/herringbone/active_record.rb +48 -19
- data/lib/herringbone/bloom_filter.rb +370 -0
- data/lib/herringbone/byte_values.rb +29 -0
- data/lib/herringbone/codecs/lz4.rb +66 -3
- data/lib/herringbone/codecs/snappy.rb +132 -3
- data/lib/herringbone/compression.rb +121 -16
- data/lib/herringbone/encodings/delta.rb +76 -19
- data/lib/herringbone/encodings/plain.rb +29 -7
- data/lib/herringbone/encodings/rle.rb +55 -3
- data/lib/herringbone/format.rb +203 -0
- data/lib/herringbone/inspector.rb +1784 -0
- data/lib/herringbone/io_buffer_support.rb +3 -1
- data/lib/herringbone/reader/column_chunk_reader.rb +388 -0
- data/lib/herringbone/reader/column_cursor.rb +225 -0
- data/lib/herringbone/reader/numo.rb +365 -0
- data/lib/herringbone/reader/page_stream.rb +615 -0
- data/lib/herringbone/reader/scan.rb +350 -0
- data/lib/herringbone/reader.rb +585 -272
- data/lib/herringbone/schema.rb +326 -90
- data/lib/herringbone/thrift.rb +155 -4
- data/lib/herringbone/types.rb +177 -42
- data/lib/herringbone/version.rb +2 -1
- data/lib/herringbone/visualizer.rb +1136 -0
- data/lib/herringbone/writer.rb +550 -105
- data/lib/herringbone/xxhash.rb +425 -0
- data/lib/herringbone.rb +64 -9
- metadata +14 -33
|
@@ -0,0 +1,370 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Herringbone
|
|
4
|
+
# A Parquet Split Block Bloom Filter (parquet-format BloomFilter.md).
|
|
5
|
+
#
|
|
6
|
+
# The bitset is made of 32-byte blocks of eight 32-bit words. A value is hashed with XXH64 over
|
|
7
|
+
# the PLAIN encoding of its physical value (without the length prefix for BYTE_ARRAY); the upper
|
|
8
|
+
# 32 bits of the hash pick a block, and the lower 32 bits set one bit in each of its words.
|
|
9
|
+
#
|
|
10
|
+
# A filter answers "definitely not in the column chunk" or "maybe": +might_contain?+ never
|
|
11
|
+
# returns false for a value that was inserted. Values are given as Ruby values and converted
|
|
12
|
+
# like the writer converts them (the column's encoder), so a Date, Time, BigDecimal or UUID
|
|
13
|
+
# String hashes the same bytes as the stored value. Nulls are never in a bloom filter.
|
|
14
|
+
# The writer builds them (bloom_filters: option) and reads with where: consult them.
|
|
15
|
+
class BloomFilter
|
|
16
|
+
# The spec's eight salt constants, one per word of a block
|
|
17
|
+
SALT = [0x47b6137b, 0x44974d91, 0x8824ad5b, 0xa2b7289d, 0x705495c7, 0x2df1424b, 0x9efc4947, 0x5c6bfb31].freeze
|
|
18
|
+
S0, S1, S2, S3, S4, S5, S6, S7 = SALT
|
|
19
|
+
# Low 16 bits of the salts: (lo * salt) mod 2**32 is computed as
|
|
20
|
+
# (lo & 0xFFFF) * salt + ((lo >> 16) * (salt & 0xFFFF) << 16), which stays a Fixnum
|
|
21
|
+
L0, L1, L2, L3, L4, L5, L6, L7 = SALT.map { |s| s & 0xFFFF }
|
|
22
|
+
# Size of a block: eight 32-bit words
|
|
23
|
+
BLOCK_BYTES = 32
|
|
24
|
+
# Smallest bitset: a single block
|
|
25
|
+
MIN_BYTES = 32
|
|
26
|
+
# Largest bitset accepted when reading or building (128 MiB, the cap parquet-mr uses)
|
|
27
|
+
MAX_BYTES = 128 * 1024 * 1024
|
|
28
|
+
# Default cap for filters sized by the writer (as in parquet-mr)
|
|
29
|
+
DEFAULT_MAX_BYTES = 1024 * 1024
|
|
30
|
+
# Default false positive probability for filters sized by the writer
|
|
31
|
+
DEFAULT_FPP = 0.01
|
|
32
|
+
# 32-bit mask
|
|
33
|
+
M32 = 0xFFFF_FFFF
|
|
34
|
+
# 64-bit mask
|
|
35
|
+
M64 = 0xFFFF_FFFF_FFFF_FFFF
|
|
36
|
+
# Shorthand for the physical type constants
|
|
37
|
+
T = Format::Type
|
|
38
|
+
# Physical types a bloom filter can be built for (the spec does not define BOOLEAN hashing)
|
|
39
|
+
TYPES = [T::INT32, T::INT64, T::INT96, T::FLOAT, T::DOUBLE, T::BYTE_ARRAY, T::FIXED_LEN_BYTE_ARRAY].freeze
|
|
40
|
+
|
|
41
|
+
# Bitset size in bytes for +ndv+ distinct values at false positive probability +fpp+, per the
|
|
42
|
+
# spec's formula (m = -8 * ndv / ln(1 - fpp ** (1/8)) bits), rounded up to a power of two
|
|
43
|
+
# and clamped to MIN_BYTES..max_bytes
|
|
44
|
+
#
|
|
45
|
+
# @param ndv [Integer] expected number of distinct values (values below 1 count as 1)
|
|
46
|
+
# @param fpp [Float] false positive probability, strictly between 0 and 1
|
|
47
|
+
# @param max_bytes [Integer] upper bound, clamped to MIN_BYTES..MAX_BYTES and rounded down to a
|
|
48
|
+
# power of two
|
|
49
|
+
# @return [Integer] bitset size in bytes, a power of two
|
|
50
|
+
# @raise [ArgumentError] when +fpp+ is not between 0 and 1
|
|
51
|
+
def self.optimal_num_bytes(ndv, fpp = DEFAULT_FPP, max_bytes: DEFAULT_MAX_BYTES)
|
|
52
|
+
fpp = Float(fpp)
|
|
53
|
+
raise ArgumentError, "fpp must be between 0 and 1, got #{fpp}" unless fpp > 0 && fpp < 1
|
|
54
|
+
max_bytes = Integer(max_bytes).clamp(MIN_BYTES, MAX_BYTES)
|
|
55
|
+
max_bytes = 1 << (max_bytes.bit_length - 1) # a power of two
|
|
56
|
+
ndv = [Integer(ndv), 1].max
|
|
57
|
+
bits = -8.0 * ndv / Math.log(1 - fpp**(1.0 / 8))
|
|
58
|
+
bytes = (bits / 8).ceil
|
|
59
|
+
return max_bytes if bytes >= max_bytes
|
|
60
|
+
return MIN_BYTES if bytes <= MIN_BYTES
|
|
61
|
+
1 << (bytes - 1).bit_length
|
|
62
|
+
end
|
|
63
|
+
|
|
64
|
+
# XXH64 of the PLAIN encoding of a physical value of +type+ (as the column encoders produce it)
|
|
65
|
+
#
|
|
66
|
+
# @param value [Integer, Float, String, Array<Integer>] the physical value; for INT96 the
|
|
67
|
+
# +[nanos_of_day, julian_day]+ pair the encoder produces
|
|
68
|
+
# @param type [Integer] Format::Type physical type
|
|
69
|
+
# @return [Integer] the unsigned 64-bit hash
|
|
70
|
+
# @raise [UnsupportedError] for BOOLEAN (or any type not in TYPES)
|
|
71
|
+
def self.hash_physical(value, type)
|
|
72
|
+
case type
|
|
73
|
+
when T::INT32 then XXHash.xxh64_u32(value)
|
|
74
|
+
when T::INT64 then XXHash.xxh64_u64(value)
|
|
75
|
+
when T::FLOAT then XXHash.xxh64_u32([value].pack("e").unpack1("L<"))
|
|
76
|
+
when T::DOUBLE then XXHash.xxh64_u64([value].pack("E").unpack1("Q<"))
|
|
77
|
+
when T::INT96 then XXHash.xxh64(value.pack("Q<L<"))
|
|
78
|
+
when T::BYTE_ARRAY, T::FIXED_LEN_BYTE_ARRAY then XXHash.xxh64(value)
|
|
79
|
+
else raise UnsupportedError, "Bloom filters are not defined for #{T::NAMES.fetch(type, type)} values"
|
|
80
|
+
end
|
|
81
|
+
end
|
|
82
|
+
|
|
83
|
+
# Hashes of many physical values of +type+, converting them in bulk. With +distinct+, each
|
|
84
|
+
# distinct physical value is hashed once (floats are compared by their bytes, so -0.0 and 0.0,
|
|
85
|
+
# or NaNs with different payloads, stay apart as they hash differently).
|
|
86
|
+
#
|
|
87
|
+
# @param values [Array] physical values, as for .hash_physical
|
|
88
|
+
# @param type [Integer] Format::Type physical type
|
|
89
|
+
# @param distinct [Boolean] hash each distinct value once (the result is then shorter)
|
|
90
|
+
# @return [Array<Integer>] the unsigned 64-bit hashes
|
|
91
|
+
# @raise [UnsupportedError] for BOOLEAN (or any type not in TYPES)
|
|
92
|
+
def self.hash_physical_all(values, type, distinct: false)
|
|
93
|
+
case type
|
|
94
|
+
when T::INT32 then XXHash.xxh64_u32_all(distinct ? values.uniq : values)
|
|
95
|
+
when T::INT64 then XXHash.xxh64_u64_all(distinct ? values.uniq : values)
|
|
96
|
+
when T::FLOAT
|
|
97
|
+
words = values.pack("e*").unpack("L<*")
|
|
98
|
+
XXHash.xxh64_u32_all(distinct ? words.uniq! || words : words)
|
|
99
|
+
when T::DOUBLE
|
|
100
|
+
lanes = values.pack("E*").unpack("Q<*")
|
|
101
|
+
XXHash.xxh64_u64_all(distinct ? lanes.uniq! || lanes : lanes)
|
|
102
|
+
when T::INT96 then XXHash.xxh64_all((distinct ? values.uniq : values).map { |v| v.pack("Q<L<") })
|
|
103
|
+
when T::BYTE_ARRAY, T::FIXED_LEN_BYTE_ARRAY then XXHash.xxh64_all(distinct ? values.uniq : values)
|
|
104
|
+
else raise UnsupportedError, "Bloom filters are not defined for #{T::NAMES.fetch(type, type)} values"
|
|
105
|
+
end
|
|
106
|
+
end
|
|
107
|
+
|
|
108
|
+
# Reads a filter (header and bitset) from +buf+ at +pos+. Returns nil for algorithms, hashes
|
|
109
|
+
# or compressions this implementation does not know.
|
|
110
|
+
#
|
|
111
|
+
# @param buf [String] bytes holding the BloomFilterHeader followed by the bitset
|
|
112
|
+
# @param pos [Integer] byte offset of the header in +buf+
|
|
113
|
+
# @param column [Schema::Column, nil] column the filter belongs to, for converting values
|
|
114
|
+
# @return [BloomFilter, nil] the filter, or nil when it is of an unsupported kind
|
|
115
|
+
# @raise [FormatError] when the bitset is shorter than the header announces
|
|
116
|
+
# @raise [Thrift::Error] when the header cannot be decoded
|
|
117
|
+
def self.decode(buf, pos = 0, column: nil)
|
|
118
|
+
reader = Thrift::Reader.new(buf, pos)
|
|
119
|
+
header = reader.read_struct(Format::BloomFilterHeader)
|
|
120
|
+
return nil unless supported_header?(header)
|
|
121
|
+
bitset = buf.byteslice(reader.pos, header.num_bytes)
|
|
122
|
+
raise FormatError, "Truncated bloom filter" if bitset.nil? || bitset.bytesize != header.num_bytes
|
|
123
|
+
new(bitset: bitset, column: column)
|
|
124
|
+
end
|
|
125
|
+
|
|
126
|
+
# Whether a header describes a filter this implementation can read: split block algorithm,
|
|
127
|
+
# XXH64, uncompressed, and a size that is a multiple of 32 bytes up to MAX_BYTES
|
|
128
|
+
#
|
|
129
|
+
# @param header [Format::BloomFilterHeader] decoded header
|
|
130
|
+
# @return [Boolean] true when supported
|
|
131
|
+
def self.supported_header?(header)
|
|
132
|
+
n = header.num_bytes
|
|
133
|
+
n.is_a?(Integer) && n >= BLOCK_BYTES && (n % BLOCK_BYTES).zero? && n <= MAX_BYTES &&
|
|
134
|
+
!(header.algorithm && header.algorithm.block).nil? &&
|
|
135
|
+
!(header.hash_function && header.hash_function.xxhash).nil? &&
|
|
136
|
+
(header.compression.nil? || !header.compression.uncompressed.nil?)
|
|
137
|
+
end
|
|
138
|
+
|
|
139
|
+
# @return [Schema::Column, nil] column whose encoder converts values, nil for raw Strings
|
|
140
|
+
attr_reader :column
|
|
141
|
+
|
|
142
|
+
# A filter of +num_bytes+ (a multiple of 32, normally a power of two), or one using an existing
|
|
143
|
+
# +bitset+ String. With a +column+ (a Schema::Column), values are Ruby values converted with
|
|
144
|
+
# the column's encoder; without one, values must be Strings and their bytes are hashed.
|
|
145
|
+
#
|
|
146
|
+
# @param num_bytes [Integer, nil] bitset size for an empty filter (MIN_BYTES when nil); ignored
|
|
147
|
+
# with +bitset+
|
|
148
|
+
# @param bitset [String, nil] existing bitset (little-endian 32-bit words)
|
|
149
|
+
# @param column [Schema::Column, nil] column the filter is for
|
|
150
|
+
# @raise [ArgumentError] when the size (or +bitset+'s size) is not a multiple of 32 bytes or
|
|
151
|
+
# exceeds MAX_BYTES
|
|
152
|
+
# @raise [UnsupportedError] when the column's physical type cannot have a bloom filter
|
|
153
|
+
def initialize(num_bytes = nil, bitset: nil, column: nil)
|
|
154
|
+
num_bytes = bitset ? bitset.bytesize : Integer(num_bytes || MIN_BYTES)
|
|
155
|
+
unless num_bytes >= BLOCK_BYTES && (num_bytes % BLOCK_BYTES).zero? && num_bytes <= MAX_BYTES
|
|
156
|
+
raise ArgumentError, "Bloom filter size must be a multiple of 32 bytes up to 128MB, got #{num_bytes}"
|
|
157
|
+
end
|
|
158
|
+
@words = bitset ? bitset.unpack("V*") : Array.new(num_bytes / 4, 0)
|
|
159
|
+
@num_blocks = @words.size / 8
|
|
160
|
+
@column = column
|
|
161
|
+
if column && !TYPES.include?(column.type)
|
|
162
|
+
raise UnsupportedError, "Bloom filters are not supported for #{T::NAMES[column.type]} column #{column.dotted_path}"
|
|
163
|
+
end
|
|
164
|
+
end
|
|
165
|
+
|
|
166
|
+
# @return [Integer] bitset size in bytes
|
|
167
|
+
def num_bytes = @words.size * 4
|
|
168
|
+
|
|
169
|
+
# Adds a value to the filter
|
|
170
|
+
#
|
|
171
|
+
# @param value [Object] Ruby value, converted with the column's encoder (a String without a
|
|
172
|
+
# column)
|
|
173
|
+
# @return [BloomFilter] self
|
|
174
|
+
# @raise [ArgumentError] for nil or a value the column cannot encode
|
|
175
|
+
def insert(value)
|
|
176
|
+
insert_hash(hash_of(value))
|
|
177
|
+
self
|
|
178
|
+
end
|
|
179
|
+
|
|
180
|
+
# Whether the value may be in the filter. Never false for an inserted value; true for a
|
|
181
|
+
# value that was not inserted with about the false positive probability the filter was sized for.
|
|
182
|
+
#
|
|
183
|
+
# @param value [Object] Ruby value, converted with the column's encoder (a String without a
|
|
184
|
+
# column)
|
|
185
|
+
# @return [Boolean] false when the value is definitely absent
|
|
186
|
+
# @raise [ArgumentError] for nil or a value the column cannot encode
|
|
187
|
+
def might_contain?(value)
|
|
188
|
+
might_contain_hash?(hash_of(value))
|
|
189
|
+
end
|
|
190
|
+
|
|
191
|
+
# XXH64 hash the filter uses for a Ruby +value+
|
|
192
|
+
#
|
|
193
|
+
# @param value [Object] Ruby value, converted with the column's encoder (a String without a
|
|
194
|
+
# column)
|
|
195
|
+
# @return [Integer] the unsigned 64-bit hash
|
|
196
|
+
# @raise [ArgumentError] for nil, a value the column cannot encode, or (without a column) a
|
|
197
|
+
# non-String
|
|
198
|
+
def hash_of(value)
|
|
199
|
+
raise ArgumentError, "Nulls are not recorded in bloom filters" if value.nil?
|
|
200
|
+
if @column
|
|
201
|
+
begin
|
|
202
|
+
physical = @column.encoder.call(value)
|
|
203
|
+
rescue ArgumentError, TypeError, NoMethodError, RangeError, EncodeError => e
|
|
204
|
+
raise ArgumentError, "#{value.inspect} is not a valid value for #{@column.dotted_path}: #{e.message}"
|
|
205
|
+
end
|
|
206
|
+
self.class.hash_physical(physical, @column.type)
|
|
207
|
+
else
|
|
208
|
+
raise ArgumentError, "Without a column, bloom filter values must be Strings, got #{value.class}" unless value.is_a?(String)
|
|
209
|
+
XXHash.xxh64(value)
|
|
210
|
+
end
|
|
211
|
+
end
|
|
212
|
+
|
|
213
|
+
# Sets the bits for a hash: the upper 32 bits pick the block, the lower 32 bits (multiplied
|
|
214
|
+
# by each salt) one bit in each of its eight words
|
|
215
|
+
#
|
|
216
|
+
# @param h [Integer] unsigned 64-bit XXH64 hash
|
|
217
|
+
# @return [BloomFilter] self
|
|
218
|
+
def insert_hash(h)
|
|
219
|
+
i = (((h >> 32) * @num_blocks) >> 32) << 3
|
|
220
|
+
x0 = h & 0xFFFF
|
|
221
|
+
x1 = (h >> 16) & 0xFFFF
|
|
222
|
+
w = @words
|
|
223
|
+
w[i] |= 1 << (((x0 * S0 + (x1 * L0 << 16)) & M32) >> 27)
|
|
224
|
+
w[i + 1] |= 1 << (((x0 * S1 + (x1 * L1 << 16)) & M32) >> 27)
|
|
225
|
+
w[i + 2] |= 1 << (((x0 * S2 + (x1 * L2 << 16)) & M32) >> 27)
|
|
226
|
+
w[i + 3] |= 1 << (((x0 * S3 + (x1 * L3 << 16)) & M32) >> 27)
|
|
227
|
+
w[i + 4] |= 1 << (((x0 * S4 + (x1 * L4 << 16)) & M32) >> 27)
|
|
228
|
+
w[i + 5] |= 1 << (((x0 * S5 + (x1 * L5 << 16)) & M32) >> 27)
|
|
229
|
+
w[i + 6] |= 1 << (((x0 * S6 + (x1 * L6 << 16)) & M32) >> 27)
|
|
230
|
+
w[i + 7] |= 1 << (((x0 * S7 + (x1 * L7 << 16)) & M32) >> 27)
|
|
231
|
+
self
|
|
232
|
+
end
|
|
233
|
+
|
|
234
|
+
# Single-bit masks, +BITS[i] == 1 << i+
|
|
235
|
+
BITS = Array.new(32) { |i| 1 << i }.freeze
|
|
236
|
+
|
|
237
|
+
# Inserts many hashes at once: faster than insert_hash, as the hashes (mostly Bignums) are
|
|
238
|
+
# split into 32-bit halves in bulk and the rest is Fixnum arithmetic
|
|
239
|
+
#
|
|
240
|
+
# @param hashes [Array<Integer>] unsigned 64-bit XXH64 hashes
|
|
241
|
+
# @return [BloomFilter] self
|
|
242
|
+
def insert_hashes(hashes)
|
|
243
|
+
halves = hashes.pack("Q<*").unpack("V*")
|
|
244
|
+
w = @words
|
|
245
|
+
num_blocks = @num_blocks
|
|
246
|
+
bit = BITS
|
|
247
|
+
j = 0
|
|
248
|
+
n = halves.size
|
|
249
|
+
while j < n
|
|
250
|
+
lo = halves[j]
|
|
251
|
+
i = halves[j + 1] * num_blocks / 4_294_967_296 * 8
|
|
252
|
+
x0 = lo & 0xFFFF
|
|
253
|
+
x1 = lo / 65_536
|
|
254
|
+
w[i] |= bit[(x0 * S0 + x1 * L0 * 65536) / 134_217_728 & 31]
|
|
255
|
+
w[i + 1] |= bit[(x0 * S1 + x1 * L1 * 65536) / 134_217_728 & 31]
|
|
256
|
+
w[i + 2] |= bit[(x0 * S2 + x1 * L2 * 65536) / 134_217_728 & 31]
|
|
257
|
+
w[i + 3] |= bit[(x0 * S3 + x1 * L3 * 65536) / 134_217_728 & 31]
|
|
258
|
+
w[i + 4] |= bit[(x0 * S4 + x1 * L4 * 65536) / 134_217_728 & 31]
|
|
259
|
+
w[i + 5] |= bit[(x0 * S5 + x1 * L5 * 65536) / 134_217_728 & 31]
|
|
260
|
+
w[i + 6] |= bit[(x0 * S6 + x1 * L6 * 65536) / 134_217_728 & 31]
|
|
261
|
+
w[i + 7] |= bit[(x0 * S7 + x1 * L7 * 65536) / 134_217_728 & 31]
|
|
262
|
+
j += 2
|
|
263
|
+
end
|
|
264
|
+
self
|
|
265
|
+
end
|
|
266
|
+
|
|
267
|
+
# Whether all the bits for a hash are set (see #insert_hash)
|
|
268
|
+
#
|
|
269
|
+
# @param h [Integer] unsigned 64-bit XXH64 hash
|
|
270
|
+
# @return [Boolean] false when the hashed value is definitely absent
|
|
271
|
+
def might_contain_hash?(h)
|
|
272
|
+
i = (((h >> 32) * @num_blocks) >> 32) << 3
|
|
273
|
+
x0 = h & 0xFFFF
|
|
274
|
+
x1 = (h >> 16) & 0xFFFF
|
|
275
|
+
w = @words
|
|
276
|
+
w[i][(((x0 * S0 + (x1 * L0 << 16)) & M32) >> 27)] == 1 &&
|
|
277
|
+
w[i + 1][(((x0 * S1 + (x1 * L1 << 16)) & M32) >> 27)] == 1 &&
|
|
278
|
+
w[i + 2][(((x0 * S2 + (x1 * L2 << 16)) & M32) >> 27)] == 1 &&
|
|
279
|
+
w[i + 3][(((x0 * S3 + (x1 * L3 << 16)) & M32) >> 27)] == 1 &&
|
|
280
|
+
w[i + 4][(((x0 * S4 + (x1 * L4 << 16)) & M32) >> 27)] == 1 &&
|
|
281
|
+
w[i + 5][(((x0 * S5 + (x1 * L5 << 16)) & M32) >> 27)] == 1 &&
|
|
282
|
+
w[i + 6][(((x0 * S6 + (x1 * L6 << 16)) & M32) >> 27)] == 1 &&
|
|
283
|
+
w[i + 7][(((x0 * S7 + (x1 * L7 << 16)) & M32) >> 27)] == 1
|
|
284
|
+
end
|
|
285
|
+
|
|
286
|
+
# The raw bitset (little-endian 32-bit words)
|
|
287
|
+
#
|
|
288
|
+
# @return [String] binary String of #num_bytes bytes
|
|
289
|
+
def bitset = @words.pack("V*")
|
|
290
|
+
|
|
291
|
+
# Header and bitset, as stored in a Parquet file
|
|
292
|
+
#
|
|
293
|
+
# @return [String] Thrift-encoded BloomFilterHeader followed by the bitset (binary)
|
|
294
|
+
def encode
|
|
295
|
+
header.encode << bitset
|
|
296
|
+
end
|
|
297
|
+
|
|
298
|
+
# @return [String] size and, when set, the column's dotted path
|
|
299
|
+
def inspect
|
|
300
|
+
"#<#{self.class.name} #{num_bytes} bytes#{" for #{@column.dotted_path}" if @column}>"
|
|
301
|
+
end
|
|
302
|
+
|
|
303
|
+
private
|
|
304
|
+
|
|
305
|
+
# Header for this filter: split block algorithm, XXH64 hash, uncompressed
|
|
306
|
+
#
|
|
307
|
+
# @return [Format::BloomFilterHeader] the header
|
|
308
|
+
def header
|
|
309
|
+
Format::BloomFilterHeader.new(
|
|
310
|
+
num_bytes: num_bytes,
|
|
311
|
+
algorithm: Format::BloomFilterAlgorithm.new(block: Format::SplitBlockAlgorithm.new),
|
|
312
|
+
hash_function: Format::BloomFilterHash.new(xxhash: Format::XxHash.new),
|
|
313
|
+
compression: Format::BloomFilterCompression.new(uncompressed: Format::BloomFilterUncompressed.new)
|
|
314
|
+
)
|
|
315
|
+
end
|
|
316
|
+
end
|
|
317
|
+
|
|
318
|
+
class Reader
|
|
319
|
+
# Internal (used by reads with where:): the bloom filter of a column chunk. +column+ is a
|
|
320
|
+
# dotted path ("a.b"), an Array path or a Schema::Column. Returns a BloomFilter, or nil when
|
|
321
|
+
# the chunk has none (or one of an unknown kind).
|
|
322
|
+
#
|
|
323
|
+
# @param row_group_index [Integer] index of the row group
|
|
324
|
+
# @param column [String, Array<String, Symbol>, Symbol, Schema::Column] the leaf column
|
|
325
|
+
# @return [BloomFilter, nil] the filter, or nil when there is none or it is unsupported
|
|
326
|
+
# @raise [IndexError] when there is no such row group
|
|
327
|
+
# @raise [ArgumentError] when there is no such column
|
|
328
|
+
# @raise [FormatError] when the filter is truncated or its header cannot be decoded
|
|
329
|
+
def bloom_filter(row_group_index, column)
|
|
330
|
+
col = bloom_filter_column(column)
|
|
331
|
+
rg = row_groups.fetch(row_group_index) { raise IndexError, "No row group #{row_group_index}" }
|
|
332
|
+
meta = rg.columns.fetch(col.index).meta_data
|
|
333
|
+
offset = meta&.bloom_filter_offset
|
|
334
|
+
return nil unless offset
|
|
335
|
+
length = meta.bloom_filter_length
|
|
336
|
+
@io.seek(offset)
|
|
337
|
+
if length
|
|
338
|
+
buf = @io.read(length)
|
|
339
|
+
raise FormatError, "Truncated bloom filter" if buf.nil? || buf.bytesize < length
|
|
340
|
+
return BloomFilter.decode(buf, column: col)
|
|
341
|
+
end
|
|
342
|
+
# Without a length (older writers), read the header first, then the bitset it announces
|
|
343
|
+
head = @io.read(256) || "".b
|
|
344
|
+
reader = Thrift::Reader.new(head)
|
|
345
|
+
header = reader.read_struct(Format::BloomFilterHeader)
|
|
346
|
+
return nil unless BloomFilter.supported_header?(header)
|
|
347
|
+
@io.seek(offset + reader.pos)
|
|
348
|
+
bitset = @io.read(header.num_bytes)
|
|
349
|
+
raise FormatError, "Truncated bloom filter" if bitset.nil? || bitset.bytesize != header.num_bytes
|
|
350
|
+
BloomFilter.new(bitset: bitset, column: col)
|
|
351
|
+
rescue Thrift::Error => e
|
|
352
|
+
raise FormatError, "Corrupt bloom filter header for #{col.dotted_path}: #{e.message}"
|
|
353
|
+
end
|
|
354
|
+
|
|
355
|
+
private
|
|
356
|
+
|
|
357
|
+
# Resolves the +column+ argument of #bloom_filter to a leaf column
|
|
358
|
+
#
|
|
359
|
+
# @param column [String, Array<String, Symbol>, Symbol, Schema::Column] dotted path, path
|
|
360
|
+
# Array or column
|
|
361
|
+
# @return [Schema::Column] the column
|
|
362
|
+
# @raise [ArgumentError] when the schema has no such column
|
|
363
|
+
def bloom_filter_column(column)
|
|
364
|
+
return column if column.is_a?(Schema::Column)
|
|
365
|
+
col = schema.column(column.is_a?(Array) ? column.map(&:to_s) : column.to_s)
|
|
366
|
+
raise ArgumentError, "No column #{column.inspect}" unless col
|
|
367
|
+
col
|
|
368
|
+
end
|
|
369
|
+
end
|
|
370
|
+
end
|
|
@@ -17,8 +17,11 @@ module Herringbone
|
|
|
17
17
|
# After this many values, give up on the dictionary if more than half of them are distinct
|
|
18
18
|
CARDINALITY_CHECK_AT = 4096
|
|
19
19
|
|
|
20
|
+
# Whether String#append_as_bytes (Ruby 3.4+) is available for appending without re-encoding
|
|
20
21
|
APPEND_AS_BYTES = "".respond_to?(:append_as_bytes)
|
|
21
22
|
|
|
23
|
+
# @param width [Integer, nil] byte length of each value for FIXED_LEN_BYTE_ARRAY, nil for BYTE_ARRAY
|
|
24
|
+
# @param dictionary [Boolean] start in dictionary mode; false appends raw bytes from the start
|
|
22
25
|
def initialize(width: nil, dictionary: true)
|
|
23
26
|
@width = width
|
|
24
27
|
@bytes = String.new(encoding: Encoding::BINARY)
|
|
@@ -31,14 +34,21 @@ module Herringbone
|
|
|
31
34
|
end
|
|
32
35
|
end
|
|
33
36
|
|
|
37
|
+
# @return [Integer] number of values held
|
|
34
38
|
def size
|
|
35
39
|
@indices ? @indices.size : @count
|
|
36
40
|
end
|
|
37
41
|
|
|
42
|
+
# @return [Boolean] true when no values are held
|
|
38
43
|
def empty? = size.zero?
|
|
39
44
|
|
|
45
|
+
# @return [Boolean] true while still dictionary-encoding (not yet switched to raw bytes)
|
|
40
46
|
def dictionary? = !@indices.nil?
|
|
41
47
|
|
|
48
|
+
# Appends a value, switching to raw bytes when the dictionary limits are exceeded.
|
|
49
|
+
# @param value [String] bytes of the value; for FIXED_LEN_BYTE_ARRAY it must be +width+ bytes
|
|
50
|
+
# long (not checked here)
|
|
51
|
+
# @return [ByteValues] self
|
|
42
52
|
def <<(value)
|
|
43
53
|
if @indices
|
|
44
54
|
index = @dictionary[value]
|
|
@@ -61,16 +71,20 @@ module Herringbone
|
|
|
61
71
|
end
|
|
62
72
|
|
|
63
73
|
# Removes the last value
|
|
74
|
+
# @return [void]
|
|
64
75
|
def pop
|
|
65
76
|
truncate(size - 1) unless size.zero?
|
|
66
77
|
end
|
|
67
78
|
|
|
68
79
|
# Supports the `slice!(n..)` form used to roll back a failed row
|
|
80
|
+
# @param range [Range] endless range; values from +range.begin+ on are dropped
|
|
81
|
+
# @return [void]
|
|
69
82
|
def slice!(range)
|
|
70
83
|
truncate(range.begin)
|
|
71
84
|
end
|
|
72
85
|
|
|
73
86
|
# Approximate memory held, used to size row groups
|
|
87
|
+
# @return [Integer] estimated bytes
|
|
74
88
|
def memory_bytes
|
|
75
89
|
if @indices
|
|
76
90
|
# Each distinct value is a String object plus a Hash entry
|
|
@@ -82,6 +96,11 @@ module Herringbone
|
|
|
82
96
|
|
|
83
97
|
# Returns [:dictionary, values, indices] when the column is worth dictionary-encoding,
|
|
84
98
|
# otherwise [:plain, values]
|
|
99
|
+
#
|
|
100
|
+
# Dictionary encoding is kept unless there are more than 16 values and more than about half
|
|
101
|
+
# of them are distinct.
|
|
102
|
+
# @return [Array(Symbol, Array<String>, Array<Integer>), Array(Symbol, Array<String>)]
|
|
103
|
+
# distinct values and an index per value, or all values in order
|
|
85
104
|
def materialize
|
|
86
105
|
if @indices
|
|
87
106
|
keys = @dictionary.keys
|
|
@@ -93,6 +112,9 @@ module Herringbone
|
|
|
93
112
|
|
|
94
113
|
private
|
|
95
114
|
|
|
115
|
+
# Appends +value+ in raw-bytes mode, recording its length for BYTE_ARRAY.
|
|
116
|
+
# @param value [String] bytes of the value, in any encoding
|
|
117
|
+
# @return [ByteValues] self
|
|
96
118
|
def append_bytes(value)
|
|
97
119
|
if value.encoding == Encoding::BINARY || value.ascii_only?
|
|
98
120
|
@bytes << value
|
|
@@ -106,6 +128,8 @@ module Herringbone
|
|
|
106
128
|
self
|
|
107
129
|
end
|
|
108
130
|
|
|
131
|
+
# Leaves dictionary mode, re-appending every value held so far as raw bytes.
|
|
132
|
+
# @return [void]
|
|
109
133
|
def switch_to_bytes
|
|
110
134
|
keys = @dictionary.keys
|
|
111
135
|
indices = @indices
|
|
@@ -113,6 +137,9 @@ module Herringbone
|
|
|
113
137
|
indices.each { |i| append_bytes(keys[i]) }
|
|
114
138
|
end
|
|
115
139
|
|
|
140
|
+
# Drops all values after the first +n+.
|
|
141
|
+
# @param n [Integer] number of values to keep
|
|
142
|
+
# @return [void]
|
|
116
143
|
def truncate(n)
|
|
117
144
|
if @indices
|
|
118
145
|
@indices.slice!(n..)
|
|
@@ -128,6 +155,8 @@ module Herringbone
|
|
|
128
155
|
end
|
|
129
156
|
end
|
|
130
157
|
|
|
158
|
+
# Splits the raw bytes back into one binary String per value.
|
|
159
|
+
# @return [Array<String>]
|
|
131
160
|
def strings
|
|
132
161
|
if @lengths
|
|
133
162
|
pos = 0
|
|
@@ -5,17 +5,26 @@ module Herringbone
|
|
|
5
5
|
# Pure-Ruby LZ4: raw block format (Parquet LZ4_RAW), Hadoop-framed blocks
|
|
6
6
|
# (Parquet's deprecated LZ4) and a decoder for the LZ4 frame format.
|
|
7
7
|
module LZ4
|
|
8
|
+
# Raised for corrupt, truncated or unsupported LZ4 input.
|
|
8
9
|
class Error < StandardError; end
|
|
9
10
|
|
|
11
|
+
# Shortest match a sequence can encode; the token's match length is stored minus this.
|
|
10
12
|
MIN_MATCH = 4
|
|
11
13
|
LAST_LITERALS = 5 # the last 5 bytes of a block are always literals
|
|
12
14
|
MFLIMIT = 12 # the last match must start at least 12 bytes before the end
|
|
15
|
+
# Largest backward distance a 2-byte match offset can express.
|
|
13
16
|
MAX_OFFSET = 65_535
|
|
17
|
+
# Size of the compressor's hash table, in bits (16384 entries).
|
|
14
18
|
HASH_LOG = 14
|
|
19
|
+
# Right shift that keeps the top HASH_LOG bits of the 32-bit multiplicative hash.
|
|
15
20
|
HASH_SHIFT = 32 - HASH_LOG
|
|
21
|
+
# Controls how fast the compressor skips ahead through input without matches (as in the reference).
|
|
16
22
|
SKIP_STRENGTH = 6
|
|
23
|
+
# Magic number at the start of an LZ4 frame (little-endian).
|
|
17
24
|
FRAME_MAGIC = 0x184D2204
|
|
25
|
+
# Hadoop block header: big-endian 32-bit uncompressed size and compressed size.
|
|
18
26
|
HADOOP_PREFIX = 8
|
|
27
|
+
# Shorthand for Encoding::BINARY.
|
|
19
28
|
BINARY = Encoding::BINARY
|
|
20
29
|
# String#unpack1 accepts offset: since Ruby 3.1
|
|
21
30
|
UNPACK_OFFSET = begin
|
|
@@ -28,6 +37,10 @@ module Herringbone
|
|
|
28
37
|
module_function
|
|
29
38
|
|
|
30
39
|
# Decompress a raw LZ4 block that must expand to exactly uncompressed_size bytes.
|
|
40
|
+
# @param input [String] raw LZ4 block
|
|
41
|
+
# @param uncompressed_size [Integer] exact decompressed size, from the page header
|
|
42
|
+
# @return [String] decompressed bytes in ASCII-8BIT
|
|
43
|
+
# @raise [Error] if the block is corrupt or does not decode to +uncompressed_size+ bytes
|
|
31
44
|
def decompress_block(input, uncompressed_size)
|
|
32
45
|
src = binary(input)
|
|
33
46
|
out = String.new(capacity: uncompressed_size, encoding: BINARY)
|
|
@@ -40,6 +53,10 @@ module Herringbone
|
|
|
40
53
|
|
|
41
54
|
# Parquet LZ4 (codec 5). Tries Hadoop framing, then the LZ4 frame format,
|
|
42
55
|
# then a bare raw block (Arrow falls back hadoop -> raw; some writers emitted frames).
|
|
56
|
+
# @param input [String] compressed page data
|
|
57
|
+
# @param uncompressed_size [Integer] exact decompressed size, from the page header
|
|
58
|
+
# @return [String] decompressed bytes in ASCII-8BIT
|
|
59
|
+
# @raise [Error] if none of the three layouts decodes
|
|
43
60
|
def decompress_hadoop(input, uncompressed_size)
|
|
44
61
|
src = binary(input)
|
|
45
62
|
result = try_hadoop(src, uncompressed_size)
|
|
@@ -58,6 +75,10 @@ module Herringbone
|
|
|
58
75
|
|
|
59
76
|
# Decode LZ4 frame format data (one or more frames, skippable frames ignored).
|
|
60
77
|
# Checksums are skipped, not verified.
|
|
78
|
+
# @param input [String] one or more concatenated LZ4 frames
|
|
79
|
+
# @param max_size [Integer] upper bound on the decompressed size
|
|
80
|
+
# @return [String] decompressed bytes in ASCII-8BIT, possibly shorter than +max_size+
|
|
81
|
+
# @raise [Error] on bad magic, truncation, unsupported frame features or output over +max_size+
|
|
61
82
|
def decompress_frame(input, max_size)
|
|
62
83
|
src = binary(input)
|
|
63
84
|
n = src.bytesize
|
|
@@ -79,6 +100,8 @@ module Herringbone
|
|
|
79
100
|
end
|
|
80
101
|
|
|
81
102
|
# Compress into a single raw LZ4 block.
|
|
103
|
+
# @param input [String] bytes to compress
|
|
104
|
+
# @return [String] raw LZ4 block in ASCII-8BIT
|
|
82
105
|
def compress_block(input)
|
|
83
106
|
src = binary(input)
|
|
84
107
|
n = src.bytesize
|
|
@@ -121,7 +144,7 @@ module Herringbone
|
|
|
121
144
|
# Emit the sequence: token, literal run, offset, extended match length
|
|
122
145
|
lit_len = ip - anchor
|
|
123
146
|
ml = len - MIN_MATCH
|
|
124
|
-
out << (((lit_len < 15 ? lit_len : 15) << 4) | (ml < 15 ? ml : 15))
|
|
147
|
+
out << ((((lit_len < 15) ? lit_len : 15) << 4) | ((ml < 15) ? ml : 15))
|
|
125
148
|
write_length(out, lit_len - 15) if lit_len >= 15
|
|
126
149
|
out << src.byteslice(anchor, lit_len) if lit_len > 0
|
|
127
150
|
offset = ip - ref
|
|
@@ -144,6 +167,8 @@ module Herringbone
|
|
|
144
167
|
end
|
|
145
168
|
|
|
146
169
|
# Single Hadoop-framed block: [BE uncompressed size][BE compressed size][raw block]
|
|
170
|
+
# @param input [String] bytes to compress
|
|
171
|
+
# @return [String] Hadoop-framed LZ4 data in ASCII-8BIT
|
|
147
172
|
def compress_hadoop(input)
|
|
148
173
|
src = binary(input)
|
|
149
174
|
block = compress_block(src)
|
|
@@ -152,18 +177,31 @@ module Herringbone
|
|
|
152
177
|
|
|
153
178
|
# -- internals --
|
|
154
179
|
|
|
180
|
+
# @param str [String] input in any encoding
|
|
181
|
+
# @return [String] +str+ itself if already binary, otherwise a binary copy
|
|
155
182
|
def binary(str)
|
|
156
|
-
str.encoding == BINARY ? str : str.b
|
|
183
|
+
(str.encoding == BINARY) ? str : str.b
|
|
157
184
|
end
|
|
158
185
|
|
|
186
|
+
# @param src [String] binary input
|
|
187
|
+
# @param i [Integer] byte offset
|
|
188
|
+
# @return [Integer] unsigned little-endian 32-bit value at +i+
|
|
159
189
|
def le32(src, i)
|
|
160
190
|
src.byteslice(i, 4).unpack1("V")
|
|
161
191
|
end
|
|
162
192
|
|
|
193
|
+
# Same as le32 but via getbyte, for Rubies without unpack1(offset:).
|
|
194
|
+
# @param src [String] binary input
|
|
195
|
+
# @param i [Integer] byte offset
|
|
196
|
+
# @return [Integer] unsigned little-endian 32-bit value at +i+
|
|
163
197
|
def u32(src, i)
|
|
164
198
|
src.getbyte(i) | (src.getbyte(i + 1) << 8) | (src.getbyte(i + 2) << 16) | (src.getbyte(i + 3) << 24)
|
|
165
199
|
end
|
|
166
200
|
|
|
201
|
+
# Appends the extension of a literal or match length: 255-bytes followed by the remainder.
|
|
202
|
+
# @param out [String] binary output buffer, appended to
|
|
203
|
+
# @param len [Integer] length minus the 15 already stored in the token
|
|
204
|
+
# @return [String] +out+
|
|
167
205
|
def write_length(out, len)
|
|
168
206
|
if len >= 255
|
|
169
207
|
out << ("\xFF".b * (len / 255))
|
|
@@ -172,14 +210,27 @@ module Herringbone
|
|
|
172
210
|
out << len
|
|
173
211
|
end
|
|
174
212
|
|
|
213
|
+
# Appends the final, literals-only sequence that terminates every block.
|
|
214
|
+
# @param out [String] binary output buffer, appended to
|
|
215
|
+
# @param src [String] binary input
|
|
216
|
+
# @param anchor [Integer] offset of the first pending literal in +src+
|
|
217
|
+
# @param lit_len [Integer] number of trailing literal bytes (may be zero)
|
|
218
|
+
# @return [String, nil] +out+, or nil when there are no literals
|
|
175
219
|
def emit_last_literals(out, src, anchor, lit_len)
|
|
176
|
-
out << ((lit_len < 15 ? lit_len : 15) << 4)
|
|
220
|
+
out << (((lit_len < 15) ? lit_len : 15) << 4)
|
|
177
221
|
write_length(out, lit_len - 15) if lit_len >= 15
|
|
178
222
|
out << src.byteslice(anchor, lit_len) if lit_len > 0
|
|
179
223
|
end
|
|
180
224
|
|
|
181
225
|
# Decode one raw block from src[ip...iend], appending to out (which may already
|
|
182
226
|
# hold earlier data that matches can reference). out may not grow beyond limit.
|
|
227
|
+
# @param src [String] binary input
|
|
228
|
+
# @param ip [Integer] offset of the block in +src+
|
|
229
|
+
# @param iend [Integer] offset just past the block
|
|
230
|
+
# @param out [String] binary output buffer, appended to
|
|
231
|
+
# @param limit [Integer] maximum total byte size of +out+
|
|
232
|
+
# @return [Integer] input offset where decoding stopped
|
|
233
|
+
# @raise [Error] if the block is truncated, has an invalid offset or overflows +limit+
|
|
183
234
|
def decode_block(src, ip, iend, out, limit)
|
|
184
235
|
while ip < iend
|
|
185
236
|
token = src.getbyte(ip)
|
|
@@ -237,6 +288,9 @@ module Herringbone
|
|
|
237
288
|
end
|
|
238
289
|
|
|
239
290
|
# Arrow-compatible Hadoop frame parsing; returns nil if the data does not fit the framing.
|
|
291
|
+
# @param src [String] binary input
|
|
292
|
+
# @param uncompressed_size [Integer] exact decompressed size expected over all blocks
|
|
293
|
+
# @return [String, nil] decompressed bytes, or nil if +src+ is not valid Hadoop-framed LZ4
|
|
240
294
|
def try_hadoop(src, uncompressed_size)
|
|
241
295
|
n = src.bytesize
|
|
242
296
|
return nil if n < HADOOP_PREFIX
|
|
@@ -262,6 +316,15 @@ module Herringbone
|
|
|
262
316
|
out
|
|
263
317
|
end
|
|
264
318
|
|
|
319
|
+
# Decodes the descriptor and data blocks of one LZ4 frame (after its magic number),
|
|
320
|
+
# skipping content size and checksums.
|
|
321
|
+
# @param src [String] binary input
|
|
322
|
+
# @param ip [Integer] offset of the frame descriptor (just past the magic)
|
|
323
|
+
# @param n [Integer] byte size of +src+
|
|
324
|
+
# @param out [String] binary output buffer, appended to
|
|
325
|
+
# @param limit [Integer] maximum total byte size of +out+
|
|
326
|
+
# @return [Integer] offset just past the frame
|
|
327
|
+
# @raise [Error] on truncation, an unsupported version or dictionary, or output over +limit+
|
|
265
328
|
def decode_frame(src, ip, n, out, limit)
|
|
266
329
|
raise Error, "Truncated LZ4 frame descriptor" if ip + 3 > n
|
|
267
330
|
flg = src.getbyte(ip)
|