herringbone 0.1.0 → 0.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/README.md +146 -74
- data/bin/herringbone +38 -26
- data/lib/herringbone/active_record.rb +10 -13
- data/lib/herringbone/bloom_filter.rb +270 -0
- data/lib/herringbone/compression.rb +73 -16
- data/lib/herringbone/encodings/delta.rb +7 -7
- data/lib/herringbone/encodings/plain.rb +7 -7
- data/lib/herringbone/encodings/rle.rb +2 -2
- data/lib/herringbone/format.rb +53 -0
- data/lib/herringbone/inspector.rb +1388 -0
- data/lib/herringbone/reader/column_chunk_reader.rb +296 -0
- data/lib/herringbone/reader/column_cursor.rb +181 -0
- data/lib/herringbone/reader/page_stream.rb +339 -0
- data/lib/herringbone/reader/scan.rb +256 -0
- data/lib/herringbone/reader.rb +294 -272
- data/lib/herringbone/schema.rb +21 -76
- data/lib/herringbone/types.rb +4 -4
- data/lib/herringbone/version.rb +1 -1
- data/lib/herringbone/visualizer.rb +1096 -0
- data/lib/herringbone/writer.rb +258 -90
- data/lib/herringbone/xxhash.rb +319 -0
- data/lib/herringbone.rb +36 -9
- metadata +12 -32
|
@@ -0,0 +1,339 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Herringbone
|
|
4
|
+
class Reader
|
|
5
|
+
# Incremental decoders for the contents of one data page. The page bytes are decoded as they
|
|
6
|
+
# are asked for, so a caller that takes a few hundred entries at a time never holds a whole
|
|
7
|
+
# page's worth of levels or values as Ruby objects.
|
|
8
|
+
module PageStream
|
|
9
|
+
# Values unpacked from a bit-packed run at a time (a multiple of 8)
|
|
10
|
+
CHUNK = 1024
|
|
11
|
+
|
|
12
|
+
# The levels and values of one data page
|
|
13
|
+
class Page
|
|
14
|
+
attr_reader :remaining, :converter
|
|
15
|
+
|
|
16
|
+
def initialize(entries, defs, reps, values, converter)
|
|
17
|
+
@remaining = entries
|
|
18
|
+
@defs = defs
|
|
19
|
+
@reps = reps
|
|
20
|
+
@values = values
|
|
21
|
+
@converter = converter
|
|
22
|
+
end
|
|
23
|
+
|
|
24
|
+
# [definition_levels, repetition_levels] of the next +n+ entries (nil when a column has no
|
|
25
|
+
# levels of that kind)
|
|
26
|
+
def read_levels(n)
|
|
27
|
+
n = @remaining if n > @remaining
|
|
28
|
+
@remaining -= n
|
|
29
|
+
[@defs&.read(n), @reps&.read(n)]
|
|
30
|
+
end
|
|
31
|
+
|
|
32
|
+
# The next +n+ values, as physical values (apply #converter for Ruby values)
|
|
33
|
+
def read_values(n)
|
|
34
|
+
n.zero? ? [] : @values.read(n)
|
|
35
|
+
end
|
|
36
|
+
|
|
37
|
+
# Moves past the next +n+ values without returning them
|
|
38
|
+
def skip_values(n)
|
|
39
|
+
return if n.zero?
|
|
40
|
+
@values.respond_to?(:skip) ? @values.skip(n) : @values.read(n)
|
|
41
|
+
end
|
|
42
|
+
end
|
|
43
|
+
|
|
44
|
+
# A decoder over an Array that is already decoded (legacy encodings, booleans, deltas)
|
|
45
|
+
class ArrayDecoder
|
|
46
|
+
def initialize(values)
|
|
47
|
+
@values = values
|
|
48
|
+
@i = 0
|
|
49
|
+
end
|
|
50
|
+
|
|
51
|
+
def read(n)
|
|
52
|
+
raise FormatError, "Page has fewer values than its levels require" if @i + n > @values.size
|
|
53
|
+
out = @values[@i, n]
|
|
54
|
+
@i += n
|
|
55
|
+
out
|
|
56
|
+
end
|
|
57
|
+
end
|
|
58
|
+
|
|
59
|
+
# The RLE / bit-packed hybrid, decoded run by run
|
|
60
|
+
class HybridDecoder
|
|
61
|
+
def initialize(data, pos, limit, width)
|
|
62
|
+
@data = data
|
|
63
|
+
@pos = pos
|
|
64
|
+
@limit = limit
|
|
65
|
+
@width = width
|
|
66
|
+
@value_bytes = (width + 7) / 8
|
|
67
|
+
@left = 0 # values left in the current run
|
|
68
|
+
@rle = true
|
|
69
|
+
@value = 0
|
|
70
|
+
@buf = nil
|
|
71
|
+
@bi = 0
|
|
72
|
+
@groups = 0 # bit-packed groups of 8 not unpacked yet
|
|
73
|
+
end
|
|
74
|
+
|
|
75
|
+
def read(n)
|
|
76
|
+
out = []
|
|
77
|
+
while n > 0
|
|
78
|
+
next_run if @left.zero?
|
|
79
|
+
t = n < @left ? n : @left
|
|
80
|
+
if @rle
|
|
81
|
+
out.fill(@value, out.size, t)
|
|
82
|
+
else
|
|
83
|
+
if @buf.nil? || @bi >= @buf.size
|
|
84
|
+
g = @groups < CHUNK / 8 ? @groups : CHUNK / 8
|
|
85
|
+
@buf = Encodings::RLE.unpack_bits(@data, @pos, g * 8, @width)
|
|
86
|
+
@pos += g * @width
|
|
87
|
+
@groups -= g
|
|
88
|
+
@bi = 0
|
|
89
|
+
end
|
|
90
|
+
avail = @buf.size - @bi
|
|
91
|
+
t = avail if t > avail
|
|
92
|
+
out.concat(@buf[@bi, t])
|
|
93
|
+
@bi += t
|
|
94
|
+
end
|
|
95
|
+
@left -= t
|
|
96
|
+
n -= t
|
|
97
|
+
end
|
|
98
|
+
out
|
|
99
|
+
end
|
|
100
|
+
|
|
101
|
+
private
|
|
102
|
+
|
|
103
|
+
def next_run
|
|
104
|
+
while @left.zero?
|
|
105
|
+
raise FormatError, "RLE data exhausted" if @pos >= @limit
|
|
106
|
+
header, @pos = Encodings::RLE.read_uleb(@data, @pos)
|
|
107
|
+
if header & 1 == 1
|
|
108
|
+
@rle = false
|
|
109
|
+
@groups = header >> 1
|
|
110
|
+
@left = @groups * 8
|
|
111
|
+
@buf = nil
|
|
112
|
+
else
|
|
113
|
+
@rle = true
|
|
114
|
+
@left = header >> 1
|
|
115
|
+
v = 0
|
|
116
|
+
@value_bytes.times { |k| v |= @data.getbyte(@pos + k).to_i << (8 * k) }
|
|
117
|
+
@value = v
|
|
118
|
+
@pos += @value_bytes
|
|
119
|
+
end
|
|
120
|
+
end
|
|
121
|
+
end
|
|
122
|
+
end
|
|
123
|
+
|
|
124
|
+
# PLAIN INT32 / INT64 / FLOAT / DOUBLE / INT96
|
|
125
|
+
class FixedDecoder
|
|
126
|
+
def initialize(data, pos, format, width)
|
|
127
|
+
@data = data
|
|
128
|
+
@pos = pos
|
|
129
|
+
@format = format
|
|
130
|
+
@width = width
|
|
131
|
+
end
|
|
132
|
+
|
|
133
|
+
def read(n)
|
|
134
|
+
bytes = n * @width
|
|
135
|
+
raise FormatError, "Truncated PLAIN data" if @pos + bytes > @data.bytesize
|
|
136
|
+
out = @data.byteslice(@pos, bytes).unpack("#{@format}#{n}")
|
|
137
|
+
@pos += bytes
|
|
138
|
+
out
|
|
139
|
+
end
|
|
140
|
+
|
|
141
|
+
def skip(n)
|
|
142
|
+
raise FormatError, "Truncated PLAIN data" if @pos + n * @width > @data.bytesize
|
|
143
|
+
@pos += n * @width
|
|
144
|
+
end
|
|
145
|
+
end
|
|
146
|
+
|
|
147
|
+
class Int96Decoder < FixedDecoder
|
|
148
|
+
def initialize(data, pos)
|
|
149
|
+
super(data, pos, "Q<L<", 12)
|
|
150
|
+
end
|
|
151
|
+
|
|
152
|
+
def read(n)
|
|
153
|
+
bytes = n * 12
|
|
154
|
+
raise FormatError, "Truncated INT96 data" if @pos + bytes > @data.bytesize
|
|
155
|
+
out = @data.byteslice(@pos, bytes).unpack("Q<L<" * n).each_slice(2).to_a
|
|
156
|
+
@pos += bytes
|
|
157
|
+
out
|
|
158
|
+
end
|
|
159
|
+
end
|
|
160
|
+
|
|
161
|
+
# PLAIN FIXED_LEN_BYTE_ARRAY
|
|
162
|
+
class FixedBytesDecoder
|
|
163
|
+
def initialize(data, pos, width)
|
|
164
|
+
@data = data
|
|
165
|
+
@pos = pos
|
|
166
|
+
@width = width
|
|
167
|
+
end
|
|
168
|
+
|
|
169
|
+
def read(n)
|
|
170
|
+
raise FormatError, "Truncated FIXED_LEN_BYTE_ARRAY data" if @pos + n * @width > @data.bytesize
|
|
171
|
+
out = Array.new(n) { |i| @data.byteslice(@pos + i * @width, @width) }
|
|
172
|
+
@pos += n * @width
|
|
173
|
+
out
|
|
174
|
+
end
|
|
175
|
+
|
|
176
|
+
def skip(n)
|
|
177
|
+
raise FormatError, "Truncated FIXED_LEN_BYTE_ARRAY data" if @pos + n * @width > @data.bytesize
|
|
178
|
+
@pos += n * @width
|
|
179
|
+
end
|
|
180
|
+
end
|
|
181
|
+
|
|
182
|
+
# PLAIN BYTE_ARRAY: 4-byte length, then the bytes
|
|
183
|
+
class ByteArrayDecoder
|
|
184
|
+
def initialize(data, pos)
|
|
185
|
+
@data = data
|
|
186
|
+
@pos = pos
|
|
187
|
+
end
|
|
188
|
+
|
|
189
|
+
def read(n)
|
|
190
|
+
data = @data
|
|
191
|
+
size = data.bytesize
|
|
192
|
+
pos = @pos
|
|
193
|
+
out = Array.new(n)
|
|
194
|
+
i = 0
|
|
195
|
+
while i < n
|
|
196
|
+
raise FormatError, "Truncated BYTE_ARRAY data" if pos + 4 > size
|
|
197
|
+
len = data.getbyte(pos) | (data.getbyte(pos + 1) << 8) | (data.getbyte(pos + 2) << 16) | (data.getbyte(pos + 3) << 24)
|
|
198
|
+
pos += 4
|
|
199
|
+
raise FormatError, "BYTE_ARRAY value overruns page" if pos + len > size
|
|
200
|
+
out[i] = data.byteslice(pos, len)
|
|
201
|
+
pos += len
|
|
202
|
+
i += 1
|
|
203
|
+
end
|
|
204
|
+
@pos = pos
|
|
205
|
+
out
|
|
206
|
+
end
|
|
207
|
+
|
|
208
|
+
# Walks the length prefixes without creating Strings
|
|
209
|
+
def skip(n)
|
|
210
|
+
data = @data
|
|
211
|
+
size = data.bytesize
|
|
212
|
+
pos = @pos
|
|
213
|
+
n.times do
|
|
214
|
+
raise FormatError, "Truncated BYTE_ARRAY data" if pos + 4 > size
|
|
215
|
+
pos += 4 + (data.getbyte(pos) | (data.getbyte(pos + 1) << 8) | (data.getbyte(pos + 2) << 16) | (data.getbyte(pos + 3) << 24))
|
|
216
|
+
end
|
|
217
|
+
raise FormatError, "BYTE_ARRAY value overruns page" if pos > size
|
|
218
|
+
@pos = pos
|
|
219
|
+
end
|
|
220
|
+
end
|
|
221
|
+
|
|
222
|
+
# PLAIN BOOLEAN: one bit per value, LSB first
|
|
223
|
+
class BooleanDecoder
|
|
224
|
+
def initialize(data, pos)
|
|
225
|
+
@bits = data.byteslice(pos, data.bytesize - pos).unpack1("b*")
|
|
226
|
+
@i = 0
|
|
227
|
+
end
|
|
228
|
+
|
|
229
|
+
def read(n)
|
|
230
|
+
raise FormatError, "Truncated BOOLEAN data" if @i + n > @bits.bytesize
|
|
231
|
+
out = Array.new(n) { |k| @bits.getbyte(@i + k) == 49 }
|
|
232
|
+
@i += n
|
|
233
|
+
out
|
|
234
|
+
end
|
|
235
|
+
end
|
|
236
|
+
|
|
237
|
+
# RLE_DICTIONARY / PLAIN_DICTIONARY indices mapped to the (already converted) dictionary
|
|
238
|
+
class DictionaryDecoder
|
|
239
|
+
def initialize(data, pos, dictionary, path)
|
|
240
|
+
@indices = HybridDecoder.new(data, pos + 1, data.bytesize, data.getbyte(pos).to_i)
|
|
241
|
+
@dictionary = dictionary
|
|
242
|
+
@path = path
|
|
243
|
+
end
|
|
244
|
+
|
|
245
|
+
# Indices are decoded but not looked up
|
|
246
|
+
def skip(n)
|
|
247
|
+
@indices.read(n)
|
|
248
|
+
end
|
|
249
|
+
|
|
250
|
+
def read(n)
|
|
251
|
+
indices = @indices.read(n)
|
|
252
|
+
dict = @dictionary
|
|
253
|
+
raise FormatError, "Dictionary index out of range in #{@path}" if indices.max >= dict.size
|
|
254
|
+
indices.map! { |i| dict[i] }
|
|
255
|
+
end
|
|
256
|
+
end
|
|
257
|
+
|
|
258
|
+
# RLE-encoded BOOLEAN values
|
|
259
|
+
class RleBooleanDecoder
|
|
260
|
+
def initialize(data, pos)
|
|
261
|
+
len = data.byteslice(pos, 4).unpack1("V")
|
|
262
|
+
@bits = HybridDecoder.new(data, pos + 4, pos + 4 + len, 1)
|
|
263
|
+
end
|
|
264
|
+
|
|
265
|
+
def read(n)
|
|
266
|
+
@bits.read(n).map! { |v| v == 1 }
|
|
267
|
+
end
|
|
268
|
+
end
|
|
269
|
+
|
|
270
|
+
# DELTA_LENGTH_BYTE_ARRAY: all lengths first (Integers), then the bytes sliced as needed
|
|
271
|
+
class DeltaLengthDecoder
|
|
272
|
+
def initialize(data, pos)
|
|
273
|
+
@lengths, @pos = Encodings::Delta.decode_binary_packed(data, pos, 32)
|
|
274
|
+
@data = data
|
|
275
|
+
@i = 0
|
|
276
|
+
end
|
|
277
|
+
|
|
278
|
+
def read(n)
|
|
279
|
+
raise FormatError, "DELTA_LENGTH_BYTE_ARRAY has too few values" if @i + n > @lengths.size
|
|
280
|
+
data = @data
|
|
281
|
+
out = Array.new(n) do |k|
|
|
282
|
+
len = @lengths[@i + k]
|
|
283
|
+
raise FormatError, "DELTA_LENGTH_BYTE_ARRAY value overruns page" if len.negative? || @pos + len > data.bytesize
|
|
284
|
+
s = data.byteslice(@pos, len)
|
|
285
|
+
@pos += len
|
|
286
|
+
s
|
|
287
|
+
end
|
|
288
|
+
@i += n
|
|
289
|
+
out
|
|
290
|
+
end
|
|
291
|
+
end
|
|
292
|
+
|
|
293
|
+
# DELTA_BYTE_ARRAY: prefix lengths and suffixes, each value built from the previous one
|
|
294
|
+
class DeltaByteArrayDecoder
|
|
295
|
+
def initialize(data, pos)
|
|
296
|
+
@prefixes, pos = Encodings::Delta.decode_binary_packed(data, pos, 32)
|
|
297
|
+
@suffixes = DeltaLengthDecoder.new(data, pos)
|
|
298
|
+
@prev = "".b
|
|
299
|
+
@i = 0
|
|
300
|
+
end
|
|
301
|
+
|
|
302
|
+
def read(n)
|
|
303
|
+
raise FormatError, "DELTA_BYTE_ARRAY has too few values" if @i + n > @prefixes.size
|
|
304
|
+
suffixes = @suffixes.read(n)
|
|
305
|
+
prev = @prev
|
|
306
|
+
out = Array.new(n) do |k|
|
|
307
|
+
prefix = @prefixes[@i + k]
|
|
308
|
+
raise FormatError, "DELTA_BYTE_ARRAY prefix longer than previous value" if prefix > prev.bytesize
|
|
309
|
+
prev = prefix.zero? ? suffixes[k] : prev.byteslice(0, prefix) + suffixes[k]
|
|
310
|
+
end
|
|
311
|
+
@prev = prev
|
|
312
|
+
@i += n
|
|
313
|
+
out
|
|
314
|
+
end
|
|
315
|
+
end
|
|
316
|
+
|
|
317
|
+
# BYTE_STREAM_SPLIT: byte k of every value lives in stream k; values are gathered per call
|
|
318
|
+
class ByteStreamSplitDecoder
|
|
319
|
+
def initialize(data, pos, width, type, type_length)
|
|
320
|
+
@data = data
|
|
321
|
+
@pos = pos
|
|
322
|
+
@width = width
|
|
323
|
+
@count = (data.bytesize - pos) / width
|
|
324
|
+
@type = type
|
|
325
|
+
@type_length = type_length
|
|
326
|
+
@i = 0
|
|
327
|
+
end
|
|
328
|
+
|
|
329
|
+
def read(n)
|
|
330
|
+
raise FormatError, "Truncated BYTE_STREAM_SPLIT data" if @i + n > @count
|
|
331
|
+
streams = Array.new(@width) { |k| @data.byteslice(@pos + k * @count + @i, n) }.join
|
|
332
|
+
@i += n
|
|
333
|
+
plain, = Encodings::ByteStreamSplit.decode(streams, 0, n, @width)
|
|
334
|
+
Encodings::Plain.decode(plain, 0, n, @type, @type_length).first
|
|
335
|
+
end
|
|
336
|
+
end
|
|
337
|
+
end
|
|
338
|
+
end
|
|
339
|
+
end
|
|
@@ -0,0 +1,256 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Herringbone
|
|
4
|
+
class Reader
|
|
5
|
+
# Row selection for where: filters. A filter maps columns to conditions:
|
|
6
|
+
#
|
|
7
|
+
# where: { "user_id" => 42, # equality
|
|
8
|
+
# "status" => %w[paid shipped], # any of these (IN)
|
|
9
|
+
# "created_at" => t1...t2, # Ranges, also endless / beginless
|
|
10
|
+
# "address.city" => "Amsterdam", # a member of a struct, by dotted path
|
|
11
|
+
# "deleted_at" => nil, # IS NULL
|
|
12
|
+
# "amount" => ->(v) { v && v > 100 } } # any callable (row check only)
|
|
13
|
+
#
|
|
14
|
+
# All conditions must hold. The filter is used three ways, from cheapest to most exact:
|
|
15
|
+
# whole row groups are ruled out with column chunk statistics (min/max, null counts) and
|
|
16
|
+
# bloom filters; within a row group, pages are ruled out with the page index (ColumnIndex +
|
|
17
|
+
# OffsetIndex), leaving ranges of rows to read; and every row that is read is checked, so
|
|
18
|
+
# results are exact.
|
|
19
|
+
class Filter
|
|
20
|
+
Condition = Struct.new(:name, :column, :field, :path, :test)
|
|
21
|
+
|
|
22
|
+
attr_reader :conditions
|
|
23
|
+
|
|
24
|
+
def initialize(schema, where)
|
|
25
|
+
raise ArgumentError, "where: must be a Hash of column => condition" unless where.is_a?(Hash)
|
|
26
|
+
@conditions = where.map do |key, test|
|
|
27
|
+
name = key.to_s
|
|
28
|
+
column = schema.column(name) or raise ArgumentError, "where: no such column #{name.inspect}"
|
|
29
|
+
if column.max_repetition_level.positive?
|
|
30
|
+
raise ArgumentError, "where: #{name} is inside a list or map; only non-repeated columns can be filtered on"
|
|
31
|
+
end
|
|
32
|
+
field = schema.field(column.path.first)
|
|
33
|
+
Condition.new(name, column, field, column.path.drop(1), normalize(test))
|
|
34
|
+
end
|
|
35
|
+
end
|
|
36
|
+
|
|
37
|
+
# Top-level fields the conditions need to read
|
|
38
|
+
def fields
|
|
39
|
+
@conditions.map(&:field).uniq
|
|
40
|
+
end
|
|
41
|
+
|
|
42
|
+
# --- statistics-based pruning ---
|
|
43
|
+
|
|
44
|
+
# Whether any row of row group +rg_index+ may match, judging by chunk statistics and bloom
|
|
45
|
+
# filters
|
|
46
|
+
def row_group_may_match?(reader, rg_index)
|
|
47
|
+
chunks = reader.row_groups[rg_index].columns
|
|
48
|
+
@conditions.all? do |c|
|
|
49
|
+
meta = chunks[c.column.index].meta_data
|
|
50
|
+
next true unless meta
|
|
51
|
+
min, max = Filter.stat_range(c.column, meta.statistics)
|
|
52
|
+
nulls = meta.statistics&.null_count
|
|
53
|
+
all_null = nulls && nulls == meta.num_values
|
|
54
|
+
Filter.may_match?(c.test, min, max, nulls, all_null) && bloom_may_match?(reader, rg_index, c)
|
|
55
|
+
end
|
|
56
|
+
end
|
|
57
|
+
|
|
58
|
+
# Sorted, non-overlapping [first_row, end_row) ranges of row group +rg_index+ that may hold
|
|
59
|
+
# matching rows, judging by the page index. Columns without a page index do not narrow
|
|
60
|
+
# the ranges.
|
|
61
|
+
def page_ranges(reader, rg_index)
|
|
62
|
+
n = reader.row_groups[rg_index].num_rows
|
|
63
|
+
ranges = [[0, n]]
|
|
64
|
+
@conditions.each do |c|
|
|
65
|
+
column_index, offset_index = reader.page_index(rg_index, c.column)
|
|
66
|
+
next unless column_index && offset_index
|
|
67
|
+
locs = offset_index.page_locations
|
|
68
|
+
next unless locs && column_index.null_pages && locs.size == column_index.null_pages.size
|
|
69
|
+
candidates = []
|
|
70
|
+
locs.each_with_index do |loc, i|
|
|
71
|
+
min = max = nil
|
|
72
|
+
unless column_index.null_pages[i]
|
|
73
|
+
min = Filter.decode_stat(c.column, column_index.min_values[i])
|
|
74
|
+
max = Filter.decode_stat(c.column, column_index.max_values[i])
|
|
75
|
+
end
|
|
76
|
+
nulls = column_index.null_counts&.[](i)
|
|
77
|
+
next unless Filter.may_match?(c.test, min, max, nulls, column_index.null_pages[i])
|
|
78
|
+
stop = i + 1 < locs.size ? locs[i + 1].first_row_index : n
|
|
79
|
+
candidates << [loc.first_row_index, stop]
|
|
80
|
+
end
|
|
81
|
+
ranges = Filter.intersect(ranges, Filter.merge(candidates))
|
|
82
|
+
break if ranges.empty?
|
|
83
|
+
end
|
|
84
|
+
ranges
|
|
85
|
+
end
|
|
86
|
+
|
|
87
|
+
# --- row checks ---
|
|
88
|
+
|
|
89
|
+
# Indexes (0...k) of the rows of an assembled batch that match. +data+ holds one Array per
|
|
90
|
+
# field in +fields+; +symbolize+ says how struct Hashes are keyed.
|
|
91
|
+
def matching_rows(data, fields, k, symbolize)
|
|
92
|
+
columns = @conditions.map do |c|
|
|
93
|
+
values = data.fetch(fields.index(c.field))
|
|
94
|
+
next values if c.path.empty?
|
|
95
|
+
path = symbolize ? c.path.map(&:to_sym) : c.path
|
|
96
|
+
values.map { |v| path.reduce(v) { |h, key| h&.[](key) } }
|
|
97
|
+
end
|
|
98
|
+
tests = @conditions.map(&:test)
|
|
99
|
+
(0...k).select do |i|
|
|
100
|
+
j = 0
|
|
101
|
+
ok = true
|
|
102
|
+
while ok && j < tests.size
|
|
103
|
+
ok = Filter.matches?(tests[j], columns[j][i])
|
|
104
|
+
j += 1
|
|
105
|
+
end
|
|
106
|
+
ok
|
|
107
|
+
end
|
|
108
|
+
end
|
|
109
|
+
|
|
110
|
+
# --- condition evaluation ---
|
|
111
|
+
|
|
112
|
+
def self.matches?(test, v)
|
|
113
|
+
case test
|
|
114
|
+
when nil then v.nil?
|
|
115
|
+
when Array then test.any? { |t| matches?(t, v) }
|
|
116
|
+
when Range then !v.nil? && (test.cover?(v) rescue false)
|
|
117
|
+
when String then v.is_a?(String) && (v == test || (v.bytesize == test.bytesize && v.b == test.b))
|
|
118
|
+
else
|
|
119
|
+
if test.respond_to?(:call) then test.call(v)
|
|
120
|
+
else v == test
|
|
121
|
+
end
|
|
122
|
+
end
|
|
123
|
+
end
|
|
124
|
+
|
|
125
|
+
# Whether a set of values with the given bounds may contain a match. Unknown bounds or
|
|
126
|
+
# values that cannot be compared never rule anything out.
|
|
127
|
+
def self.may_match?(test, min, max, nulls, all_null)
|
|
128
|
+
case test
|
|
129
|
+
when nil then nulls.nil? || nulls.positive?
|
|
130
|
+
when Array then test.any? { |t| may_match?(t, min, max, nulls, all_null) }
|
|
131
|
+
when Range
|
|
132
|
+
return false if all_null
|
|
133
|
+
lo = test.begin
|
|
134
|
+
hi = test.end
|
|
135
|
+
if !hi.nil? && !min.nil?
|
|
136
|
+
c = compare(min, hi)
|
|
137
|
+
return false if c && (c.positive? || (test.exclude_end? && c.zero?))
|
|
138
|
+
end
|
|
139
|
+
if !lo.nil? && !max.nil?
|
|
140
|
+
c = compare(max, lo)
|
|
141
|
+
return false if c&.negative?
|
|
142
|
+
end
|
|
143
|
+
true
|
|
144
|
+
else
|
|
145
|
+
return true if test.respond_to?(:call)
|
|
146
|
+
return false if all_null
|
|
147
|
+
lo = min.nil? ? nil : compare(test, min)
|
|
148
|
+
hi = max.nil? ? nil : compare(test, max)
|
|
149
|
+
!(lo&.negative? || hi&.positive?)
|
|
150
|
+
end
|
|
151
|
+
end
|
|
152
|
+
|
|
153
|
+
BOOLEAN_ORDER = { false => 0, true => 1 }.freeze
|
|
154
|
+
|
|
155
|
+
def self.compare(a, b)
|
|
156
|
+
a = a.b if a.is_a?(String)
|
|
157
|
+
b = b.b if b.is_a?(String)
|
|
158
|
+
a = BOOLEAN_ORDER.fetch(a, a)
|
|
159
|
+
b = BOOLEAN_ORDER.fetch(b, b)
|
|
160
|
+
a <=> b
|
|
161
|
+
rescue StandardError
|
|
162
|
+
nil
|
|
163
|
+
end
|
|
164
|
+
|
|
165
|
+
# [min, max] of a chunk's statistics as Ruby values, or [nil, nil] when unusable
|
|
166
|
+
def self.stat_range(column, stats)
|
|
167
|
+
return [nil, nil] unless stats
|
|
168
|
+
if stats.min_value && stats.max_value
|
|
169
|
+
[decode_stat(column, stats.min_value), decode_stat(column, stats.max_value)]
|
|
170
|
+
elsif stats.min && stats.max && legacy_order_ok?(column)
|
|
171
|
+
[decode_stat(column, stats.min), decode_stat(column, stats.max)]
|
|
172
|
+
else
|
|
173
|
+
[nil, nil]
|
|
174
|
+
end
|
|
175
|
+
end
|
|
176
|
+
|
|
177
|
+
# The deprecated min/max fields were written with signed comparisons, which only agree
|
|
178
|
+
# with the logical order for signed numbers and booleans
|
|
179
|
+
def self.legacy_order_ok?(column)
|
|
180
|
+
kind, _, signed = Types.logical_of(column.node)
|
|
181
|
+
case column.type
|
|
182
|
+
when Format::Type::BOOLEAN, Format::Type::FLOAT, Format::Type::DOUBLE then true
|
|
183
|
+
when Format::Type::INT32, Format::Type::INT64 then !(kind == :integer && !signed)
|
|
184
|
+
else false
|
|
185
|
+
end
|
|
186
|
+
end
|
|
187
|
+
|
|
188
|
+
def self.decode_stat(column, bytes)
|
|
189
|
+
return nil if bytes.nil?
|
|
190
|
+
type = column.type
|
|
191
|
+
value = case type
|
|
192
|
+
when Format::Type::INT96 then return nil # no defined order
|
|
193
|
+
when Format::Type::BYTE_ARRAY, Format::Type::FIXED_LEN_BYTE_ARRAY
|
|
194
|
+
# Truncated bounds (shorter than the type length) still bound byte-wise
|
|
195
|
+
return type == Format::Type::FIXED_LEN_BYTE_ARRAY && bytes.bytesize != column.type_length ? nil : convert(column, bytes.dup)
|
|
196
|
+
when Format::Type::BOOLEAN then bytes.getbyte(0) == 1
|
|
197
|
+
else
|
|
198
|
+
return nil if bytes.bytesize < { Format::Type::INT32 => 4, Format::Type::FLOAT => 4 }.fetch(type, 8)
|
|
199
|
+
Encodings::Plain.decode(bytes, 0, 1, type).first.first
|
|
200
|
+
end
|
|
201
|
+
convert(column, value)
|
|
202
|
+
rescue StandardError
|
|
203
|
+
nil
|
|
204
|
+
end
|
|
205
|
+
|
|
206
|
+
def self.convert(column, value)
|
|
207
|
+
conv = column.converter
|
|
208
|
+
conv ? conv.call(value) : value
|
|
209
|
+
end
|
|
210
|
+
|
|
211
|
+
def self.merge(ranges)
|
|
212
|
+
ranges.sort.each_with_object([]) do |(s, e), out|
|
|
213
|
+
if out.any? && s <= out.last[1]
|
|
214
|
+
out.last[1] = e if e > out.last[1]
|
|
215
|
+
else
|
|
216
|
+
out << [s, e]
|
|
217
|
+
end
|
|
218
|
+
end
|
|
219
|
+
end
|
|
220
|
+
|
|
221
|
+
def self.intersect(a, b)
|
|
222
|
+
out = []
|
|
223
|
+
i = j = 0
|
|
224
|
+
while i < a.size && j < b.size
|
|
225
|
+
s = [a[i][0], b[j][0]].max
|
|
226
|
+
e = [a[i][1], b[j][1]].min
|
|
227
|
+
out << [s, e] if s < e
|
|
228
|
+
a[i][1] < b[j][1] ? i += 1 : j += 1
|
|
229
|
+
end
|
|
230
|
+
out
|
|
231
|
+
end
|
|
232
|
+
|
|
233
|
+
private
|
|
234
|
+
|
|
235
|
+
def normalize(test)
|
|
236
|
+
case test
|
|
237
|
+
when Symbol then test.to_s
|
|
238
|
+
when Array then test.map { |t| normalize(t) }
|
|
239
|
+
else test
|
|
240
|
+
end
|
|
241
|
+
end
|
|
242
|
+
|
|
243
|
+
# Bloom filters answer equality lookups (a value or a list of values)
|
|
244
|
+
def bloom_may_match?(reader, rg_index, condition)
|
|
245
|
+
values = Array(condition.test.is_a?(Array) ? condition.test : [condition.test])
|
|
246
|
+
return true if values.empty? || values.any? { |v| v.nil? || v.is_a?(Range) || v.respond_to?(:call) }
|
|
247
|
+
return true unless reader.respond_to?(:bloom_filter)
|
|
248
|
+
filter = reader.bloom_filter(rg_index, condition.column.path)
|
|
249
|
+
return true unless filter
|
|
250
|
+
values.any? { |v| filter.might_contain?(v) }
|
|
251
|
+
rescue StandardError
|
|
252
|
+
true # values the column cannot store never rule a row group out here; rows are checked anyway
|
|
253
|
+
end
|
|
254
|
+
end
|
|
255
|
+
end
|
|
256
|
+
end
|