herringbone 0.1.0 → 0.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,350 @@
1
+ # frozen_string_literal: true
2
+
3
+ module Herringbone
4
+ class Reader
5
+ # Row selection for where: filters. A filter maps columns to conditions:
6
+ #
7
+ # where: { "user_id" => 42, # equality
8
+ # "status" => %w[paid shipped], # any of these (IN)
9
+ # "created_at" => t1...t2, # Ranges, also endless / beginless
10
+ # "address.city" => "Amsterdam", # a member of a struct, by dotted path
11
+ # "deleted_at" => nil, # IS NULL
12
+ # "amount" => ->(v) { v && v > 100 } } # any callable (row check only)
13
+ #
14
+ # All conditions must hold. The filter is used three ways, from cheapest to most exact:
15
+ # whole row groups are ruled out with column chunk statistics (min/max, null counts) and
16
+ # bloom filters; within a row group, pages are ruled out with the page index (ColumnIndex +
17
+ # OffsetIndex), leaving ranges of rows to read; and every row that is read is checked, so
18
+ # results are exact.
19
+ class Filter
20
+ # One where: entry, resolved against the schema
21
+ #
22
+ # @!attribute [rw] name
23
+ # @return [String] the key as given (dotted path of a leaf column)
24
+ # @!attribute [rw] column
25
+ # @return [Schema::Column] the leaf column the condition reads
26
+ # @!attribute [rw] field
27
+ # @return [Schema::Field] the top-level field holding that column
28
+ # @!attribute [rw] path
29
+ # @return [Array<String>] member names from the top-level field down to the column
30
+ # (empty for a top-level column)
31
+ # @!attribute [rw] test
32
+ # @return [Object] the condition, with Symbols turned into Strings
33
+ Condition = Struct.new(:name, :column, :field, :path, :test)
34
+
35
+ # @return [Array<Condition>] one per where: entry, in the order given
36
+ attr_reader :conditions
37
+
38
+ # @param schema [Schema] schema of the file being read
39
+ # @param where [Hash{String, Symbol => Object}] column => condition (see the class docs)
40
+ # @raise [ArgumentError] when +where+ is not a Hash, names an unknown column, or a column
41
+ # inside a list or map
42
+ def initialize(schema, where)
43
+ raise ArgumentError, "where: must be a Hash of column => condition" unless where.is_a?(Hash)
44
+ @conditions = where.map do |key, test|
45
+ name = key.to_s
46
+ column = schema.column(name) or raise ArgumentError, "where: no such column #{name.inspect}"
47
+ if column.max_repetition_level.positive?
48
+ raise ArgumentError, "where: #{name} is inside a list or map; only non-repeated columns can be filtered on"
49
+ end
50
+ field = schema.field(column.path.first)
51
+ Condition.new(name, column, field, column.path.drop(1), normalize(test))
52
+ end
53
+ end
54
+
55
+ # Top-level fields the conditions need to read
56
+ #
57
+ # @return [Array<Schema::Field>] distinct fields, in condition order
58
+ def fields
59
+ @conditions.map(&:field).uniq
60
+ end
61
+
62
+ # --- statistics-based pruning ---
63
+
64
+ # Whether any row of row group +rg_index+ may match, judging by chunk statistics and bloom
65
+ # filters
66
+ #
67
+ # @param reader [Reader] reader of the file (for its footer and bloom filters)
68
+ # @param rg_index [Integer] position of the row group in the footer
69
+ # @return [Boolean] false only when no row of the row group can match
70
+ def row_group_may_match?(reader, rg_index)
71
+ chunks = reader.row_groups[rg_index].columns
72
+ @conditions.all? do |c|
73
+ meta = chunks[c.column.index].meta_data
74
+ next true unless meta
75
+ min, max = Filter.stat_range(c.column, meta.statistics)
76
+ nulls = meta.statistics&.null_count
77
+ all_null = nulls && nulls == meta.num_values
78
+ Filter.may_match?(c.test, min, max, nulls, all_null) && bloom_may_match?(reader, rg_index, c)
79
+ end
80
+ end
81
+
82
+ # Sorted, non-overlapping [first_row, end_row) ranges of row group +rg_index+ that may hold
83
+ # matching rows, judging by the page index. Columns without a page index do not narrow
84
+ # the ranges.
85
+ #
86
+ # @param reader [Reader] reader of the file (for its page index)
87
+ # @param rg_index [Integer] position of the row group in the footer
88
+ # @return [Array<Array(Integer, Integer)>] half-open row ranges within the row group
89
+ def page_ranges(reader, rg_index)
90
+ n = reader.row_groups[rg_index].num_rows
91
+ ranges = [[0, n]]
92
+ @conditions.each do |c|
93
+ column_index, offset_index = reader.page_index(rg_index, c.column)
94
+ next unless column_index && offset_index
95
+ locs = offset_index.page_locations
96
+ next unless locs && column_index.null_pages && locs.size == column_index.null_pages.size
97
+ candidates = []
98
+ locs.each_with_index do |loc, i|
99
+ min = max = nil
100
+ unless column_index.null_pages[i]
101
+ min = Filter.decode_stat(c.column, column_index.min_values[i])
102
+ max = Filter.decode_stat(c.column, column_index.max_values[i])
103
+ end
104
+ nulls = column_index.null_counts&.[](i)
105
+ next unless Filter.may_match?(c.test, min, max, nulls, column_index.null_pages[i])
106
+ stop = (i + 1 < locs.size) ? locs[i + 1].first_row_index : n
107
+ candidates << [loc.first_row_index, stop]
108
+ end
109
+ ranges = Filter.intersect(ranges, Filter.merge(candidates))
110
+ break if ranges.empty?
111
+ end
112
+ ranges
113
+ end
114
+
115
+ # --- row checks ---
116
+
117
+ # Indexes (0...k) of the rows of an assembled batch that match. +data+ holds one Array per
118
+ # field in +fields+; +symbolize+ says how struct Hashes are keyed.
119
+ #
120
+ # @param data [Array<Array>] assembled values, one Array of +k+ entries per field
121
+ # @param fields [Array<Schema::Field>] the fields +data+ holds, in the same order
122
+ # @param k [Integer] number of rows in the batch
123
+ # @param symbolize [Boolean] whether struct Hashes are keyed by Symbol
124
+ # @return [Array<Integer>] indexes of matching rows, ascending
125
+ def matching_rows(data, fields, k, symbolize)
126
+ columns = @conditions.map do |c|
127
+ values = data.fetch(fields.index(c.field))
128
+ next values if c.path.empty?
129
+ path = symbolize ? c.path.map(&:to_sym) : c.path
130
+ values.map { |v| path.reduce(v) { |h, key| h&.[](key) } }
131
+ end
132
+ tests = @conditions.map(&:test)
133
+ (0...k).select do |i|
134
+ j = 0
135
+ ok = true
136
+ while ok && j < tests.size
137
+ ok = Filter.matches?(tests[j], columns[j][i])
138
+ j += 1
139
+ end
140
+ ok
141
+ end
142
+ end
143
+
144
+ # --- condition evaluation ---
145
+
146
+ # Whether a value satisfies a condition: nil matches nulls, an Array any of its elements,
147
+ # a Range by #cover? (never nil, incomparable values do not match), a String by bytes
148
+ # (ignoring the encoding), a callable by its result, anything else by ==.
149
+ #
150
+ # @param test [Object] the condition
151
+ # @param v [Object] the row's value
152
+ # @return [Boolean] true when the value matches
153
+ def self.matches?(test, v)
154
+ case test
155
+ when nil then v.nil?
156
+ when Array then test.any? { |t| matches?(t, v) }
157
+ when Range then !v.nil? && begin
158
+ test.cover?(v)
159
+ rescue
160
+ false
161
+ end
162
+ when String then v.is_a?(String) && (v == test || (v.bytesize == test.bytesize && v.b == test.b))
163
+ else
164
+ if test.respond_to?(:call) then test.call(v)
165
+ else v == test
166
+ end
167
+ end
168
+ end
169
+
170
+ # Whether a set of values with the given bounds may contain a match. Unknown bounds or
171
+ # values that cannot be compared never rule anything out.
172
+ #
173
+ # @param test [Object] the condition
174
+ # @param min [Object, nil] lower bound of the values, nil when unknown
175
+ # @param max [Object, nil] upper bound of the values, nil when unknown
176
+ # @param nulls [Integer, nil] number of nulls, nil when unknown
177
+ # @param all_null [Boolean, nil] whether every value is null
178
+ # @return [Boolean] false only when no value can match
179
+ def self.may_match?(test, min, max, nulls, all_null)
180
+ case test
181
+ when nil then nulls.nil? || nulls.positive?
182
+ when Array then test.any? { |t| may_match?(t, min, max, nulls, all_null) }
183
+ when Range
184
+ return false if all_null
185
+ lo = test.begin
186
+ hi = test.end
187
+ if !hi.nil? && !min.nil?
188
+ c = compare(min, hi)
189
+ return false if c && (c.positive? || (test.exclude_end? && c.zero?))
190
+ end
191
+ if !lo.nil? && !max.nil?
192
+ c = compare(max, lo)
193
+ return false if c&.negative?
194
+ end
195
+ true
196
+ else
197
+ return true if test.respond_to?(:call)
198
+ return false if all_null
199
+ lo = min.nil? ? nil : compare(test, min)
200
+ hi = max.nil? ? nil : compare(test, max)
201
+ !(lo&.negative? || hi&.positive?)
202
+ end
203
+ end
204
+
205
+ # Booleans compare as integers, false before true (the Parquet sort order)
206
+ BOOLEAN_ORDER = {false => 0, true => 1}.freeze
207
+
208
+ # Compares two values in Parquet order: Strings byte-wise, false before true
209
+ #
210
+ # @param a [Object] left-hand value
211
+ # @param b [Object] right-hand value
212
+ # @return [Integer, nil] -1, 0 or 1, or nil when the values cannot be compared
213
+ def self.compare(a, b)
214
+ a = a.b if a.is_a?(String)
215
+ b = b.b if b.is_a?(String)
216
+ a = BOOLEAN_ORDER.fetch(a, a)
217
+ b = BOOLEAN_ORDER.fetch(b, b)
218
+ a <=> b
219
+ rescue
220
+ nil
221
+ end
222
+
223
+ # [min, max] of a chunk's statistics as Ruby values, or [nil, nil] when unusable
224
+ #
225
+ # @param column [Schema::Column] the chunk's column, for decoding the bounds
226
+ # @param stats [Format::Statistics, nil] the chunk's statistics
227
+ # @return [Array(Object, Object)] min and max, each nil when unknown
228
+ def self.stat_range(column, stats)
229
+ return [nil, nil] unless stats
230
+ if stats.min_value && stats.max_value
231
+ [decode_stat(column, stats.min_value), decode_stat(column, stats.max_value)]
232
+ elsif stats.min && stats.max && legacy_order_ok?(column)
233
+ [decode_stat(column, stats.min), decode_stat(column, stats.max)]
234
+ else
235
+ [nil, nil]
236
+ end
237
+ end
238
+
239
+ # The deprecated min/max fields were written with signed comparisons, which only agree
240
+ # with the logical order for signed numbers and booleans
241
+ #
242
+ # @param column [Schema::Column] column whose statistics are being read
243
+ # @return [Boolean] true when the legacy min/max fields can be trusted
244
+ def self.legacy_order_ok?(column)
245
+ kind, _, signed = Types.logical_of(column.node)
246
+ case column.type
247
+ when Format::Type::BOOLEAN, Format::Type::FLOAT, Format::Type::DOUBLE then true
248
+ when Format::Type::INT32, Format::Type::INT64 then !(kind == :integer && !signed)
249
+ else false
250
+ end
251
+ end
252
+
253
+ # Decodes a min/max bound (PLAIN encoding of one value) into a Ruby value
254
+ #
255
+ # @param column [Schema::Column] the column the bound belongs to
256
+ # @param bytes [String, nil] the encoded bound
257
+ # @return [Object, nil] the value, or nil when it is absent or cannot be used (INT96,
258
+ # truncated FIXED_LEN_BYTE_ARRAY, too short, undecodable)
259
+ def self.decode_stat(column, bytes)
260
+ return nil if bytes.nil?
261
+ type = column.type
262
+ value = case type
263
+ when Format::Type::INT96 then return nil # no defined order
264
+ when Format::Type::BYTE_ARRAY, Format::Type::FIXED_LEN_BYTE_ARRAY
265
+ # Truncated bounds (shorter than the type length) still bound byte-wise
266
+ return (type == Format::Type::FIXED_LEN_BYTE_ARRAY && bytes.bytesize != column.type_length) ? nil : convert(column, bytes.dup)
267
+ when Format::Type::BOOLEAN then bytes.getbyte(0) == 1
268
+ else
269
+ return nil if bytes.bytesize < {Format::Type::INT32 => 4, Format::Type::FLOAT => 4}.fetch(type, 8)
270
+ Encodings::Plain.decode(bytes, 0, 1, type).first.first
271
+ end
272
+ convert(column, value)
273
+ rescue
274
+ nil
275
+ end
276
+
277
+ # Applies the column's converter, so bounds compare with the values rows hold
278
+ #
279
+ # @param column [Schema::Column] the column the value belongs to
280
+ # @param value [Object] physical value
281
+ # @return [Object] the Ruby value
282
+ def self.convert(column, value)
283
+ conv = column.converter
284
+ conv ? conv.call(value) : value
285
+ end
286
+
287
+ # Sorts ranges and merges overlapping or touching ones
288
+ #
289
+ # @param ranges [Array<Array(Integer, Integer)>] half-open row ranges
290
+ # @return [Array<Array(Integer, Integer)>] sorted, non-overlapping ranges
291
+ def self.merge(ranges)
292
+ ranges.sort.each_with_object([]) do |(s, e), out|
293
+ if out.any? && s <= out.last[1]
294
+ out.last[1] = e if e > out.last[1]
295
+ else
296
+ out << [s, e]
297
+ end
298
+ end
299
+ end
300
+
301
+ # Intersection of two sorted, non-overlapping lists of half-open ranges
302
+ #
303
+ # @param a [Array<Array(Integer, Integer)>] first list of ranges
304
+ # @param b [Array<Array(Integer, Integer)>] second list of ranges
305
+ # @return [Array<Array(Integer, Integer)>] the rows in both, sorted
306
+ def self.intersect(a, b)
307
+ out = []
308
+ i = j = 0
309
+ while i < a.size && j < b.size
310
+ s = [a[i][0], b[j][0]].max
311
+ e = [a[i][1], b[j][1]].min
312
+ out << [s, e] if s < e
313
+ (a[i][1] < b[j][1]) ? i += 1 : j += 1
314
+ end
315
+ out
316
+ end
317
+
318
+ private
319
+
320
+ # Symbols (also inside Arrays) become Strings, since columns never hold Symbols
321
+ #
322
+ # @param test [Object] a condition as given in where:
323
+ # @return [Object] the condition to store in Condition#test
324
+ def normalize(test)
325
+ case test
326
+ when Symbol then test.to_s
327
+ when Array then test.map { |t| normalize(t) }
328
+ else test
329
+ end
330
+ end
331
+
332
+ # Bloom filters answer equality lookups (a value or a list of values)
333
+ #
334
+ # @param reader [Reader] reader of the file (for its bloom filters)
335
+ # @param rg_index [Integer] position of the row group in the footer
336
+ # @param condition [Condition] the condition to check
337
+ # @return [Boolean] false only when the bloom filter rules out every value of the condition
338
+ def bloom_may_match?(reader, rg_index, condition)
339
+ values = Array(condition.test.is_a?(Array) ? condition.test : [condition.test])
340
+ return true if values.empty? || values.any? { |v| v.nil? || v.is_a?(Range) || v.respond_to?(:call) }
341
+ return true unless reader.respond_to?(:bloom_filter)
342
+ filter = reader.bloom_filter(rg_index, condition.column.path)
343
+ return true unless filter
344
+ values.any? { |v| filter.might_contain?(v) }
345
+ rescue
346
+ true # values the column cannot store never rule a row group out here; rows are checked anyway
347
+ end
348
+ end
349
+ end
350
+ end