herringbone 0.1.0 → 0.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/README.md +198 -74
- data/bin/herringbone +38 -26
- data/lib/herringbone/active_record.rb +48 -19
- data/lib/herringbone/bloom_filter.rb +370 -0
- data/lib/herringbone/byte_values.rb +29 -0
- data/lib/herringbone/codecs/lz4.rb +66 -3
- data/lib/herringbone/codecs/snappy.rb +132 -3
- data/lib/herringbone/compression.rb +121 -16
- data/lib/herringbone/encodings/delta.rb +76 -19
- data/lib/herringbone/encodings/plain.rb +29 -7
- data/lib/herringbone/encodings/rle.rb +55 -3
- data/lib/herringbone/format.rb +203 -0
- data/lib/herringbone/inspector.rb +1784 -0
- data/lib/herringbone/io_buffer_support.rb +3 -1
- data/lib/herringbone/reader/column_chunk_reader.rb +388 -0
- data/lib/herringbone/reader/column_cursor.rb +225 -0
- data/lib/herringbone/reader/numo.rb +365 -0
- data/lib/herringbone/reader/page_stream.rb +615 -0
- data/lib/herringbone/reader/scan.rb +350 -0
- data/lib/herringbone/reader.rb +585 -272
- data/lib/herringbone/schema.rb +326 -90
- data/lib/herringbone/thrift.rb +155 -4
- data/lib/herringbone/types.rb +177 -42
- data/lib/herringbone/version.rb +2 -1
- data/lib/herringbone/visualizer.rb +1136 -0
- data/lib/herringbone/writer.rb +550 -105
- data/lib/herringbone/xxhash.rb +425 -0
- data/lib/herringbone.rb +64 -9
- metadata +14 -33
|
@@ -0,0 +1,350 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Herringbone
|
|
4
|
+
class Reader
|
|
5
|
+
# Row selection for where: filters. A filter maps columns to conditions:
|
|
6
|
+
#
|
|
7
|
+
# where: { "user_id" => 42, # equality
|
|
8
|
+
# "status" => %w[paid shipped], # any of these (IN)
|
|
9
|
+
# "created_at" => t1...t2, # Ranges, also endless / beginless
|
|
10
|
+
# "address.city" => "Amsterdam", # a member of a struct, by dotted path
|
|
11
|
+
# "deleted_at" => nil, # IS NULL
|
|
12
|
+
# "amount" => ->(v) { v && v > 100 } } # any callable (row check only)
|
|
13
|
+
#
|
|
14
|
+
# All conditions must hold. The filter is used three ways, from cheapest to most exact:
|
|
15
|
+
# whole row groups are ruled out with column chunk statistics (min/max, null counts) and
|
|
16
|
+
# bloom filters; within a row group, pages are ruled out with the page index (ColumnIndex +
|
|
17
|
+
# OffsetIndex), leaving ranges of rows to read; and every row that is read is checked, so
|
|
18
|
+
# results are exact.
|
|
19
|
+
class Filter
|
|
20
|
+
# One where: entry, resolved against the schema
|
|
21
|
+
#
|
|
22
|
+
# @!attribute [rw] name
|
|
23
|
+
# @return [String] the key as given (dotted path of a leaf column)
|
|
24
|
+
# @!attribute [rw] column
|
|
25
|
+
# @return [Schema::Column] the leaf column the condition reads
|
|
26
|
+
# @!attribute [rw] field
|
|
27
|
+
# @return [Schema::Field] the top-level field holding that column
|
|
28
|
+
# @!attribute [rw] path
|
|
29
|
+
# @return [Array<String>] member names from the top-level field down to the column
|
|
30
|
+
# (empty for a top-level column)
|
|
31
|
+
# @!attribute [rw] test
|
|
32
|
+
# @return [Object] the condition, with Symbols turned into Strings
|
|
33
|
+
Condition = Struct.new(:name, :column, :field, :path, :test)
|
|
34
|
+
|
|
35
|
+
# @return [Array<Condition>] one per where: entry, in the order given
|
|
36
|
+
attr_reader :conditions
|
|
37
|
+
|
|
38
|
+
# @param schema [Schema] schema of the file being read
|
|
39
|
+
# @param where [Hash{String, Symbol => Object}] column => condition (see the class docs)
|
|
40
|
+
# @raise [ArgumentError] when +where+ is not a Hash, names an unknown column, or a column
|
|
41
|
+
# inside a list or map
|
|
42
|
+
def initialize(schema, where)
|
|
43
|
+
raise ArgumentError, "where: must be a Hash of column => condition" unless where.is_a?(Hash)
|
|
44
|
+
@conditions = where.map do |key, test|
|
|
45
|
+
name = key.to_s
|
|
46
|
+
column = schema.column(name) or raise ArgumentError, "where: no such column #{name.inspect}"
|
|
47
|
+
if column.max_repetition_level.positive?
|
|
48
|
+
raise ArgumentError, "where: #{name} is inside a list or map; only non-repeated columns can be filtered on"
|
|
49
|
+
end
|
|
50
|
+
field = schema.field(column.path.first)
|
|
51
|
+
Condition.new(name, column, field, column.path.drop(1), normalize(test))
|
|
52
|
+
end
|
|
53
|
+
end
|
|
54
|
+
|
|
55
|
+
# Top-level fields the conditions need to read
|
|
56
|
+
#
|
|
57
|
+
# @return [Array<Schema::Field>] distinct fields, in condition order
|
|
58
|
+
def fields
|
|
59
|
+
@conditions.map(&:field).uniq
|
|
60
|
+
end
|
|
61
|
+
|
|
62
|
+
# --- statistics-based pruning ---
|
|
63
|
+
|
|
64
|
+
# Whether any row of row group +rg_index+ may match, judging by chunk statistics and bloom
|
|
65
|
+
# filters
|
|
66
|
+
#
|
|
67
|
+
# @param reader [Reader] reader of the file (for its footer and bloom filters)
|
|
68
|
+
# @param rg_index [Integer] position of the row group in the footer
|
|
69
|
+
# @return [Boolean] false only when no row of the row group can match
|
|
70
|
+
def row_group_may_match?(reader, rg_index)
|
|
71
|
+
chunks = reader.row_groups[rg_index].columns
|
|
72
|
+
@conditions.all? do |c|
|
|
73
|
+
meta = chunks[c.column.index].meta_data
|
|
74
|
+
next true unless meta
|
|
75
|
+
min, max = Filter.stat_range(c.column, meta.statistics)
|
|
76
|
+
nulls = meta.statistics&.null_count
|
|
77
|
+
all_null = nulls && nulls == meta.num_values
|
|
78
|
+
Filter.may_match?(c.test, min, max, nulls, all_null) && bloom_may_match?(reader, rg_index, c)
|
|
79
|
+
end
|
|
80
|
+
end
|
|
81
|
+
|
|
82
|
+
# Sorted, non-overlapping [first_row, end_row) ranges of row group +rg_index+ that may hold
|
|
83
|
+
# matching rows, judging by the page index. Columns without a page index do not narrow
|
|
84
|
+
# the ranges.
|
|
85
|
+
#
|
|
86
|
+
# @param reader [Reader] reader of the file (for its page index)
|
|
87
|
+
# @param rg_index [Integer] position of the row group in the footer
|
|
88
|
+
# @return [Array<Array(Integer, Integer)>] half-open row ranges within the row group
|
|
89
|
+
def page_ranges(reader, rg_index)
|
|
90
|
+
n = reader.row_groups[rg_index].num_rows
|
|
91
|
+
ranges = [[0, n]]
|
|
92
|
+
@conditions.each do |c|
|
|
93
|
+
column_index, offset_index = reader.page_index(rg_index, c.column)
|
|
94
|
+
next unless column_index && offset_index
|
|
95
|
+
locs = offset_index.page_locations
|
|
96
|
+
next unless locs && column_index.null_pages && locs.size == column_index.null_pages.size
|
|
97
|
+
candidates = []
|
|
98
|
+
locs.each_with_index do |loc, i|
|
|
99
|
+
min = max = nil
|
|
100
|
+
unless column_index.null_pages[i]
|
|
101
|
+
min = Filter.decode_stat(c.column, column_index.min_values[i])
|
|
102
|
+
max = Filter.decode_stat(c.column, column_index.max_values[i])
|
|
103
|
+
end
|
|
104
|
+
nulls = column_index.null_counts&.[](i)
|
|
105
|
+
next unless Filter.may_match?(c.test, min, max, nulls, column_index.null_pages[i])
|
|
106
|
+
stop = (i + 1 < locs.size) ? locs[i + 1].first_row_index : n
|
|
107
|
+
candidates << [loc.first_row_index, stop]
|
|
108
|
+
end
|
|
109
|
+
ranges = Filter.intersect(ranges, Filter.merge(candidates))
|
|
110
|
+
break if ranges.empty?
|
|
111
|
+
end
|
|
112
|
+
ranges
|
|
113
|
+
end
|
|
114
|
+
|
|
115
|
+
# --- row checks ---
|
|
116
|
+
|
|
117
|
+
# Indexes (0...k) of the rows of an assembled batch that match. +data+ holds one Array per
|
|
118
|
+
# field in +fields+; +symbolize+ says how struct Hashes are keyed.
|
|
119
|
+
#
|
|
120
|
+
# @param data [Array<Array>] assembled values, one Array of +k+ entries per field
|
|
121
|
+
# @param fields [Array<Schema::Field>] the fields +data+ holds, in the same order
|
|
122
|
+
# @param k [Integer] number of rows in the batch
|
|
123
|
+
# @param symbolize [Boolean] whether struct Hashes are keyed by Symbol
|
|
124
|
+
# @return [Array<Integer>] indexes of matching rows, ascending
|
|
125
|
+
def matching_rows(data, fields, k, symbolize)
|
|
126
|
+
columns = @conditions.map do |c|
|
|
127
|
+
values = data.fetch(fields.index(c.field))
|
|
128
|
+
next values if c.path.empty?
|
|
129
|
+
path = symbolize ? c.path.map(&:to_sym) : c.path
|
|
130
|
+
values.map { |v| path.reduce(v) { |h, key| h&.[](key) } }
|
|
131
|
+
end
|
|
132
|
+
tests = @conditions.map(&:test)
|
|
133
|
+
(0...k).select do |i|
|
|
134
|
+
j = 0
|
|
135
|
+
ok = true
|
|
136
|
+
while ok && j < tests.size
|
|
137
|
+
ok = Filter.matches?(tests[j], columns[j][i])
|
|
138
|
+
j += 1
|
|
139
|
+
end
|
|
140
|
+
ok
|
|
141
|
+
end
|
|
142
|
+
end
|
|
143
|
+
|
|
144
|
+
# --- condition evaluation ---
|
|
145
|
+
|
|
146
|
+
# Whether a value satisfies a condition: nil matches nulls, an Array any of its elements,
|
|
147
|
+
# a Range by #cover? (never nil, incomparable values do not match), a String by bytes
|
|
148
|
+
# (ignoring the encoding), a callable by its result, anything else by ==.
|
|
149
|
+
#
|
|
150
|
+
# @param test [Object] the condition
|
|
151
|
+
# @param v [Object] the row's value
|
|
152
|
+
# @return [Boolean] true when the value matches
|
|
153
|
+
def self.matches?(test, v)
|
|
154
|
+
case test
|
|
155
|
+
when nil then v.nil?
|
|
156
|
+
when Array then test.any? { |t| matches?(t, v) }
|
|
157
|
+
when Range then !v.nil? && begin
|
|
158
|
+
test.cover?(v)
|
|
159
|
+
rescue
|
|
160
|
+
false
|
|
161
|
+
end
|
|
162
|
+
when String then v.is_a?(String) && (v == test || (v.bytesize == test.bytesize && v.b == test.b))
|
|
163
|
+
else
|
|
164
|
+
if test.respond_to?(:call) then test.call(v)
|
|
165
|
+
else v == test
|
|
166
|
+
end
|
|
167
|
+
end
|
|
168
|
+
end
|
|
169
|
+
|
|
170
|
+
# Whether a set of values with the given bounds may contain a match. Unknown bounds or
|
|
171
|
+
# values that cannot be compared never rule anything out.
|
|
172
|
+
#
|
|
173
|
+
# @param test [Object] the condition
|
|
174
|
+
# @param min [Object, nil] lower bound of the values, nil when unknown
|
|
175
|
+
# @param max [Object, nil] upper bound of the values, nil when unknown
|
|
176
|
+
# @param nulls [Integer, nil] number of nulls, nil when unknown
|
|
177
|
+
# @param all_null [Boolean, nil] whether every value is null
|
|
178
|
+
# @return [Boolean] false only when no value can match
|
|
179
|
+
def self.may_match?(test, min, max, nulls, all_null)
|
|
180
|
+
case test
|
|
181
|
+
when nil then nulls.nil? || nulls.positive?
|
|
182
|
+
when Array then test.any? { |t| may_match?(t, min, max, nulls, all_null) }
|
|
183
|
+
when Range
|
|
184
|
+
return false if all_null
|
|
185
|
+
lo = test.begin
|
|
186
|
+
hi = test.end
|
|
187
|
+
if !hi.nil? && !min.nil?
|
|
188
|
+
c = compare(min, hi)
|
|
189
|
+
return false if c && (c.positive? || (test.exclude_end? && c.zero?))
|
|
190
|
+
end
|
|
191
|
+
if !lo.nil? && !max.nil?
|
|
192
|
+
c = compare(max, lo)
|
|
193
|
+
return false if c&.negative?
|
|
194
|
+
end
|
|
195
|
+
true
|
|
196
|
+
else
|
|
197
|
+
return true if test.respond_to?(:call)
|
|
198
|
+
return false if all_null
|
|
199
|
+
lo = min.nil? ? nil : compare(test, min)
|
|
200
|
+
hi = max.nil? ? nil : compare(test, max)
|
|
201
|
+
!(lo&.negative? || hi&.positive?)
|
|
202
|
+
end
|
|
203
|
+
end
|
|
204
|
+
|
|
205
|
+
# Booleans compare as integers, false before true (the Parquet sort order)
|
|
206
|
+
BOOLEAN_ORDER = {false => 0, true => 1}.freeze
|
|
207
|
+
|
|
208
|
+
# Compares two values in Parquet order: Strings byte-wise, false before true
|
|
209
|
+
#
|
|
210
|
+
# @param a [Object] left-hand value
|
|
211
|
+
# @param b [Object] right-hand value
|
|
212
|
+
# @return [Integer, nil] -1, 0 or 1, or nil when the values cannot be compared
|
|
213
|
+
def self.compare(a, b)
|
|
214
|
+
a = a.b if a.is_a?(String)
|
|
215
|
+
b = b.b if b.is_a?(String)
|
|
216
|
+
a = BOOLEAN_ORDER.fetch(a, a)
|
|
217
|
+
b = BOOLEAN_ORDER.fetch(b, b)
|
|
218
|
+
a <=> b
|
|
219
|
+
rescue
|
|
220
|
+
nil
|
|
221
|
+
end
|
|
222
|
+
|
|
223
|
+
# [min, max] of a chunk's statistics as Ruby values, or [nil, nil] when unusable
|
|
224
|
+
#
|
|
225
|
+
# @param column [Schema::Column] the chunk's column, for decoding the bounds
|
|
226
|
+
# @param stats [Format::Statistics, nil] the chunk's statistics
|
|
227
|
+
# @return [Array(Object, Object)] min and max, each nil when unknown
|
|
228
|
+
def self.stat_range(column, stats)
|
|
229
|
+
return [nil, nil] unless stats
|
|
230
|
+
if stats.min_value && stats.max_value
|
|
231
|
+
[decode_stat(column, stats.min_value), decode_stat(column, stats.max_value)]
|
|
232
|
+
elsif stats.min && stats.max && legacy_order_ok?(column)
|
|
233
|
+
[decode_stat(column, stats.min), decode_stat(column, stats.max)]
|
|
234
|
+
else
|
|
235
|
+
[nil, nil]
|
|
236
|
+
end
|
|
237
|
+
end
|
|
238
|
+
|
|
239
|
+
# The deprecated min/max fields were written with signed comparisons, which only agree
|
|
240
|
+
# with the logical order for signed numbers and booleans
|
|
241
|
+
#
|
|
242
|
+
# @param column [Schema::Column] column whose statistics are being read
|
|
243
|
+
# @return [Boolean] true when the legacy min/max fields can be trusted
|
|
244
|
+
def self.legacy_order_ok?(column)
|
|
245
|
+
kind, _, signed = Types.logical_of(column.node)
|
|
246
|
+
case column.type
|
|
247
|
+
when Format::Type::BOOLEAN, Format::Type::FLOAT, Format::Type::DOUBLE then true
|
|
248
|
+
when Format::Type::INT32, Format::Type::INT64 then !(kind == :integer && !signed)
|
|
249
|
+
else false
|
|
250
|
+
end
|
|
251
|
+
end
|
|
252
|
+
|
|
253
|
+
# Decodes a min/max bound (PLAIN encoding of one value) into a Ruby value
|
|
254
|
+
#
|
|
255
|
+
# @param column [Schema::Column] the column the bound belongs to
|
|
256
|
+
# @param bytes [String, nil] the encoded bound
|
|
257
|
+
# @return [Object, nil] the value, or nil when it is absent or cannot be used (INT96,
|
|
258
|
+
# truncated FIXED_LEN_BYTE_ARRAY, too short, undecodable)
|
|
259
|
+
def self.decode_stat(column, bytes)
|
|
260
|
+
return nil if bytes.nil?
|
|
261
|
+
type = column.type
|
|
262
|
+
value = case type
|
|
263
|
+
when Format::Type::INT96 then return nil # no defined order
|
|
264
|
+
when Format::Type::BYTE_ARRAY, Format::Type::FIXED_LEN_BYTE_ARRAY
|
|
265
|
+
# Truncated bounds (shorter than the type length) still bound byte-wise
|
|
266
|
+
return (type == Format::Type::FIXED_LEN_BYTE_ARRAY && bytes.bytesize != column.type_length) ? nil : convert(column, bytes.dup)
|
|
267
|
+
when Format::Type::BOOLEAN then bytes.getbyte(0) == 1
|
|
268
|
+
else
|
|
269
|
+
return nil if bytes.bytesize < {Format::Type::INT32 => 4, Format::Type::FLOAT => 4}.fetch(type, 8)
|
|
270
|
+
Encodings::Plain.decode(bytes, 0, 1, type).first.first
|
|
271
|
+
end
|
|
272
|
+
convert(column, value)
|
|
273
|
+
rescue
|
|
274
|
+
nil
|
|
275
|
+
end
|
|
276
|
+
|
|
277
|
+
# Applies the column's converter, so bounds compare with the values rows hold
|
|
278
|
+
#
|
|
279
|
+
# @param column [Schema::Column] the column the value belongs to
|
|
280
|
+
# @param value [Object] physical value
|
|
281
|
+
# @return [Object] the Ruby value
|
|
282
|
+
def self.convert(column, value)
|
|
283
|
+
conv = column.converter
|
|
284
|
+
conv ? conv.call(value) : value
|
|
285
|
+
end
|
|
286
|
+
|
|
287
|
+
# Sorts ranges and merges overlapping or touching ones
|
|
288
|
+
#
|
|
289
|
+
# @param ranges [Array<Array(Integer, Integer)>] half-open row ranges
|
|
290
|
+
# @return [Array<Array(Integer, Integer)>] sorted, non-overlapping ranges
|
|
291
|
+
def self.merge(ranges)
|
|
292
|
+
ranges.sort.each_with_object([]) do |(s, e), out|
|
|
293
|
+
if out.any? && s <= out.last[1]
|
|
294
|
+
out.last[1] = e if e > out.last[1]
|
|
295
|
+
else
|
|
296
|
+
out << [s, e]
|
|
297
|
+
end
|
|
298
|
+
end
|
|
299
|
+
end
|
|
300
|
+
|
|
301
|
+
# Intersection of two sorted, non-overlapping lists of half-open ranges
|
|
302
|
+
#
|
|
303
|
+
# @param a [Array<Array(Integer, Integer)>] first list of ranges
|
|
304
|
+
# @param b [Array<Array(Integer, Integer)>] second list of ranges
|
|
305
|
+
# @return [Array<Array(Integer, Integer)>] the rows in both, sorted
|
|
306
|
+
def self.intersect(a, b)
|
|
307
|
+
out = []
|
|
308
|
+
i = j = 0
|
|
309
|
+
while i < a.size && j < b.size
|
|
310
|
+
s = [a[i][0], b[j][0]].max
|
|
311
|
+
e = [a[i][1], b[j][1]].min
|
|
312
|
+
out << [s, e] if s < e
|
|
313
|
+
(a[i][1] < b[j][1]) ? i += 1 : j += 1
|
|
314
|
+
end
|
|
315
|
+
out
|
|
316
|
+
end
|
|
317
|
+
|
|
318
|
+
private
|
|
319
|
+
|
|
320
|
+
# Symbols (also inside Arrays) become Strings, since columns never hold Symbols
|
|
321
|
+
#
|
|
322
|
+
# @param test [Object] a condition as given in where:
|
|
323
|
+
# @return [Object] the condition to store in Condition#test
|
|
324
|
+
def normalize(test)
|
|
325
|
+
case test
|
|
326
|
+
when Symbol then test.to_s
|
|
327
|
+
when Array then test.map { |t| normalize(t) }
|
|
328
|
+
else test
|
|
329
|
+
end
|
|
330
|
+
end
|
|
331
|
+
|
|
332
|
+
# Bloom filters answer equality lookups (a value or a list of values)
|
|
333
|
+
#
|
|
334
|
+
# @param reader [Reader] reader of the file (for its bloom filters)
|
|
335
|
+
# @param rg_index [Integer] position of the row group in the footer
|
|
336
|
+
# @param condition [Condition] the condition to check
|
|
337
|
+
# @return [Boolean] false only when the bloom filter rules out every value of the condition
|
|
338
|
+
def bloom_may_match?(reader, rg_index, condition)
|
|
339
|
+
values = Array(condition.test.is_a?(Array) ? condition.test : [condition.test])
|
|
340
|
+
return true if values.empty? || values.any? { |v| v.nil? || v.is_a?(Range) || v.respond_to?(:call) }
|
|
341
|
+
return true unless reader.respond_to?(:bloom_filter)
|
|
342
|
+
filter = reader.bloom_filter(rg_index, condition.column.path)
|
|
343
|
+
return true unless filter
|
|
344
|
+
values.any? { |v| filter.might_contain?(v) }
|
|
345
|
+
rescue
|
|
346
|
+
true # values the column cannot store never rule a row group out here; rows are checked anyway
|
|
347
|
+
end
|
|
348
|
+
end
|
|
349
|
+
end
|
|
350
|
+
end
|