herringbone 0.2.0 → 0.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/README.md +56 -4
- data/lib/herringbone/active_record.rb +44 -12
- data/lib/herringbone/bloom_filter.rb +112 -12
- data/lib/herringbone/byte_values.rb +29 -0
- data/lib/herringbone/codecs/lz4.rb +66 -3
- data/lib/herringbone/codecs/snappy.rb +132 -3
- data/lib/herringbone/compression.rb +48 -0
- data/lib/herringbone/encodings/delta.rb +70 -13
- data/lib/herringbone/encodings/plain.rb +22 -0
- data/lib/herringbone/encodings/rle.rb +53 -1
- data/lib/herringbone/format.rb +150 -0
- data/lib/herringbone/inspector.rb +477 -81
- data/lib/herringbone/io_buffer_support.rb +3 -1
- data/lib/herringbone/reader/column_chunk_reader.rb +98 -6
- data/lib/herringbone/reader/column_cursor.rb +46 -2
- data/lib/herringbone/reader/numo.rb +365 -0
- data/lib/herringbone/reader/page_stream.rb +280 -4
- data/lib/herringbone/reader/scan.rb +103 -9
- data/lib/herringbone/reader.rb +308 -17
- data/lib/herringbone/schema.rb +306 -15
- data/lib/herringbone/thrift.rb +155 -4
- data/lib/herringbone/types.rb +176 -41
- data/lib/herringbone/version.rb +2 -1
- data/lib/herringbone/visualizer.rb +51 -11
- data/lib/herringbone/writer.rb +308 -31
- data/lib/herringbone/xxhash.rb +108 -2
- data/lib/herringbone.rb +28 -0
- metadata +3 -2
data/lib/herringbone/reader.rb
CHANGED
|
@@ -22,18 +22,33 @@ module Herringbone
|
|
|
22
22
|
# TZInfo::Timezone), or anything responding to #at such as an
|
|
23
23
|
# ActiveSupport::TimeZone (Time.zone), which yields ActiveSupport::TimeWithZone.
|
|
24
24
|
class Reader
|
|
25
|
+
# The 4 bytes a Parquet file starts and ends with
|
|
25
26
|
MAGIC = "PAR1"
|
|
27
|
+
# Rows per batch in #each_batch when no size is given
|
|
26
28
|
DEFAULT_BATCH_SIZE = 1024
|
|
29
|
+
# Accepted values of the +keys:+ option
|
|
27
30
|
KEY_MODES = %i[string symbol].freeze
|
|
31
|
+
# Accepted values of the +as:+ option
|
|
32
|
+
AS_MODES = %i[rows columns numo].freeze
|
|
28
33
|
|
|
29
|
-
#
|
|
30
|
-
attr_reader :schema
|
|
34
|
+
# @return [Schema] the schema, built from the footer's flattened SchemaElements
|
|
35
|
+
attr_reader :schema
|
|
36
|
+
|
|
37
|
+
# @return [Format::FileMetaData] the file's FileMetaData (the decoded Thrift footer)
|
|
38
|
+
attr_reader :file_metadata
|
|
31
39
|
|
|
32
40
|
# +io+ must support #seek and #read (a File opened with "rb", StringIO, Tempfile...)
|
|
41
|
+
#
|
|
42
|
+
# @param io [IO, StringIO] random-access source of the Parquet bytes; the caller closes it
|
|
43
|
+
# @param keys [Symbol, String] +:string+ or +:symbol+, the key type of row and struct Hashes
|
|
44
|
+
# @param time_zone [String, Integer, Object, nil] zone timestamps are returned in (see the
|
|
45
|
+
# class docs); nil keeps them in UTC
|
|
46
|
+
# @raise [ArgumentError] when +io+ cannot seek and read, or +keys+ / +time_zone+ are invalid
|
|
47
|
+
# @raise [FormatError] when the footer is missing or cannot be decoded
|
|
33
48
|
def initialize(io, keys: :string, time_zone: nil)
|
|
34
49
|
unless io.respond_to?(:seek) && io.respond_to?(:read)
|
|
35
50
|
raise ArgumentError, "Herringbone::Reader expects an IO that supports #seek and #read " \
|
|
36
|
-
"(e.g. File.open(path, \"rb\")), got #{io.
|
|
51
|
+
"(e.g. File.open(path, \"rb\")), got #{io.is_a?(String) ? "a String" : io.class}" \
|
|
37
52
|
"#{" (wrap Parquet bytes in a StringIO)" if io.is_a?(String)}"
|
|
38
53
|
end
|
|
39
54
|
keys = keys.to_sym if keys.is_a?(String)
|
|
@@ -45,10 +60,19 @@ module Herringbone
|
|
|
45
60
|
@schema = Schema.from_elements(@file_metadata.schema)
|
|
46
61
|
end
|
|
47
62
|
|
|
63
|
+
# Total number of rows, as stated in the footer (some writers store 0)
|
|
64
|
+
#
|
|
65
|
+
# @return [Integer] the footer's num_rows
|
|
48
66
|
def num_rows = @file_metadata.num_rows
|
|
67
|
+
|
|
68
|
+
# The row groups listed in the footer
|
|
69
|
+
#
|
|
70
|
+
# @return [Array<Format::RowGroup>] row group metadata, in file order
|
|
49
71
|
def row_groups = @file_metadata.row_groups
|
|
50
72
|
|
|
51
73
|
# The footer's key/value metadata as a Hash (what the writer's metadata: option stores)
|
|
74
|
+
#
|
|
75
|
+
# @return [Hash{String => String, nil}] key => value; empty when the footer has none
|
|
52
76
|
def metadata
|
|
53
77
|
(@file_metadata.key_value_metadata || []).to_h { |kv| [kv.key, kv.value] }
|
|
54
78
|
end
|
|
@@ -58,24 +82,47 @@ module Herringbone
|
|
|
58
82
|
#
|
|
59
83
|
# as: :rows (default) yields an Array of row Hashes; as: :columns yields a Hash of top-level
|
|
60
84
|
# field name => Array of that field's values in the batch, which skips building a Hash per row
|
|
61
|
-
# and is noticeably faster when you process data column by column.
|
|
85
|
+
# and is noticeably faster when you process data column by column. as: :numo yields a Hash of
|
|
86
|
+
# field name => Numo array (needs the numo-narray-alt or numo-narray gem; see NumoColumns
|
|
87
|
+
# for the type mapping). With as: :numo, whether an integer column becomes DFloat (nulls) or
|
|
88
|
+
# a list column 2-D is decided per batch, from the values in it.
|
|
62
89
|
#
|
|
63
90
|
# where: only yields rows matching all conditions (see Reader::Filter). Row groups and pages
|
|
64
91
|
# that cannot match are skipped using statistics, bloom filters and the page index, and the
|
|
65
92
|
# remaining rows are checked one by one. Filtered columns need not be in +columns+.
|
|
66
93
|
# from: skips the first rows of the file (jumping over pages with the page index), and
|
|
67
94
|
# limit: stops after yielding that many rows.
|
|
95
|
+
#
|
|
96
|
+
# @param size [Integer] maximum number of rows per batch
|
|
97
|
+
# @param columns [Array<String, Symbol>, String, Symbol, nil] top-level fields to return;
|
|
98
|
+
# nil returns all of them
|
|
99
|
+
# @param as [Symbol] +:rows+, +:columns+ or +:numo+, the shape of each batch
|
|
100
|
+
# @param where [Hash{String, Symbol => Object}, nil] column => condition, see Filter
|
|
101
|
+
# @param from [Integer, nil] number of rows at the start of the file to skip
|
|
102
|
+
# @param limit [Integer, nil] maximum number of rows to yield in total
|
|
103
|
+
# @yield [batch] once per batch
|
|
104
|
+
# @yieldparam batch [Array<Hash>, Hash{String, Symbol => Array}, Hash{String, Symbol => Numo::NArray}]
|
|
105
|
+
# row Hashes (+:rows+), or field name => values of the batch (+:columns+, +:numo+)
|
|
106
|
+
# @yieldreturn [void]
|
|
107
|
+
# @return [Reader, Enumerator] self, or an Enumerator of batches when no block is given
|
|
108
|
+
# @raise [ArgumentError] for a non-positive +size+, unknown +as+, negative +from+ / +limit+,
|
|
109
|
+
# or an unknown column in +columns+ or +where+
|
|
110
|
+
# @raise [UnsupportedError] with +as: :numo+ when no Numo gem can be loaded
|
|
68
111
|
def each_batch(size = DEFAULT_BATCH_SIZE, columns: nil, as: :rows, where: nil, from: nil, limit: nil)
|
|
69
112
|
return enum_for(:each_batch, size, columns: columns, as: as, where: where, from: from, limit: limit) unless block_given?
|
|
70
113
|
size = Integer(size)
|
|
71
114
|
raise ArgumentError, "Batch size must be positive, got #{size}" unless size.positive?
|
|
72
|
-
raise ArgumentError, "as: must be :rows or :
|
|
73
|
-
raise ArgumentError, "limit: must not be negative" if limit
|
|
74
|
-
return
|
|
115
|
+
raise ArgumentError, "as: must be :rows, :columns or :numo, got #{as.inspect}" unless AS_MODES.include?(as)
|
|
116
|
+
raise ArgumentError, "limit: must not be negative" if limit&.negative?
|
|
117
|
+
return validate_read_options(columns, where, from) if limit&.zero?
|
|
118
|
+
if as == :numo
|
|
119
|
+
each_numo_batch(size, columns, where, from, limit) { |batch| yield batch }
|
|
120
|
+
return self
|
|
121
|
+
end
|
|
75
122
|
columnar = as == :columns
|
|
76
123
|
symbolize = @symbolize
|
|
77
124
|
out_fields = select_fields(columns)
|
|
78
|
-
filter = where && !where.empty? ? Filter.new(@schema, where) : nil
|
|
125
|
+
filter = (where && !where.empty?) ? Filter.new(@schema, where) : nil
|
|
79
126
|
fields = filter ? out_fields | filter.fields : out_fields
|
|
80
127
|
names = row_keys(out_fields, symbolize)
|
|
81
128
|
nout = out_fields.size
|
|
@@ -157,15 +204,32 @@ module Herringbone
|
|
|
157
204
|
# What a read with +where:+ / +from:+ would touch, without reading any data: an Array of
|
|
158
205
|
# { row_group:, rows:, ranges: [[first_row, end_row), ...] } for the row groups that are
|
|
159
206
|
# read. Row groups ruled out entirely are left out.
|
|
207
|
+
#
|
|
208
|
+
# @param where [Hash{String, Symbol => Object}, nil] column => condition, as in #each_batch
|
|
209
|
+
# @param from [Integer, nil] number of rows at the start of the file to skip
|
|
210
|
+
# @return [Array<Hash{Symbol => Object}>] +{row_group: Integer, rows: Integer,
|
|
211
|
+
# ranges: Array<Array(Integer, Integer)>}+ per row group that would be read
|
|
212
|
+
# @raise [ArgumentError] for a negative +from+ or an unknown column in +where+
|
|
160
213
|
def scan_plan(where: nil, from: nil)
|
|
161
|
-
filter = where && !where.empty? ? Filter.new(@schema, where) : nil
|
|
214
|
+
filter = (where && !where.empty?) ? Filter.new(@schema, where) : nil
|
|
162
215
|
plan_rows(filter, from).map do |rg_index, ranges|
|
|
163
|
-
{
|
|
216
|
+
{row_group: rg_index, rows: ranges.sum { |s, e| e - s }, ranges: ranges}
|
|
164
217
|
end
|
|
165
218
|
end
|
|
166
219
|
|
|
167
220
|
# Yields each row as a Hash of top-level field name => value. Takes the options of each_batch
|
|
168
221
|
# except as:.
|
|
222
|
+
#
|
|
223
|
+
# @param columns [Array<String, Symbol>, String, Symbol, nil] top-level fields to return;
|
|
224
|
+
# nil returns all of them
|
|
225
|
+
# @param where [Hash{String, Symbol => Object}, nil] column => condition, see Filter
|
|
226
|
+
# @param from [Integer, nil] number of rows at the start of the file to skip
|
|
227
|
+
# @param limit [Integer, nil] maximum number of rows to yield
|
|
228
|
+
# @yield [row] once per row
|
|
229
|
+
# @yieldparam row [Hash{String, Symbol => Object}] top-level field name => value
|
|
230
|
+
# @yieldreturn [void]
|
|
231
|
+
# @return [Reader, Enumerator] self, or an Enumerator of rows when no block is given
|
|
232
|
+
# @raise [ArgumentError] for a negative +from+ / +limit+ or an unknown column
|
|
169
233
|
def each_row(columns: nil, where: nil, from: nil, limit: nil, &block)
|
|
170
234
|
return enum_for(:each_row, columns: columns, where: where, from: from, limit: limit) unless block
|
|
171
235
|
each_batch(columns: columns, where: where, from: from, limit: limit) { |rows| rows.each(&block) }
|
|
@@ -173,9 +237,32 @@ module Herringbone
|
|
|
173
237
|
end
|
|
174
238
|
|
|
175
239
|
# Reads the whole file (or the selected rows) at once: an Array of row Hashes, or with
|
|
176
|
-
# as: :columns a Hash of top-level field name => Array of values
|
|
240
|
+
# as: :columns a Hash of top-level field name => Array of values, or with as: :numo a Hash of
|
|
241
|
+
# top-level field name => Numo array. Takes the options of each_batch.
|
|
242
|
+
#
|
|
243
|
+
# @param columns [Array<String, Symbol>, String, Symbol, nil] top-level fields to return;
|
|
244
|
+
# nil returns all of them
|
|
245
|
+
# @param as [Symbol] +:rows+, +:columns+ or +:numo+, the shape of the result
|
|
246
|
+
# @param where [Hash{String, Symbol => Object}, nil] column => condition, see Filter
|
|
247
|
+
# @param from [Integer, nil] number of rows at the start of the file to skip
|
|
248
|
+
# @param limit [Integer, nil] maximum number of rows to return
|
|
249
|
+
# @return [Array<Hash>, Hash{String, Symbol => Array}, Hash{String, Symbol => Numo::NArray}]
|
|
250
|
+
# row Hashes (+:rows+), or field name => all values (+:columns+, +:numo+)
|
|
251
|
+
# @raise [ArgumentError] for an unknown +as+, negative +from+ / +limit+ or an unknown column
|
|
252
|
+
# @raise [UnsupportedError] with +as: :numo+ when no Numo gem can be loaded
|
|
177
253
|
def read(columns: nil, as: :rows, where: nil, from: nil, limit: nil)
|
|
178
|
-
if as == :
|
|
254
|
+
if as == :numo
|
|
255
|
+
raise ArgumentError, "limit: must not be negative" if limit&.negative?
|
|
256
|
+
out = nil
|
|
257
|
+
if limit&.zero?
|
|
258
|
+
validate_read_options(columns, where, from)
|
|
259
|
+
else
|
|
260
|
+
# One batch (num_rows is not trusted: some writers store 0), so the column types are
|
|
261
|
+
# decided from all the rows read
|
|
262
|
+
each_numo_batch(1 << 62, columns, where, from, limit) { |batch| out = batch }
|
|
263
|
+
end
|
|
264
|
+
out ||= numo_empty(columns)
|
|
265
|
+
elsif as == :columns
|
|
179
266
|
out = row_keys(select_fields(columns), @symbolize).to_h { |name| [name, []] }
|
|
180
267
|
each_batch(65_536, columns: columns, as: :columns, where: where, from: from, limit: limit) do |batch|
|
|
181
268
|
batch.each { |name, values| out[name].concat(values) }
|
|
@@ -188,7 +275,13 @@ module Herringbone
|
|
|
188
275
|
end
|
|
189
276
|
|
|
190
277
|
# Internal (used by reads with where:/from:): [ColumnIndex or nil, OffsetIndex or nil] of a
|
|
191
|
-
# leaf column in a row group
|
|
278
|
+
# leaf column in a row group. Cached per chunk; a missing or damaged index reads as nil.
|
|
279
|
+
#
|
|
280
|
+
# @param row_group_index [Integer] position of the row group in the footer
|
|
281
|
+
# @param column [Schema::Column, String, Array<String>] leaf column, or its dotted / Array path
|
|
282
|
+
# @return [Array(Format::ColumnIndex, Format::OffsetIndex)] either element may be nil
|
|
283
|
+
# @raise [ArgumentError] when +column+ does not name a leaf column
|
|
284
|
+
# @raise [IndexError] when the row group does not exist
|
|
192
285
|
def page_index(row_group_index, column)
|
|
193
286
|
column = @schema.column(column) unless column.is_a?(Schema::Column)
|
|
194
287
|
raise ArgumentError, "No such leaf column" unless column
|
|
@@ -200,14 +293,146 @@ module Herringbone
|
|
|
200
293
|
end
|
|
201
294
|
end
|
|
202
295
|
|
|
296
|
+
# Short summary for the console, without the schema
|
|
297
|
+
#
|
|
298
|
+
# @return [String] row count, row group count and the writer's created_by
|
|
203
299
|
def inspect
|
|
204
300
|
"#<#{self.class.name} rows=#{num_rows} row_groups=#{row_groups.size} created_by=#{@file_metadata.created_by.inspect}>"
|
|
205
301
|
end
|
|
206
302
|
|
|
207
303
|
private
|
|
208
304
|
|
|
305
|
+
# as: :numo. Flat numeric/boolean output columns go through NumoCursors (no Ruby object per
|
|
306
|
+
# value); the other output columns, and every column a where: filter needs, are assembled as
|
|
307
|
+
# Ruby values like as: :columns and converted when a batch is complete.
|
|
308
|
+
#
|
|
309
|
+
# @param size [Integer] maximum number of rows per batch
|
|
310
|
+
# @param columns [Array<String, Symbol>, String, Symbol, nil] top-level fields to return
|
|
311
|
+
# @param where [Hash{String, Symbol => Object}, nil] column => condition, see Filter
|
|
312
|
+
# @param from [Integer, nil] number of rows at the start of the file to skip
|
|
313
|
+
# @param limit [Integer, nil] maximum number of rows to yield in total
|
|
314
|
+
# @yield [batch] once per batch
|
|
315
|
+
# @yieldparam batch [Hash{String, Symbol => Numo::NArray}] field name => values of the batch
|
|
316
|
+
# @yieldreturn [void]
|
|
317
|
+
# @return [void]
|
|
318
|
+
# @raise [UnsupportedError] when no Numo gem can be loaded
|
|
319
|
+
def each_numo_batch(size, columns, where, from, limit)
|
|
320
|
+
NumoColumns.load!
|
|
321
|
+
symbolize = @symbolize
|
|
322
|
+
out_fields = select_fields(columns)
|
|
323
|
+
filter = (where && !where.empty?) ? Filter.new(@schema, where) : nil
|
|
324
|
+
filter_fields = filter ? filter.fields : []
|
|
325
|
+
names = row_keys(out_fields, symbolize)
|
|
326
|
+
specs = out_fields.map { |f| NumoColumns.spec_for(f) }
|
|
327
|
+
fast = out_fields.each_index.select { |j| specs[j].fast? && !filter_fields.include?(out_fields[j]) }
|
|
328
|
+
ruby_fields = (out_fields.each_index.to_a - fast).map { |j| out_fields[j] } | filter_fields
|
|
329
|
+
# Where each output column comes from: index into the fast cursors or the Ruby fields
|
|
330
|
+
sources = out_fields.each_index.map { |j| (i = fast.index(j)) ? [:fast, i] : [:ruby, ruby_fields.index(out_fields[j])] }
|
|
331
|
+
converters = @schema.columns.map { |col| converter_for(col) }
|
|
332
|
+
assemblers = ruby_fields.map { |f| Assembler.new(f, symbolize) }
|
|
333
|
+
# Rows decoded per step: Ruby values are kept to slices of the batch, Numo arrays need not be
|
|
334
|
+
step = ruby_fields.empty? ? 1 << 20 : 65_536
|
|
335
|
+
left_to_yield = limit
|
|
336
|
+
pending = Array.new(out_fields.size) { [] }
|
|
337
|
+
pending_rows = 0
|
|
338
|
+
flush = lambda do
|
|
339
|
+
batch = sources.each_with_index.map do |(kind, _), j|
|
|
340
|
+
(kind == :fast) ? NumoColumns.finish_fixed(specs[j], pending[j]) : NumoColumns.finish_values(specs[j], pending[j])
|
|
341
|
+
end
|
|
342
|
+
pending = Array.new(out_fields.size) { [] }
|
|
343
|
+
pending_rows = 0
|
|
344
|
+
yield names.zip(batch).to_h
|
|
345
|
+
end
|
|
346
|
+
|
|
347
|
+
plan_rows(filter, from).each do |rg_index, ranges|
|
|
348
|
+
rg = row_groups[rg_index]
|
|
349
|
+
partial = ranges != [[0, rg.num_rows]]
|
|
350
|
+
open = lambda do |col|
|
|
351
|
+
reader = ColumnChunkReader.new(@io, rg.columns.fetch(col.index), col, converter: converters[col.index], lazy: true)
|
|
352
|
+
reader.locations = page_index(rg_index, col)[1]&.page_locations if partial
|
|
353
|
+
reader
|
|
354
|
+
end
|
|
355
|
+
ruby_cursors = ruby_fields.map { |f| f.leaves.map { |col| [col.index, ColumnCursor.new(open.call(col))] } }
|
|
356
|
+
fast_cursors = fast.map { |j| NumoCursor.new(open.call(out_fields[j].column), specs[j]) }
|
|
357
|
+
ranges.each do |first, stop|
|
|
358
|
+
ruby_cursors.each { |cs| cs.each { |_, cursor| cursor.seek(first) } }
|
|
359
|
+
fast_cursors.each { |cursor| cursor.seek(first) }
|
|
360
|
+
left = stop - first
|
|
361
|
+
while left.positive?
|
|
362
|
+
k = size - pending_rows
|
|
363
|
+
raise Error, "Internal error: batch needs #{k} rows" unless k.positive? # never loop without progress
|
|
364
|
+
k = step if k > step
|
|
365
|
+
k = left if left < k
|
|
366
|
+
data = assemblers.each_with_index.map do |asm, j|
|
|
367
|
+
asm.read_rows(k, ruby_cursors[j].to_h { |idx, cursor| [idx, cursor.take(k)] })
|
|
368
|
+
end
|
|
369
|
+
numo = fast_cursors.map { |cursor| cursor.take(k) }
|
|
370
|
+
left -= k
|
|
371
|
+
kept = k
|
|
372
|
+
if filter
|
|
373
|
+
keep = filter.matching_rows(data, ruby_fields, k, symbolize)
|
|
374
|
+
kept = keep.size
|
|
375
|
+
next if kept.zero?
|
|
376
|
+
if kept < k
|
|
377
|
+
data = data.map { |col| keep.map { |i| col[i] } }
|
|
378
|
+
index = Numo::Int64.cast(keep)
|
|
379
|
+
numo = numo.map { |values, valid| [values[index].dup, valid && valid[index].dup] }
|
|
380
|
+
end
|
|
381
|
+
end
|
|
382
|
+
if left_to_yield && kept > left_to_yield
|
|
383
|
+
kept = left_to_yield
|
|
384
|
+
data = data.map { |col| col.first(kept) }
|
|
385
|
+
numo = numo.map { |values, valid| [values[0...kept].dup, valid && valid[0...kept].dup] }
|
|
386
|
+
end
|
|
387
|
+
sources.each_with_index do |(kind, i), j|
|
|
388
|
+
pending[j] << ((kind == :fast) ? numo[i] : data[i])
|
|
389
|
+
end
|
|
390
|
+
pending_rows += kept
|
|
391
|
+
left_to_yield -= kept if left_to_yield
|
|
392
|
+
flush.call if pending_rows >= size || left_to_yield&.zero?
|
|
393
|
+
return if left_to_yield&.zero? # standard:disable Lint/NonLocalExitFromIterator
|
|
394
|
+
end
|
|
395
|
+
end
|
|
396
|
+
end
|
|
397
|
+
flush.call if pending_rows.positive?
|
|
398
|
+
end
|
|
399
|
+
|
|
400
|
+
# Checks the columns:, where: and from: options of a read that returns no rows (limit: 0),
|
|
401
|
+
# so it raises for bad options like any other read
|
|
402
|
+
#
|
|
403
|
+
# @param columns [Array<String, Symbol>, String, Symbol, nil] top-level fields to return
|
|
404
|
+
# @param where [Hash{String, Symbol => Object}, nil] column => condition, see Filter
|
|
405
|
+
# @param from [Integer, nil] number of rows at the start of the file to skip
|
|
406
|
+
# @return [Reader] self
|
|
407
|
+
# @raise [ArgumentError] for an unknown column, a bad condition or a negative +from+
|
|
408
|
+
def validate_read_options(columns, where, from)
|
|
409
|
+
select_fields(columns)
|
|
410
|
+
Filter.new(@schema, where) if where && !where.empty?
|
|
411
|
+
raise ArgumentError, "from: must not be negative" if Integer(from || 0).negative?
|
|
412
|
+
self
|
|
413
|
+
end
|
|
414
|
+
|
|
415
|
+
# read(as: :numo) of no rows: an empty array of each column's type
|
|
416
|
+
#
|
|
417
|
+
# @param columns [Array<String, Symbol>, String, Symbol, nil] top-level fields to return
|
|
418
|
+
# @return [Hash{String, Symbol => Numo::NArray}] field name => zero-length Numo array
|
|
419
|
+
def numo_empty(columns)
|
|
420
|
+
NumoColumns.load!
|
|
421
|
+
fields = select_fields(columns)
|
|
422
|
+
row_keys(fields, @symbolize).zip(fields.map { |f|
|
|
423
|
+
spec = NumoColumns.spec_for(f)
|
|
424
|
+
(spec.kind == :fixed) ? spec.klass.new(0) : Numo::RObject.new(0)
|
|
425
|
+
}).to_h
|
|
426
|
+
end
|
|
427
|
+
|
|
209
428
|
# [[row_group_index, [[first_row, end_row), ...]], ...] to read, after ruling out row groups
|
|
210
429
|
# (statistics, bloom filters) and pages (page index), and skipping the first +from+ rows
|
|
430
|
+
#
|
|
431
|
+
# @param filter [Filter, nil] the where: conditions, if any
|
|
432
|
+
# @param from [Integer, nil] number of rows at the start of the file to skip
|
|
433
|
+
# @return [Array<Array(Integer, Array<Array(Integer, Integer)>)>] row group index and its
|
|
434
|
+
# half-open row ranges, for row groups that have rows left to read
|
|
435
|
+
# @raise [ArgumentError] for a negative +from+
|
|
211
436
|
def plan_rows(filter, from)
|
|
212
437
|
from = Integer(from || 0)
|
|
213
438
|
raise ArgumentError, "from: must not be negative" if from.negative?
|
|
@@ -228,6 +453,12 @@ module Herringbone
|
|
|
228
453
|
plan
|
|
229
454
|
end
|
|
230
455
|
|
|
456
|
+
# Decodes one Thrift struct stored elsewhere in the file (ColumnIndex, OffsetIndex)
|
|
457
|
+
#
|
|
458
|
+
# @param klass [Class] Format struct class to decode with (responds to .decode)
|
|
459
|
+
# @param offset [Integer, nil] file offset of the struct
|
|
460
|
+
# @param length [Integer, nil] byte length of the struct
|
|
461
|
+
# @return [Object, nil] the decoded +klass+ instance, or nil when absent, truncated or corrupt
|
|
231
462
|
def read_struct(klass, offset, length)
|
|
232
463
|
return nil unless offset && length&.positive?
|
|
233
464
|
@io.seek(offset)
|
|
@@ -238,18 +469,34 @@ module Herringbone
|
|
|
238
469
|
nil # a damaged index only means pages cannot be skipped
|
|
239
470
|
end
|
|
240
471
|
|
|
472
|
+
# The top-level fields named by a +columns:+ option
|
|
473
|
+
#
|
|
474
|
+
# @param columns [Array<String, Symbol>, String, Symbol, nil] field names; nil selects all
|
|
475
|
+
# @return [Array<Schema::Field>] the fields, in the order requested
|
|
476
|
+
# @raise [ArgumentError] when a name is not a top-level field
|
|
241
477
|
def select_fields(columns)
|
|
242
478
|
return @schema.fields unless columns
|
|
243
479
|
Array(columns).map { |c| @schema.field(c) or raise ArgumentError, "No such column #{c.inspect}" }
|
|
244
480
|
end
|
|
245
481
|
|
|
482
|
+
# Hash keys for the given fields (frozen, deduplicated Strings, or Symbols)
|
|
483
|
+
#
|
|
484
|
+
# @param fields [Array<Schema::Field>] top-level fields being returned
|
|
485
|
+
# @param symbolize [Boolean] whether to key by Symbol instead of String
|
|
486
|
+
# @return [Array<String>, Array<Symbol>] one key per field
|
|
246
487
|
def row_keys(fields, symbolize)
|
|
247
488
|
fields.map { |f| symbolize ? f.name.to_sym : -f.name }
|
|
248
489
|
end
|
|
249
490
|
|
|
491
|
+
# Turns column-wise batch data into row Hashes
|
|
492
|
+
#
|
|
493
|
+
# @param names [Array<String>, Array<Symbol>] Hash key per output field
|
|
494
|
+
# @param data [Array<Array>] values per output field, each holding +k+ entries
|
|
495
|
+
# @param k [Integer] number of rows in the batch
|
|
496
|
+
# @return [Array<Hash>] +k+ row Hashes
|
|
250
497
|
def build_rows(names, data, k)
|
|
251
498
|
return Array.new(k) { {} } if names.empty?
|
|
252
|
-
return data.first.map { |v| {
|
|
499
|
+
return data.first.map { |v| {names.first => v} } if names.size == 1
|
|
253
500
|
nf = names.size
|
|
254
501
|
Array.new(k) do |i|
|
|
255
502
|
row = {}
|
|
@@ -263,6 +510,9 @@ module Herringbone
|
|
|
263
510
|
end
|
|
264
511
|
|
|
265
512
|
# The column's value converter, with the time zone applied to timestamps
|
|
513
|
+
#
|
|
514
|
+
# @param column [Schema::Column] leaf column being read
|
|
515
|
+
# @return [Proc, nil] physical value => Ruby value, or nil when values are used as decoded
|
|
266
516
|
def converter_for(column)
|
|
267
517
|
base = column.converter
|
|
268
518
|
zc = @zone_converter
|
|
@@ -272,15 +522,25 @@ module Herringbone
|
|
|
272
522
|
|
|
273
523
|
# Timestamps that denote an instant (UTC-adjusted TIMESTAMP, INT96). Local timestamps
|
|
274
524
|
# (isAdjustedToUTC = false) are wall-clock values and are left as they are.
|
|
525
|
+
#
|
|
526
|
+
# @param column [Schema::Column] leaf column to check
|
|
527
|
+
# @return [Boolean] true when the time zone applies to this column's values
|
|
275
528
|
def instant_column?(column)
|
|
276
529
|
return true if column.type == Format::Type::INT96
|
|
277
530
|
kind, _unit, utc = Types.logical_of(column.node)
|
|
278
531
|
kind == :timestamp && utc
|
|
279
532
|
end
|
|
280
533
|
|
|
534
|
+
# A UTC offset string: "+02", "+0200", "+02:00" or "+02:00:00" (sign required)
|
|
281
535
|
OFFSET_PATTERN = /\A([+-])(\d\d)(?::?(\d\d)(?::?(\d\d))?)?\z/
|
|
282
536
|
|
|
283
537
|
# A lambda turning a UTC Time into the zone, or nil for UTC
|
|
538
|
+
#
|
|
539
|
+
# @param zone [String, Integer, Object, nil] the +time_zone:+ option: a UTC offset (String or
|
|
540
|
+
# seconds), a zone name (ActiveSupport or TZInfo), an object responding to #at or
|
|
541
|
+
# #utc_to_local, or nil
|
|
542
|
+
# @return [Proc, nil] Time => Time (or ActiveSupport::TimeWithZone), nil for UTC
|
|
543
|
+
# @raise [ArgumentError] for an unknown, unsupported or out-of-range zone
|
|
284
544
|
def zone_converter(zone)
|
|
285
545
|
case zone
|
|
286
546
|
when nil then nil
|
|
@@ -292,7 +552,7 @@ module Herringbone
|
|
|
292
552
|
return nil if zone.match?(/\A(?:utc|z)\z/i)
|
|
293
553
|
if (m = OFFSET_PATTERN.match(zone))
|
|
294
554
|
# As seconds, since Ruby 3.0 only parses "+HH:MM" offset strings
|
|
295
|
-
return zone_converter((m[1] == "-" ? -1 : 1) * (m[2].to_i * 3600 + m[3].to_i * 60 + m[4].to_i))
|
|
555
|
+
return zone_converter(((m[1] == "-") ? -1 : 1) * (m[2].to_i * 3600 + m[3].to_i * 60 + m[4].to_i))
|
|
296
556
|
end
|
|
297
557
|
if defined?(::ActiveSupport::TimeZone) && (tz = ::ActiveSupport::TimeZone[zone])
|
|
298
558
|
return ->(t) { tz.at(t) }
|
|
@@ -316,14 +576,20 @@ module Herringbone
|
|
|
316
576
|
raise ArgumentError, "Invalid time_zone #{zone.inspect}: #{e.message}"
|
|
317
577
|
end
|
|
318
578
|
|
|
579
|
+
# Reads and decodes the footer: the FileMetaData Thrift struct, its 4-byte little-endian
|
|
580
|
+
# length and the closing magic.
|
|
581
|
+
#
|
|
582
|
+
# @return [Format::FileMetaData] the decoded footer
|
|
583
|
+
# @raise [FormatError] when the file is too short, lacks the magic or the footer is corrupt
|
|
584
|
+
# @raise [UnsupportedError] for an encrypted file (+PARE+ magic)
|
|
319
585
|
def read_footer
|
|
320
586
|
@io.seek(0, IO::SEEK_END)
|
|
321
587
|
size = @io.pos
|
|
322
588
|
raise FormatError, "File too small to be Parquet (#{size} bytes)" if size < 12
|
|
323
589
|
@io.seek(size - 8)
|
|
324
590
|
tail = @io.read(8)
|
|
325
|
-
raise FormatError, "Missing PAR1 footer magic" unless tail.byteslice(4, 4) == MAGIC
|
|
326
591
|
raise UnsupportedError, "Encrypted Parquet files are not supported" if tail.byteslice(4, 4) == "PARE"
|
|
592
|
+
raise FormatError, "Missing PAR1 footer magic" unless tail.byteslice(4, 4) == MAGIC
|
|
327
593
|
footer_len = tail.unpack1("V")
|
|
328
594
|
raise FormatError, "Footer length #{footer_len} exceeds file size" if footer_len + 12 > size
|
|
329
595
|
@io.seek(size - 8 - footer_len)
|
|
@@ -336,6 +602,8 @@ module Herringbone
|
|
|
336
602
|
# Rebuilds nested values of one top-level field from the levels of its leaf columns
|
|
337
603
|
# (the "record assembly" half of the Dremel algorithm).
|
|
338
604
|
class Assembler
|
|
605
|
+
# @param field [Schema::Field] top-level field to assemble
|
|
606
|
+
# @param symbolize [Boolean] whether struct Hashes are keyed by Symbol instead of String
|
|
339
607
|
def initialize(field, symbolize = false)
|
|
340
608
|
@field = field
|
|
341
609
|
@symbolize = symbolize
|
|
@@ -344,6 +612,12 @@ module Herringbone
|
|
|
344
612
|
|
|
345
613
|
# Assembles +n+ values of the field from +chunks+ (leaf column index =>
|
|
346
614
|
# [defs, reps, values] holding exactly those rows)
|
|
615
|
+
#
|
|
616
|
+
# @param n [Integer] number of rows in +chunks+
|
|
617
|
+
# @param chunks [Hash{Integer => Array(Array<Integer>, Array<Integer>, Array)}] leaf column
|
|
618
|
+
# index => [definition levels, repetition levels, values]; levels may be nil
|
|
619
|
+
# @return [Array] +n+ assembled values (nil, scalars, Arrays, Hashes)
|
|
620
|
+
# @raise [FormatError] when the levels do not add up to +n+ rows
|
|
347
621
|
def read_rows(n, chunks)
|
|
348
622
|
@defs = {}
|
|
349
623
|
@reps = {}
|
|
@@ -365,7 +639,7 @@ module Herringbone
|
|
|
365
639
|
max = f.column.max_definition_level
|
|
366
640
|
return vals if vals.size == defs.size
|
|
367
641
|
vi = -1
|
|
368
|
-
return defs.map { |d| d == max ? vals[vi += 1] : nil }
|
|
642
|
+
return defs.map { |d| (d == max) ? vals[vi += 1] : nil }
|
|
369
643
|
end
|
|
370
644
|
return read_simple_list(f) if f.kind == :list && f.element.leaf? && f.element.column.max_repetition_level == 1
|
|
371
645
|
|
|
@@ -381,6 +655,9 @@ module Herringbone
|
|
|
381
655
|
private
|
|
382
656
|
|
|
383
657
|
# Fast path for a top-level list of primitives (the most common nested shape)
|
|
658
|
+
#
|
|
659
|
+
# @param field [Schema::Field] list field whose element is a leaf with repetition level 1
|
|
660
|
+
# @return [Array<Array, nil>] one list (or nil) per row
|
|
384
661
|
def read_simple_list(field)
|
|
385
662
|
col = field.element.column
|
|
386
663
|
defs = @defs[col.index]
|
|
@@ -420,10 +697,19 @@ module Herringbone
|
|
|
420
697
|
out
|
|
421
698
|
end
|
|
422
699
|
|
|
700
|
+
# Hash key for a struct member, memoized
|
|
701
|
+
#
|
|
702
|
+
# @param field [Schema::Field] struct child
|
|
703
|
+
# @return [String, Symbol] frozen name, or Symbol when symbolizing
|
|
423
704
|
def key_for(field)
|
|
424
705
|
@keys[field] ||= @symbolize ? field.name.to_sym : -field.name
|
|
425
706
|
end
|
|
426
707
|
|
|
708
|
+
# Assembles one value of +field+ at the current entry cursors, recursing into children
|
|
709
|
+
#
|
|
710
|
+
# @param field [Schema::Field] field to assemble
|
|
711
|
+
# @return [Object, nil] scalar, Hash (struct, map), Array (list) or nil
|
|
712
|
+
# @raise [FormatError] when the definition levels run out
|
|
427
713
|
def read(field)
|
|
428
714
|
c = field.first_leaf.index
|
|
429
715
|
kind = field.kind
|
|
@@ -478,6 +764,10 @@ module Herringbone
|
|
|
478
764
|
end
|
|
479
765
|
end
|
|
480
766
|
|
|
767
|
+
# Moves every leaf of +field+ past one entry (a null or empty value takes a single entry)
|
|
768
|
+
#
|
|
769
|
+
# @param field [Schema::Field] field whose leaves to advance
|
|
770
|
+
# @return [void]
|
|
481
771
|
def skip(field)
|
|
482
772
|
field.leaves.each { |col| @ei[col.index] += 1 }
|
|
483
773
|
end
|
|
@@ -489,3 +779,4 @@ require_relative "reader/page_stream"
|
|
|
489
779
|
require_relative "reader/column_chunk_reader"
|
|
490
780
|
require_relative "reader/column_cursor"
|
|
491
781
|
require_relative "reader/scan"
|
|
782
|
+
require_relative "reader/numo"
|