herringbone 0.1.0 → 0.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,133 +1,595 @@
1
1
  # frozen_string_literal: true
2
2
 
3
3
  module Herringbone
4
- # Reads Parquet files.
4
+ # Reads Parquet files from a random-access IO. Herringbone never opens files by path: the
5
+ # caller opens (and closes) the IO.
5
6
  #
6
- # Herringbone::Reader.open("data.parquet") do |reader|
7
- # reader.each_row { |row| p row } # rows as Hashes with String keys
8
- # reader.column("name") # all values of a top-level field
9
- # reader.each_row(columns: ["id"]) { ... } # projection
7
+ # File.open("data.parquet", "rb") do |file|
8
+ # reader = Herringbone::Reader.new(file)
9
+ # reader.each_row { |row| p row } # rows as Hashes with String keys
10
+ # reader.each_batch(1000, as: :columns) { |batch| ... } # { "id" => [...], ... } per batch
11
+ # reader.read(columns: ["id"], where: { id: 1..10 }) # everything at once
10
12
  # end
13
+ #
14
+ # Rows are read in batches: pages are read and decoded one at a time per column, so memory use
15
+ # depends on the batch size and the page size, not on the size of the row groups.
16
+ #
17
+ # Options:
18
+ # keys: :string (default) or :symbol, for row Hashes and Hashes built from structs.
19
+ # Map keys are always the stored values.
20
+ # time_zone: return timestamps in this zone instead of UTC. A UTC offset ("+02:00", or
21
+ # seconds as an Integer), a timezone object Time#getlocal accepts (e.g. a
22
+ # TZInfo::Timezone), or anything responding to #at such as an
23
+ # ActiveSupport::TimeZone (Time.zone), which yields ActiveSupport::TimeWithZone.
11
24
  class Reader
12
- include Enumerable
13
-
25
+ # The 4 bytes a Parquet file starts and ends with
14
26
  MAGIC = "PAR1"
27
+ # Rows per batch in #each_batch when no size is given
28
+ DEFAULT_BATCH_SIZE = 1024
29
+ # Accepted values of the +keys:+ option
30
+ KEY_MODES = %i[string symbol].freeze
31
+ # Accepted values of the +as:+ option
32
+ AS_MODES = %i[rows columns numo].freeze
33
+
34
+ # @return [Schema] the schema, built from the footer's flattened SchemaElements
35
+ attr_reader :schema
36
+
37
+ # @return [Format::FileMetaData] the file's FileMetaData (the decoded Thrift footer)
38
+ attr_reader :file_metadata
39
+
40
+ # +io+ must support #seek and #read (a File opened with "rb", StringIO, Tempfile...)
41
+ #
42
+ # @param io [IO, StringIO] random-access source of the Parquet bytes; the caller closes it
43
+ # @param keys [Symbol, String] +:string+ or +:symbol+, the key type of row and struct Hashes
44
+ # @param time_zone [String, Integer, Object, nil] zone timestamps are returned in (see the
45
+ # class docs); nil keeps them in UTC
46
+ # @raise [ArgumentError] when +io+ cannot seek and read, or +keys+ / +time_zone+ are invalid
47
+ # @raise [FormatError] when the footer is missing or cannot be decoded
48
+ def initialize(io, keys: :string, time_zone: nil)
49
+ unless io.respond_to?(:seek) && io.respond_to?(:read)
50
+ raise ArgumentError, "Herringbone::Reader expects an IO that supports #seek and #read " \
51
+ "(e.g. File.open(path, \"rb\")), got #{io.is_a?(String) ? "a String" : io.class}" \
52
+ "#{" (wrap Parquet bytes in a StringIO)" if io.is_a?(String)}"
53
+ end
54
+ keys = keys.to_sym if keys.is_a?(String)
55
+ raise ArgumentError, "keys: must be :string or :symbol, got #{keys.inspect}" unless KEY_MODES.include?(keys)
56
+ @symbolize = keys == :symbol
57
+ @zone_converter = zone_converter(time_zone)
58
+ @io = io
59
+ @file_metadata = read_footer
60
+ @schema = Schema.from_elements(@file_metadata.schema)
61
+ end
15
62
 
16
- attr_reader :metadata, :schema
63
+ # Total number of rows, as stated in the footer (some writers store 0)
64
+ #
65
+ # @return [Integer] the footer's num_rows
66
+ def num_rows = @file_metadata.num_rows
67
+
68
+ # The row groups listed in the footer
69
+ #
70
+ # @return [Array<Format::RowGroup>] row group metadata, in file order
71
+ def row_groups = @file_metadata.row_groups
72
+
73
+ # The footer's key/value metadata as a Hash (what the writer's metadata: option stores)
74
+ #
75
+ # @return [Hash{String => String, nil}] key => value; empty when the footer has none
76
+ def metadata
77
+ (@file_metadata.key_value_metadata || []).to_h { |kv| [kv.key, kv.value] }
78
+ end
17
79
 
18
- def self.open(path)
19
- io = File.open(path, "rb")
20
- reader = new(io)
21
- return reader unless block_given?
22
- begin
23
- yield reader
24
- ensure
25
- io.close
80
+ # Yields batches of up to +size+ rows (all batches are full except the last one; batches span
81
+ # row groups). Only one page per column and the current batch are held in memory.
82
+ #
83
+ # as: :rows (default) yields an Array of row Hashes; as: :columns yields a Hash of top-level
84
+ # field name => Array of that field's values in the batch, which skips building a Hash per row
85
+ # and is noticeably faster when you process data column by column. as: :numo yields a Hash of
86
+ # field name => Numo array (needs the numo-narray-alt or numo-narray gem; see NumoColumns
87
+ # for the type mapping). With as: :numo, whether an integer column becomes DFloat (nulls) or
88
+ # a list column 2-D is decided per batch, from the values in it.
89
+ #
90
+ # where: only yields rows matching all conditions (see Reader::Filter). Row groups and pages
91
+ # that cannot match are skipped using statistics, bloom filters and the page index, and the
92
+ # remaining rows are checked one by one. Filtered columns need not be in +columns+.
93
+ # from: skips the first rows of the file (jumping over pages with the page index), and
94
+ # limit: stops after yielding that many rows.
95
+ #
96
+ # @param size [Integer] maximum number of rows per batch
97
+ # @param columns [Array<String, Symbol>, String, Symbol, nil] top-level fields to return;
98
+ # nil returns all of them
99
+ # @param as [Symbol] +:rows+, +:columns+ or +:numo+, the shape of each batch
100
+ # @param where [Hash{String, Symbol => Object}, nil] column => condition, see Filter
101
+ # @param from [Integer, nil] number of rows at the start of the file to skip
102
+ # @param limit [Integer, nil] maximum number of rows to yield in total
103
+ # @yield [batch] once per batch
104
+ # @yieldparam batch [Array<Hash>, Hash{String, Symbol => Array}, Hash{String, Symbol => Numo::NArray}]
105
+ # row Hashes (+:rows+), or field name => values of the batch (+:columns+, +:numo+)
106
+ # @yieldreturn [void]
107
+ # @return [Reader, Enumerator] self, or an Enumerator of batches when no block is given
108
+ # @raise [ArgumentError] for a non-positive +size+, unknown +as+, negative +from+ / +limit+,
109
+ # or an unknown column in +columns+ or +where+
110
+ # @raise [UnsupportedError] with +as: :numo+ when no Numo gem can be loaded
111
+ def each_batch(size = DEFAULT_BATCH_SIZE, columns: nil, as: :rows, where: nil, from: nil, limit: nil)
112
+ return enum_for(:each_batch, size, columns: columns, as: as, where: where, from: from, limit: limit) unless block_given?
113
+ size = Integer(size)
114
+ raise ArgumentError, "Batch size must be positive, got #{size}" unless size.positive?
115
+ raise ArgumentError, "as: must be :rows, :columns or :numo, got #{as.inspect}" unless AS_MODES.include?(as)
116
+ raise ArgumentError, "limit: must not be negative" if limit&.negative?
117
+ return validate_read_options(columns, where, from) if limit&.zero?
118
+ if as == :numo
119
+ each_numo_batch(size, columns, where, from, limit) { |batch| yield batch }
120
+ return self
121
+ end
122
+ columnar = as == :columns
123
+ symbolize = @symbolize
124
+ out_fields = select_fields(columns)
125
+ filter = (where && !where.empty?) ? Filter.new(@schema, where) : nil
126
+ fields = filter ? out_fields | filter.fields : out_fields
127
+ names = row_keys(out_fields, symbolize)
128
+ nout = out_fields.size
129
+ converters = @schema.columns.map { |col| converter_for(col) }
130
+ assemblers = fields.map { |f| Assembler.new(f, symbolize) }
131
+ left_to_yield = limit
132
+ pending = pending_rows = nil
133
+ emit = lambda do |data, k|
134
+ if columnar
135
+ if pending
136
+ pending.each_with_index { |col, j| col.concat(data[j]) }
137
+ else
138
+ pending = data
139
+ end
140
+ pending_rows = (pending_rows || 0) + k
141
+ if pending_rows >= size
142
+ yield names.zip(pending).to_h
143
+ pending = pending_rows = nil
144
+ end
145
+ else
146
+ rows = build_rows(names, data, k)
147
+ if pending
148
+ pending.concat(rows)
149
+ else
150
+ pending = rows
151
+ end
152
+ pending_rows = pending.size
153
+ if pending_rows >= size
154
+ yield pending
155
+ pending = pending_rows = nil
156
+ end
157
+ end
26
158
  end
159
+
160
+ plan_rows(filter, from).each do |rg_index, ranges|
161
+ rg = row_groups[rg_index]
162
+ partial = ranges != [[0, rg.num_rows]]
163
+ cursors = fields.map do |f|
164
+ f.leaves.map do |col|
165
+ reader = ColumnChunkReader.new(@io, rg.columns.fetch(col.index), col, converter: converters[col.index], lazy: true)
166
+ reader.locations = page_index(rg_index, col)[1]&.page_locations if partial
167
+ [col.index, ColumnCursor.new(reader)]
168
+ end
169
+ end
170
+ ranges.each do |first, stop|
171
+ cursors.each { |cs| cs.each { |_, cursor| cursor.seek(first) } }
172
+ left = stop - first
173
+ while left.positive?
174
+ # Rows still missing from the current batch (pending_rows counts rows in both modes)
175
+ k = pending ? size - pending_rows : size
176
+ raise Error, "Internal error: batch needs #{k} rows" unless k.positive? # never loop without progress
177
+ k = left if left < k
178
+ data = assemblers.each_with_index.map do |asm, j|
179
+ asm.read_rows(k, cursors[j].to_h { |idx, cursor| [idx, cursor.take(k)] })
180
+ end
181
+ left -= k
182
+ kept = k
183
+ if filter
184
+ keep = filter.matching_rows(data, fields, k, symbolize)
185
+ kept = keep.size
186
+ next if kept.zero?
187
+ data = data.first(nout).map { |col| keep.map { |i| col[i] } } if kept < k
188
+ end
189
+ data = data.first(nout) if data.size > nout
190
+ if left_to_yield && kept >= left_to_yield
191
+ emit.call(data.map { |col| col.first(left_to_yield) }, left_to_yield)
192
+ yield(columnar ? names.zip(pending).to_h : pending) if pending
193
+ return self
194
+ end
195
+ left_to_yield -= kept if left_to_yield
196
+ emit.call(data, kept)
197
+ end
198
+ end
199
+ end
200
+ yield(columnar ? names.zip(pending).to_h : pending) if pending
201
+ self
27
202
  end
28
203
 
29
- # +source+ is an IO (opened in binary mode), a path, or a String of Parquet bytes when
30
- # +string: true+ is given.
31
- def initialize(source)
32
- @io = case source
33
- when String then File.open(source, "rb")
34
- when Pathname then File.open(source.to_s, "rb")
35
- else source
204
+ # What a read with +where:+ / +from:+ would touch, without reading any data: an Array of
205
+ # { row_group:, rows:, ranges: [[first_row, end_row), ...] } for the row groups that are
206
+ # read. Row groups ruled out entirely are left out.
207
+ #
208
+ # @param where [Hash{String, Symbol => Object}, nil] column => condition, as in #each_batch
209
+ # @param from [Integer, nil] number of rows at the start of the file to skip
210
+ # @return [Array<Hash{Symbol => Object}>] +{row_group: Integer, rows: Integer,
211
+ # ranges: Array<Array(Integer, Integer)>}+ per row group that would be read
212
+ # @raise [ArgumentError] for a negative +from+ or an unknown column in +where+
213
+ def scan_plan(where: nil, from: nil)
214
+ filter = (where && !where.empty?) ? Filter.new(@schema, where) : nil
215
+ plan_rows(filter, from).map do |rg_index, ranges|
216
+ {row_group: rg_index, rows: ranges.sum { |s, e| e - s }, ranges: ranges}
36
217
  end
37
- @metadata = read_footer
38
- @schema = Schema.from_elements(@metadata.schema)
39
218
  end
40
219
 
41
- def self.from_string(bytes)
42
- new(StringIO.new(bytes.b))
220
+ # Yields each row as a Hash of top-level field name => value. Takes the options of each_batch
221
+ # except as:.
222
+ #
223
+ # @param columns [Array<String, Symbol>, String, Symbol, nil] top-level fields to return;
224
+ # nil returns all of them
225
+ # @param where [Hash{String, Symbol => Object}, nil] column => condition, see Filter
226
+ # @param from [Integer, nil] number of rows at the start of the file to skip
227
+ # @param limit [Integer, nil] maximum number of rows to yield
228
+ # @yield [row] once per row
229
+ # @yieldparam row [Hash{String, Symbol => Object}] top-level field name => value
230
+ # @yieldreturn [void]
231
+ # @return [Reader, Enumerator] self, or an Enumerator of rows when no block is given
232
+ # @raise [ArgumentError] for a negative +from+ / +limit+ or an unknown column
233
+ def each_row(columns: nil, where: nil, from: nil, limit: nil, &block)
234
+ return enum_for(:each_row, columns: columns, where: where, from: from, limit: limit) unless block
235
+ each_batch(columns: columns, where: where, from: from, limit: limit) { |rows| rows.each(&block) }
236
+ self
43
237
  end
44
238
 
45
- def close
46
- @io.close
239
+ # Reads the whole file (or the selected rows) at once: an Array of row Hashes, or with
240
+ # as: :columns a Hash of top-level field name => Array of values, or with as: :numo a Hash of
241
+ # top-level field name => Numo array. Takes the options of each_batch.
242
+ #
243
+ # @param columns [Array<String, Symbol>, String, Symbol, nil] top-level fields to return;
244
+ # nil returns all of them
245
+ # @param as [Symbol] +:rows+, +:columns+ or +:numo+, the shape of the result
246
+ # @param where [Hash{String, Symbol => Object}, nil] column => condition, see Filter
247
+ # @param from [Integer, nil] number of rows at the start of the file to skip
248
+ # @param limit [Integer, nil] maximum number of rows to return
249
+ # @return [Array<Hash>, Hash{String, Symbol => Array}, Hash{String, Symbol => Numo::NArray}]
250
+ # row Hashes (+:rows+), or field name => all values (+:columns+, +:numo+)
251
+ # @raise [ArgumentError] for an unknown +as+, negative +from+ / +limit+ or an unknown column
252
+ # @raise [UnsupportedError] with +as: :numo+ when no Numo gem can be loaded
253
+ def read(columns: nil, as: :rows, where: nil, from: nil, limit: nil)
254
+ if as == :numo
255
+ raise ArgumentError, "limit: must not be negative" if limit&.negative?
256
+ out = nil
257
+ if limit&.zero?
258
+ validate_read_options(columns, where, from)
259
+ else
260
+ # One batch (num_rows is not trusted: some writers store 0), so the column types are
261
+ # decided from all the rows read
262
+ each_numo_batch(1 << 62, columns, where, from, limit) { |batch| out = batch }
263
+ end
264
+ out ||= numo_empty(columns)
265
+ elsif as == :columns
266
+ out = row_keys(select_fields(columns), @symbolize).to_h { |name| [name, []] }
267
+ each_batch(65_536, columns: columns, as: :columns, where: where, from: from, limit: limit) do |batch|
268
+ batch.each { |name, values| out[name].concat(values) }
269
+ end
270
+ else
271
+ out = []
272
+ each_batch(columns: columns, as: as, where: where, from: from, limit: limit) { |rows| out.concat(rows) }
273
+ end
274
+ out
47
275
  end
48
276
 
49
- def num_rows = @metadata.num_rows
50
- def row_groups = @metadata.row_groups
51
- def num_row_groups = @metadata.row_groups.size
52
- def created_by = @metadata.created_by
277
+ # Internal (used by reads with where:/from:): [ColumnIndex or nil, OffsetIndex or nil] of a
278
+ # leaf column in a row group. Cached per chunk; a missing or damaged index reads as nil.
279
+ #
280
+ # @param row_group_index [Integer] position of the row group in the footer
281
+ # @param column [Schema::Column, String, Array<String>] leaf column, or its dotted / Array path
282
+ # @return [Array(Format::ColumnIndex, Format::OffsetIndex)] either element may be nil
283
+ # @raise [ArgumentError] when +column+ does not name a leaf column
284
+ # @raise [IndexError] when the row group does not exist
285
+ def page_index(row_group_index, column)
286
+ column = @schema.column(column) unless column.is_a?(Schema::Column)
287
+ raise ArgumentError, "No such leaf column" unless column
288
+ @page_indexes ||= {}
289
+ @page_indexes[[row_group_index, column.index]] ||= begin
290
+ chunk = row_groups.fetch(row_group_index).columns.fetch(column.index)
291
+ [read_struct(Format::ColumnIndex, chunk.column_index_offset, chunk.column_index_length),
292
+ read_struct(Format::OffsetIndex, chunk.offset_index_offset, chunk.offset_index_length)]
293
+ end
294
+ end
53
295
 
54
- def key_value_metadata
55
- (@metadata.key_value_metadata || []).to_h { |kv| [kv.key, kv.value] }
296
+ # Short summary for the console, without the schema
297
+ #
298
+ # @return [String] row count, row group count and the writer's created_by
299
+ def inspect
300
+ "#<#{self.class.name} rows=#{num_rows} row_groups=#{row_groups.size} created_by=#{@file_metadata.created_by.inspect}>"
56
301
  end
57
302
 
58
- # Yields each row as a Hash of top-level field name => value
59
- def each_row(columns: nil, &block)
60
- return enum_for(:each_row, columns: columns) unless block
61
- fields = select_fields(columns)
62
- row_groups.each_index do |rg|
63
- data = read_row_group_fields(rg, fields)
64
- n = row_groups[rg].num_rows
65
- names = fields.map(&:name)
66
- n.times do |i|
67
- row = {}
68
- names.each_with_index { |name, j| row[name] = data[j][i] }
69
- yield row
303
+ private
304
+
305
+ # as: :numo. Flat numeric/boolean output columns go through NumoCursors (no Ruby object per
306
+ # value); the other output columns, and every column a where: filter needs, are assembled as
307
+ # Ruby values like as: :columns and converted when a batch is complete.
308
+ #
309
+ # @param size [Integer] maximum number of rows per batch
310
+ # @param columns [Array<String, Symbol>, String, Symbol, nil] top-level fields to return
311
+ # @param where [Hash{String, Symbol => Object}, nil] column => condition, see Filter
312
+ # @param from [Integer, nil] number of rows at the start of the file to skip
313
+ # @param limit [Integer, nil] maximum number of rows to yield in total
314
+ # @yield [batch] once per batch
315
+ # @yieldparam batch [Hash{String, Symbol => Numo::NArray}] field name => values of the batch
316
+ # @yieldreturn [void]
317
+ # @return [void]
318
+ # @raise [UnsupportedError] when no Numo gem can be loaded
319
+ def each_numo_batch(size, columns, where, from, limit)
320
+ NumoColumns.load!
321
+ symbolize = @symbolize
322
+ out_fields = select_fields(columns)
323
+ filter = (where && !where.empty?) ? Filter.new(@schema, where) : nil
324
+ filter_fields = filter ? filter.fields : []
325
+ names = row_keys(out_fields, symbolize)
326
+ specs = out_fields.map { |f| NumoColumns.spec_for(f) }
327
+ fast = out_fields.each_index.select { |j| specs[j].fast? && !filter_fields.include?(out_fields[j]) }
328
+ ruby_fields = (out_fields.each_index.to_a - fast).map { |j| out_fields[j] } | filter_fields
329
+ # Where each output column comes from: index into the fast cursors or the Ruby fields
330
+ sources = out_fields.each_index.map { |j| (i = fast.index(j)) ? [:fast, i] : [:ruby, ruby_fields.index(out_fields[j])] }
331
+ converters = @schema.columns.map { |col| converter_for(col) }
332
+ assemblers = ruby_fields.map { |f| Assembler.new(f, symbolize) }
333
+ # Rows decoded per step: Ruby values are kept to slices of the batch, Numo arrays need not be
334
+ step = ruby_fields.empty? ? 1 << 20 : 65_536
335
+ left_to_yield = limit
336
+ pending = Array.new(out_fields.size) { [] }
337
+ pending_rows = 0
338
+ flush = lambda do
339
+ batch = sources.each_with_index.map do |(kind, _), j|
340
+ (kind == :fast) ? NumoColumns.finish_fixed(specs[j], pending[j]) : NumoColumns.finish_values(specs[j], pending[j])
70
341
  end
342
+ pending = Array.new(out_fields.size) { [] }
343
+ pending_rows = 0
344
+ yield names.zip(batch).to_h
71
345
  end
72
- self
73
- end
74
- alias_method :each, :each_row
75
346
 
76
- # Rows as an Array of Hashes
77
- def rows(columns: nil) = each_row(columns: columns).to_a
347
+ plan_rows(filter, from).each do |rg_index, ranges|
348
+ rg = row_groups[rg_index]
349
+ partial = ranges != [[0, rg.num_rows]]
350
+ open = lambda do |col|
351
+ reader = ColumnChunkReader.new(@io, rg.columns.fetch(col.index), col, converter: converters[col.index], lazy: true)
352
+ reader.locations = page_index(rg_index, col)[1]&.page_locations if partial
353
+ reader
354
+ end
355
+ ruby_cursors = ruby_fields.map { |f| f.leaves.map { |col| [col.index, ColumnCursor.new(open.call(col))] } }
356
+ fast_cursors = fast.map { |j| NumoCursor.new(open.call(out_fields[j].column), specs[j]) }
357
+ ranges.each do |first, stop|
358
+ ruby_cursors.each { |cs| cs.each { |_, cursor| cursor.seek(first) } }
359
+ fast_cursors.each { |cursor| cursor.seek(first) }
360
+ left = stop - first
361
+ while left.positive?
362
+ k = size - pending_rows
363
+ raise Error, "Internal error: batch needs #{k} rows" unless k.positive? # never loop without progress
364
+ k = step if k > step
365
+ k = left if left < k
366
+ data = assemblers.each_with_index.map do |asm, j|
367
+ asm.read_rows(k, ruby_cursors[j].to_h { |idx, cursor| [idx, cursor.take(k)] })
368
+ end
369
+ numo = fast_cursors.map { |cursor| cursor.take(k) }
370
+ left -= k
371
+ kept = k
372
+ if filter
373
+ keep = filter.matching_rows(data, ruby_fields, k, symbolize)
374
+ kept = keep.size
375
+ next if kept.zero?
376
+ if kept < k
377
+ data = data.map { |col| keep.map { |i| col[i] } }
378
+ index = Numo::Int64.cast(keep)
379
+ numo = numo.map { |values, valid| [values[index].dup, valid && valid[index].dup] }
380
+ end
381
+ end
382
+ if left_to_yield && kept > left_to_yield
383
+ kept = left_to_yield
384
+ data = data.map { |col| col.first(kept) }
385
+ numo = numo.map { |values, valid| [values[0...kept].dup, valid && valid[0...kept].dup] }
386
+ end
387
+ sources.each_with_index do |(kind, i), j|
388
+ pending[j] << ((kind == :fast) ? numo[i] : data[i])
389
+ end
390
+ pending_rows += kept
391
+ left_to_yield -= kept if left_to_yield
392
+ flush.call if pending_rows >= size || left_to_yield&.zero?
393
+ return if left_to_yield&.zero? # standard:disable Lint/NonLocalExitFromIterator
394
+ end
395
+ end
396
+ end
397
+ flush.call if pending_rows.positive?
398
+ end
78
399
 
79
- # All values of a single top-level field, across all row groups
80
- def column(name)
81
- field = @schema.field(name) or raise ArgumentError, "No such column #{name.inspect}"
82
- row_groups.each_index.flat_map { |rg| read_row_group_fields(rg, [field]).first }
400
+ # Checks the columns:, where: and from: options of a read that returns no rows (limit: 0),
401
+ # so it raises for bad options like any other read
402
+ #
403
+ # @param columns [Array<String, Symbol>, String, Symbol, nil] top-level fields to return
404
+ # @param where [Hash{String, Symbol => Object}, nil] column => condition, see Filter
405
+ # @param from [Integer, nil] number of rows at the start of the file to skip
406
+ # @return [Reader] self
407
+ # @raise [ArgumentError] for an unknown column, a bad condition or a negative +from+
408
+ def validate_read_options(columns, where, from)
409
+ select_fields(columns)
410
+ Filter.new(@schema, where) if where && !where.empty?
411
+ raise ArgumentError, "from: must not be negative" if Integer(from || 0).negative?
412
+ self
83
413
  end
84
414
 
85
- # Hash of field name => Array of values, for the given row group
86
- def read_row_group(index, columns: nil)
415
+ # read(as: :numo) of no rows: an empty array of each column's type
416
+ #
417
+ # @param columns [Array<String, Symbol>, String, Symbol, nil] top-level fields to return
418
+ # @return [Hash{String, Symbol => Numo::NArray}] field name => zero-length Numo array
419
+ def numo_empty(columns)
420
+ NumoColumns.load!
87
421
  fields = select_fields(columns)
88
- fields.map(&:name).zip(read_row_group_fields(index, fields)).to_h
422
+ row_keys(fields, @symbolize).zip(fields.map { |f|
423
+ spec = NumoColumns.spec_for(f)
424
+ (spec.kind == :fixed) ? spec.klass.new(0) : Numo::RObject.new(0)
425
+ }).to_h
89
426
  end
90
427
 
91
- # Raw column data for a leaf column: [definition_levels, repetition_levels, values].
92
- # Levels are nil when the column's max level is 0.
93
- def read_column_chunk(row_group_index, column)
94
- column = @schema.column(column) unless column.is_a?(Schema::Column)
95
- raise ArgumentError, "No such leaf column" unless column
96
- chunk = row_groups.fetch(row_group_index).columns.fetch(column.index)
97
- ColumnChunkReader.new(@io, chunk, column).read
428
+ # [[row_group_index, [[first_row, end_row), ...]], ...] to read, after ruling out row groups
429
+ # (statistics, bloom filters) and pages (page index), and skipping the first +from+ rows
430
+ #
431
+ # @param filter [Filter, nil] the where: conditions, if any
432
+ # @param from [Integer, nil] number of rows at the start of the file to skip
433
+ # @return [Array<Array(Integer, Array<Array(Integer, Integer)>)>] row group index and its
434
+ # half-open row ranges, for row groups that have rows left to read
435
+ # @raise [ArgumentError] for a negative +from+
436
+ def plan_rows(filter, from)
437
+ from = Integer(from || 0)
438
+ raise ArgumentError, "from: must not be negative" if from.negative?
439
+ offset = 0
440
+ plan = []
441
+ row_groups.each_with_index do |rg, i|
442
+ n = rg.num_rows
443
+ first = from - offset
444
+ offset += n
445
+ next if n.zero? || first >= n
446
+ ranges = [[first.positive? ? first : 0, n]]
447
+ if filter
448
+ next unless filter.row_group_may_match?(self, i)
449
+ ranges = Filter.intersect(ranges, filter.page_ranges(self, i))
450
+ end
451
+ plan << [i, ranges] unless ranges.empty?
452
+ end
453
+ plan
98
454
  end
99
455
 
100
- def inspect
101
- "#<#{self.class.name} rows=#{num_rows} row_groups=#{num_row_groups} created_by=#{created_by.inspect}>"
456
+ # Decodes one Thrift struct stored elsewhere in the file (ColumnIndex, OffsetIndex)
457
+ #
458
+ # @param klass [Class] Format struct class to decode with (responds to .decode)
459
+ # @param offset [Integer, nil] file offset of the struct
460
+ # @param length [Integer, nil] byte length of the struct
461
+ # @return [Object, nil] the decoded +klass+ instance, or nil when absent, truncated or corrupt
462
+ def read_struct(klass, offset, length)
463
+ return nil unless offset && length&.positive?
464
+ @io.seek(offset)
465
+ bytes = @io.read(length)
466
+ return nil unless bytes&.bytesize == length
467
+ klass.decode(bytes.b).first
468
+ rescue Thrift::Error
469
+ nil # a damaged index only means pages cannot be skipped
102
470
  end
103
471
 
104
- private
105
-
472
+ # The top-level fields named by a +columns:+ option
473
+ #
474
+ # @param columns [Array<String, Symbol>, String, Symbol, nil] field names; nil selects all
475
+ # @return [Array<Schema::Field>] the fields, in the order requested
476
+ # @raise [ArgumentError] when a name is not a top-level field
106
477
  def select_fields(columns)
107
478
  return @schema.fields unless columns
108
479
  Array(columns).map { |c| @schema.field(c) or raise ArgumentError, "No such column #{c.inspect}" }
109
480
  end
110
481
 
111
- def read_row_group_fields(rg, fields)
112
- rg_meta = row_groups.fetch(rg)
113
- n = rg_meta.num_rows
114
- fields.map do |field|
115
- chunks = {}
116
- field.leaves.each do |col|
117
- chunks[col.index] = ColumnChunkReader.new(@io, rg_meta.columns.fetch(col.index), col).read
482
+ # Hash keys for the given fields (frozen, deduplicated Strings, or Symbols)
483
+ #
484
+ # @param fields [Array<Schema::Field>] top-level fields being returned
485
+ # @param symbolize [Boolean] whether to key by Symbol instead of String
486
+ # @return [Array<String>, Array<Symbol>] one key per field
487
+ def row_keys(fields, symbolize)
488
+ fields.map { |f| symbolize ? f.name.to_sym : -f.name }
489
+ end
490
+
491
+ # Turns column-wise batch data into row Hashes
492
+ #
493
+ # @param names [Array<String>, Array<Symbol>] Hash key per output field
494
+ # @param data [Array<Array>] values per output field, each holding +k+ entries
495
+ # @param k [Integer] number of rows in the batch
496
+ # @return [Array<Hash>] +k+ row Hashes
497
+ def build_rows(names, data, k)
498
+ return Array.new(k) { {} } if names.empty?
499
+ return data.first.map { |v| {names.first => v} } if names.size == 1
500
+ nf = names.size
501
+ Array.new(k) do |i|
502
+ row = {}
503
+ j = 0
504
+ while j < nf
505
+ row[names[j]] = data[j][i]
506
+ j += 1
118
507
  end
119
- Assembler.new(field, chunks).read_rows(n)
508
+ row
120
509
  end
121
510
  end
122
511
 
512
+ # The column's value converter, with the time zone applied to timestamps
513
+ #
514
+ # @param column [Schema::Column] leaf column being read
515
+ # @return [Proc, nil] physical value => Ruby value, or nil when values are used as decoded
516
+ def converter_for(column)
517
+ base = column.converter
518
+ zc = @zone_converter
519
+ return base unless zc && base && instant_column?(column)
520
+ ->(v) { zc.call(base.call(v)) }
521
+ end
522
+
523
+ # Timestamps that denote an instant (UTC-adjusted TIMESTAMP, INT96). Local timestamps
524
+ # (isAdjustedToUTC = false) are wall-clock values and are left as they are.
525
+ #
526
+ # @param column [Schema::Column] leaf column to check
527
+ # @return [Boolean] true when the time zone applies to this column's values
528
+ def instant_column?(column)
529
+ return true if column.type == Format::Type::INT96
530
+ kind, _unit, utc = Types.logical_of(column.node)
531
+ kind == :timestamp && utc
532
+ end
533
+
534
+ # A UTC offset string: "+02", "+0200", "+02:00" or "+02:00:00" (sign required)
535
+ OFFSET_PATTERN = /\A([+-])(\d\d)(?::?(\d\d)(?::?(\d\d))?)?\z/
536
+
537
+ # A lambda turning a UTC Time into the zone, or nil for UTC
538
+ #
539
+ # @param zone [String, Integer, Object, nil] the +time_zone:+ option: a UTC offset (String or
540
+ # seconds), a zone name (ActiveSupport or TZInfo), an object responding to #at or
541
+ # #utc_to_local, or nil
542
+ # @return [Proc, nil] Time => Time (or ActiveSupport::TimeWithZone), nil for UTC
543
+ # @raise [ArgumentError] for an unknown, unsupported or out-of-range zone
544
+ def zone_converter(zone)
545
+ case zone
546
+ when nil then nil
547
+ when Integer
548
+ return nil if zone.zero?
549
+ Time.at(0).getlocal(zone) # validates the range
550
+ ->(t) { t.getlocal(zone) }
551
+ when String
552
+ return nil if zone.match?(/\A(?:utc|z)\z/i)
553
+ if (m = OFFSET_PATTERN.match(zone))
554
+ # As seconds, since Ruby 3.0 only parses "+HH:MM" offset strings
555
+ return zone_converter(((m[1] == "-") ? -1 : 1) * (m[2].to_i * 3600 + m[3].to_i * 60 + m[4].to_i))
556
+ end
557
+ if defined?(::ActiveSupport::TimeZone) && (tz = ::ActiveSupport::TimeZone[zone])
558
+ return ->(t) { tz.at(t) }
559
+ end
560
+ if defined?(::TZInfo::Timezone)
561
+ tz = ::TZInfo::Timezone.get(zone)
562
+ return ->(t) { t.getlocal(tz) }
563
+ end
564
+ raise ArgumentError, "Unknown time zone #{zone.inspect}: use a UTC offset like \"+02:00\", " \
565
+ "a timezone object (e.g. TZInfo::Timezone.get(#{zone.inspect})) or an ActiveSupport::TimeZone"
566
+ else
567
+ return ->(t) { zone.at(t) } if zone.respond_to?(:at)
568
+ if zone.respond_to?(:utc_to_local)
569
+ Time.at(0).getlocal(zone)
570
+ return ->(t) { t.getlocal(zone) }
571
+ end
572
+ raise ArgumentError, "Unsupported time_zone: #{zone.inspect}"
573
+ end
574
+ rescue ArgumentError => e
575
+ raise if e.message.start_with?("Unknown time zone", "Unsupported time_zone", "Invalid time_zone")
576
+ raise ArgumentError, "Invalid time_zone #{zone.inspect}: #{e.message}"
577
+ end
578
+
579
+ # Reads and decodes the footer: the FileMetaData Thrift struct, its 4-byte little-endian
580
+ # length and the closing magic.
581
+ #
582
+ # @return [Format::FileMetaData] the decoded footer
583
+ # @raise [FormatError] when the file is too short, lacks the magic or the footer is corrupt
584
+ # @raise [UnsupportedError] for an encrypted file (+PARE+ magic)
123
585
  def read_footer
124
586
  @io.seek(0, IO::SEEK_END)
125
587
  size = @io.pos
126
588
  raise FormatError, "File too small to be Parquet (#{size} bytes)" if size < 12
127
589
  @io.seek(size - 8)
128
590
  tail = @io.read(8)
129
- raise FormatError, "Missing PAR1 footer magic" unless tail.byteslice(4, 4) == MAGIC
130
591
  raise UnsupportedError, "Encrypted Parquet files are not supported" if tail.byteslice(4, 4) == "PARE"
592
+ raise FormatError, "Missing PAR1 footer magic" unless tail.byteslice(4, 4) == MAGIC
131
593
  footer_len = tail.unpack1("V")
132
594
  raise FormatError, "Footer length #{footer_len} exceeds file size" if footer_len + 12 > size
133
595
  @io.seek(size - 8 - footer_len)
@@ -137,199 +599,26 @@ module Herringbone
137
599
  raise FormatError, "Corrupt file metadata: #{e.message}"
138
600
  end
139
601
 
140
- # Decodes all pages of one column chunk into levels and values
141
- class ColumnChunkReader
142
- T = Format::Type
143
- E = Format::Encoding
144
-
145
- def initialize(io, chunk, column)
146
- @io = io
147
- @chunk = chunk
148
- @column = column
149
- @meta = chunk.meta_data or raise UnsupportedError, "Column chunk without metadata (encrypted?)"
150
- raise UnsupportedError, "Column chunks in external files are not supported" if chunk.file_path
151
- @max_def = column.max_definition_level
152
- @max_rep = column.max_repetition_level
153
- @converter = column.converter
154
- end
155
-
156
- def read
157
- buf = read_bytes
158
- defs = @max_def.positive? ? [] : nil
159
- reps = @max_rep.positive? ? [] : nil
160
- values = []
161
- pos = 0
162
- total = @meta.num_values
163
- seen = 0
164
- while seen < total && pos < buf.bytesize
165
- header, pos = decode_page_header(buf, pos)
166
- size = header.compressed_page_size
167
- # Some old writers under-report total_compressed_size; read on past the declared end
168
- extend_buffer(buf, pos + size - buf.bytesize) if pos + size > buf.bytesize
169
- raise FormatError, "Page overruns column chunk" if pos + size > buf.bytesize
170
- body = buf.byteslice(pos, size)
171
- pos += size
172
- case header.type
173
- when Format::PageType::DICTIONARY_PAGE
174
- read_dictionary(header, body)
175
- when Format::PageType::DATA_PAGE
176
- seen += read_data_page_v1(header, body, defs, reps, values)
177
- when Format::PageType::DATA_PAGE_V2
178
- seen += read_data_page_v2(header, body, defs, reps, values)
179
- end
180
- end
181
- raise FormatError, "Column #{@column.dotted_path}: read #{seen} of #{total} values" if seen < total
182
- [defs, reps, values]
183
- rescue Thrift::Error => e
184
- raise FormatError, "Corrupt page header in #{@column.dotted_path}: #{e.message}"
185
- end
186
-
187
- private
188
-
189
- def read_bytes
190
- start = @meta.data_page_offset
191
- dict = @meta.dictionary_page_offset
192
- # Some writers store 0 when there is no dictionary page
193
- start = dict if dict && dict.positive? && dict < start
194
- @start = start
195
- @io.seek(start)
196
- len = @meta.total_compressed_size
197
- (@io.read(len) || "".b).b
198
- end
199
-
200
- def extend_buffer(buf, nbytes)
201
- @io.seek(@start + buf.bytesize)
202
- more = @io.read(nbytes)
203
- buf << more.b if more
204
- end
205
-
206
- def decode_page_header(buf, pos)
207
- Format::PageHeader.decode(buf, pos)
208
- rescue Thrift::Error
209
- # The header may straddle the declared end of the chunk
210
- before = buf.bytesize
211
- extend_buffer(buf, 1024)
212
- raise if buf.bytesize == before
213
- Format::PageHeader.decode(buf, pos)
214
- end
215
-
216
- def decompress(body, size)
217
- Compression.decompress(@meta.codec, body, size)
218
- end
219
-
220
- def read_dictionary(header, body)
221
- dh = header.dictionary_page_header
222
- data = decompress(body, header.uncompressed_page_size)
223
- vals, = Encodings::Plain.decode(data, 0, dh.num_values, @column.type, @column.type_length)
224
- vals.map!(&@converter) if @converter
225
- @dictionary = vals
226
- end
227
-
228
- def read_data_page_v1(header, body, defs, reps, values)
229
- dh = header.data_page_header
230
- n = dh.num_values
231
- data = decompress(body, header.uncompressed_page_size)
232
- pos = 0
233
- if @max_rep.positive?
234
- levels, pos = read_levels(data, pos, dh.repetition_level_encoding, @max_rep, n)
235
- reps.concat(levels)
236
- end
237
- non_null = n
238
- if @max_def.positive?
239
- levels, pos = read_levels(data, pos, dh.definition_level_encoding, @max_def, n)
240
- non_null = levels.count(@max_def)
241
- defs.concat(levels)
242
- end
243
- values.concat(decode_values(data, pos, non_null, dh.encoding))
244
- n
245
- end
246
-
247
- def read_data_page_v2(header, body, defs, reps, values)
248
- dh = header.data_page_header_v2
249
- n = dh.num_values
250
- rep_len = dh.repetition_levels_byte_length
251
- def_len = dh.definition_levels_byte_length
252
- if @max_rep.positive?
253
- reps.concat(Encodings::RLE.decode_hybrid(body, 0, rep_len, RLE_WIDTH[@max_rep], n))
254
- end
255
- non_null = n
256
- if @max_def.positive?
257
- levels = Encodings::RLE.decode_hybrid(body, rep_len, rep_len + def_len, RLE_WIDTH[@max_def], n)
258
- non_null = levels.count(@max_def)
259
- defs.concat(levels)
260
- end
261
- data = body.byteslice(rep_len + def_len, body.bytesize - rep_len - def_len)
262
- if dh.is_compressed != false
263
- data = decompress(data, header.uncompressed_page_size - rep_len - def_len)
264
- end
265
- values.concat(decode_values(data, 0, non_null, dh.encoding))
266
- n
267
- end
268
-
269
- RLE_WIDTH = Hash.new { |h, k| h[k] = k.bit_length }
270
-
271
- def read_levels(data, pos, encoding, max, n)
272
- width = RLE_WIDTH[max]
273
- case encoding
274
- when E::RLE
275
- len = data.byteslice(pos, 4).unpack1("V")
276
- start = pos + 4
277
- [Encodings::RLE.decode_hybrid(data, start, start + len, width, n), start + len]
278
- when E::BIT_PACKED
279
- [Encodings::RLE.decode_legacy_bit_packed(data, pos, width, n), pos + (n * width + 7) / 8]
280
- else
281
- raise UnsupportedError, "Unsupported level encoding #{E::NAMES[encoding] || encoding}"
282
- end
283
- end
284
-
285
- def decode_values(data, pos, count, encoding)
286
- type = @column.type
287
- vals = case encoding
288
- when E::PLAIN
289
- Encodings::Plain.decode(data, pos, count, type, @column.type_length).first
290
- when E::PLAIN_DICTIONARY, E::RLE_DICTIONARY
291
- raise FormatError, "Dictionary-encoded page without a dictionary in #{@column.dotted_path}" unless @dictionary
292
- return [] if count.zero?
293
- width = data.getbyte(pos)
294
- indices = Encodings::RLE.decode_hybrid(data, pos + 1, data.bytesize, width, count)
295
- dict = @dictionary
296
- raise FormatError, "Dictionary index out of range in #{@column.dotted_path}" if indices.max >= dict.size
297
- return indices.map! { |i| dict[i] }
298
- when E::RLE
299
- raise UnsupportedError, "RLE value encoding is only supported for BOOLEAN" unless type == T::BOOLEAN
300
- len = data.byteslice(pos, 4).unpack1("V")
301
- Encodings::RLE.decode_hybrid(data, pos + 4, pos + 4 + len, 1, count).map! { |v| v == 1 }
302
- when E::DELTA_BINARY_PACKED
303
- bits = type == T::INT32 ? 32 : 64
304
- vals, = Encodings::Delta.decode_binary_packed(data, pos, bits, count)
305
- raise FormatError, "DELTA_BINARY_PACKED page has #{vals.size} values, need #{count}" if vals.size < count
306
- vals
307
- when E::DELTA_LENGTH_BYTE_ARRAY
308
- Encodings::Delta.decode_length_byte_array(data, pos, count).first
309
- when E::DELTA_BYTE_ARRAY
310
- Encodings::Delta.decode_byte_array(data, pos, count).first
311
- when E::BYTE_STREAM_SPLIT
312
- width = case type
313
- when T::INT32, T::FLOAT then 4
314
- when T::INT64, T::DOUBLE then 8
315
- when T::FIXED_LEN_BYTE_ARRAY then @column.type_length
316
- else raise UnsupportedError, "BYTE_STREAM_SPLIT is not valid for #{T::NAMES[type]}"
317
- end
318
- plain, = Encodings::ByteStreamSplit.decode(data, pos, count, width)
319
- Encodings::Plain.decode(plain, 0, count, type, @column.type_length).first
320
- else
321
- raise UnsupportedError, "Unsupported encoding #{E::NAMES[encoding] || encoding}"
322
- end
323
- vals.map!(&@converter) if @converter
324
- vals
325
- end
326
- end
327
-
328
602
  # Rebuilds nested values of one top-level field from the levels of its leaf columns
329
603
  # (the "record assembly" half of the Dremel algorithm).
330
604
  class Assembler
331
- def initialize(field, chunks)
605
+ # @param field [Schema::Field] top-level field to assemble
606
+ # @param symbolize [Boolean] whether struct Hashes are keyed by Symbol instead of String
607
+ def initialize(field, symbolize = false)
332
608
  @field = field
609
+ @symbolize = symbolize
610
+ @keys = {} # struct child Field => Hash key
611
+ end
612
+
613
+ # Assembles +n+ values of the field from +chunks+ (leaf column index =>
614
+ # [defs, reps, values] holding exactly those rows)
615
+ #
616
+ # @param n [Integer] number of rows in +chunks+
617
+ # @param chunks [Hash{Integer => Array(Array<Integer>, Array<Integer>, Array)}] leaf column
618
+ # index => [definition levels, repetition levels, values]; levels may be nil
619
+ # @return [Array] +n+ assembled values (nil, scalars, Arrays, Hashes)
620
+ # @raise [FormatError] when the levels do not add up to +n+ rows
621
+ def read_rows(n, chunks)
333
622
  @defs = {}
334
623
  @reps = {}
335
624
  @vals = {}
@@ -340,9 +629,7 @@ module Herringbone
340
629
  end
341
630
  @ei = Hash.new(0) # entry cursor per leaf column
342
631
  @vi = Hash.new(0) # value cursor per leaf column
343
- end
344
632
 
345
- def read_rows(n)
346
633
  f = @field
347
634
  if f.leaf? && f.column.max_repetition_level.zero?
348
635
  idx = f.column.index
@@ -352,7 +639,7 @@ module Herringbone
352
639
  max = f.column.max_definition_level
353
640
  return vals if vals.size == defs.size
354
641
  vi = -1
355
- return defs.map { |d| d == max ? vals[vi += 1] : nil }
642
+ return defs.map { |d| (d == max) ? vals[vi += 1] : nil }
356
643
  end
357
644
  return read_simple_list(f) if f.kind == :list && f.element.leaf? && f.element.column.max_repetition_level == 1
358
645
 
@@ -368,6 +655,9 @@ module Herringbone
368
655
  private
369
656
 
370
657
  # Fast path for a top-level list of primitives (the most common nested shape)
658
+ #
659
+ # @param field [Schema::Field] list field whose element is a leaf with repetition level 1
660
+ # @return [Array<Array, nil>] one list (or nil) per row
371
661
  def read_simple_list(field)
372
662
  col = field.element.column
373
663
  defs = @defs[col.index]
@@ -407,6 +697,19 @@ module Herringbone
407
697
  out
408
698
  end
409
699
 
700
+ # Hash key for a struct member, memoized
701
+ #
702
+ # @param field [Schema::Field] struct child
703
+ # @return [String, Symbol] frozen name, or Symbol when symbolizing
704
+ def key_for(field)
705
+ @keys[field] ||= @symbolize ? field.name.to_sym : -field.name
706
+ end
707
+
708
+ # Assembles one value of +field+ at the current entry cursors, recursing into children
709
+ #
710
+ # @param field [Schema::Field] field to assemble
711
+ # @return [Object, nil] scalar, Hash (struct, map), Array (list) or nil
712
+ # @raise [FormatError] when the definition levels run out
410
713
  def read(field)
411
714
  c = field.first_leaf.index
412
715
  kind = field.kind
@@ -427,7 +730,7 @@ module Herringbone
427
730
  @vals[c][v]
428
731
  when :struct
429
732
  h = {}
430
- field.children.each { |ch| h[ch.name] = read(ch) }
733
+ field.children.each { |ch| h[key_for(ch)] = read(ch) }
431
734
  h
432
735
  when :list
433
736
  if d < field.item_def
@@ -461,9 +764,19 @@ module Herringbone
461
764
  end
462
765
  end
463
766
 
767
+ # Moves every leaf of +field+ past one entry (a null or empty value takes a single entry)
768
+ #
769
+ # @param field [Schema::Field] field whose leaves to advance
770
+ # @return [void]
464
771
  def skip(field)
465
772
  field.leaves.each { |col| @ei[col.index] += 1 }
466
773
  end
467
774
  end
468
775
  end
469
776
  end
777
+
778
+ require_relative "reader/page_stream"
779
+ require_relative "reader/column_chunk_reader"
780
+ require_relative "reader/column_cursor"
781
+ require_relative "reader/scan"
782
+ require_relative "reader/numo"