herringbone 0.2.0 → 0.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -22,18 +22,33 @@ module Herringbone
22
22
  # TZInfo::Timezone), or anything responding to #at such as an
23
23
  # ActiveSupport::TimeZone (Time.zone), which yields ActiveSupport::TimeWithZone.
24
24
  class Reader
25
+ # The 4 bytes a Parquet file starts and ends with
25
26
  MAGIC = "PAR1"
27
+ # Rows per batch in #each_batch when no size is given
26
28
  DEFAULT_BATCH_SIZE = 1024
29
+ # Accepted values of the +keys:+ option
27
30
  KEY_MODES = %i[string symbol].freeze
31
+ # Accepted values of the +as:+ option
32
+ AS_MODES = %i[rows columns numo].freeze
28
33
 
29
- # The schema, and the file's FileMetaData (the decoded Thrift footer)
30
- attr_reader :schema, :file_metadata
34
+ # @return [Schema] the schema, built from the footer's flattened SchemaElements
35
+ attr_reader :schema
36
+
37
+ # @return [Format::FileMetaData] the file's FileMetaData (the decoded Thrift footer)
38
+ attr_reader :file_metadata
31
39
 
32
40
  # +io+ must support #seek and #read (a File opened with "rb", StringIO, Tempfile...)
41
+ #
42
+ # @param io [IO, StringIO] random-access source of the Parquet bytes; the caller closes it
43
+ # @param keys [Symbol, String] +:string+ or +:symbol+, the key type of row and struct Hashes
44
+ # @param time_zone [String, Integer, Object, nil] zone timestamps are returned in (see the
45
+ # class docs); nil keeps them in UTC
46
+ # @raise [ArgumentError] when +io+ cannot seek and read, or +keys+ / +time_zone+ are invalid
47
+ # @raise [FormatError] when the footer is missing or cannot be decoded
33
48
  def initialize(io, keys: :string, time_zone: nil)
34
49
  unless io.respond_to?(:seek) && io.respond_to?(:read)
35
50
  raise ArgumentError, "Herringbone::Reader expects an IO that supports #seek and #read " \
36
- "(e.g. File.open(path, \"rb\")), got #{io.class == String ? "a String" : io.class}" \
51
+ "(e.g. File.open(path, \"rb\")), got #{io.is_a?(String) ? "a String" : io.class}" \
37
52
  "#{" (wrap Parquet bytes in a StringIO)" if io.is_a?(String)}"
38
53
  end
39
54
  keys = keys.to_sym if keys.is_a?(String)
@@ -45,10 +60,19 @@ module Herringbone
45
60
  @schema = Schema.from_elements(@file_metadata.schema)
46
61
  end
47
62
 
63
+ # Total number of rows, as stated in the footer (some writers store 0)
64
+ #
65
+ # @return [Integer] the footer's num_rows
48
66
  def num_rows = @file_metadata.num_rows
67
+
68
+ # The row groups listed in the footer
69
+ #
70
+ # @return [Array<Format::RowGroup>] row group metadata, in file order
49
71
  def row_groups = @file_metadata.row_groups
50
72
 
51
73
  # The footer's key/value metadata as a Hash (what the writer's metadata: option stores)
74
+ #
75
+ # @return [Hash{String => String, nil}] key => value; empty when the footer has none
52
76
  def metadata
53
77
  (@file_metadata.key_value_metadata || []).to_h { |kv| [kv.key, kv.value] }
54
78
  end
@@ -58,24 +82,47 @@ module Herringbone
58
82
  #
59
83
  # as: :rows (default) yields an Array of row Hashes; as: :columns yields a Hash of top-level
60
84
  # field name => Array of that field's values in the batch, which skips building a Hash per row
61
- # and is noticeably faster when you process data column by column.
85
+ # and is noticeably faster when you process data column by column. as: :numo yields a Hash of
86
+ # field name => Numo array (needs the numo-narray-alt or numo-narray gem; see NumoColumns
87
+ # for the type mapping). With as: :numo, whether an integer column becomes DFloat (nulls) or
88
+ # a list column 2-D is decided per batch, from the values in it.
62
89
  #
63
90
  # where: only yields rows matching all conditions (see Reader::Filter). Row groups and pages
64
91
  # that cannot match are skipped using statistics, bloom filters and the page index, and the
65
92
  # remaining rows are checked one by one. Filtered columns need not be in +columns+.
66
93
  # from: skips the first rows of the file (jumping over pages with the page index), and
67
94
  # limit: stops after yielding that many rows.
95
+ #
96
+ # @param size [Integer] maximum number of rows per batch
97
+ # @param columns [Array<String, Symbol>, String, Symbol, nil] top-level fields to return;
98
+ # nil returns all of them
99
+ # @param as [Symbol] +:rows+, +:columns+ or +:numo+, the shape of each batch
100
+ # @param where [Hash{String, Symbol => Object}, nil] column => condition, see Filter
101
+ # @param from [Integer, nil] number of rows at the start of the file to skip
102
+ # @param limit [Integer, nil] maximum number of rows to yield in total
103
+ # @yield [batch] once per batch
104
+ # @yieldparam batch [Array<Hash>, Hash{String, Symbol => Array}, Hash{String, Symbol => Numo::NArray}]
105
+ # row Hashes (+:rows+), or field name => values of the batch (+:columns+, +:numo+)
106
+ # @yieldreturn [void]
107
+ # @return [Reader, Enumerator] self, or an Enumerator of batches when no block is given
108
+ # @raise [ArgumentError] for a non-positive +size+, unknown +as+, negative +from+ / +limit+,
109
+ # or an unknown column in +columns+ or +where+
110
+ # @raise [UnsupportedError] with +as: :numo+ when no Numo gem can be loaded
68
111
  def each_batch(size = DEFAULT_BATCH_SIZE, columns: nil, as: :rows, where: nil, from: nil, limit: nil)
69
112
  return enum_for(:each_batch, size, columns: columns, as: as, where: where, from: from, limit: limit) unless block_given?
70
113
  size = Integer(size)
71
114
  raise ArgumentError, "Batch size must be positive, got #{size}" unless size.positive?
72
- raise ArgumentError, "as: must be :rows or :columns, got #{as.inspect}" unless as == :rows || as == :columns
73
- raise ArgumentError, "limit: must not be negative" if limit && limit.negative?
74
- return self if limit&.zero?
115
+ raise ArgumentError, "as: must be :rows, :columns or :numo, got #{as.inspect}" unless AS_MODES.include?(as)
116
+ raise ArgumentError, "limit: must not be negative" if limit&.negative?
117
+ return validate_read_options(columns, where, from) if limit&.zero?
118
+ if as == :numo
119
+ each_numo_batch(size, columns, where, from, limit) { |batch| yield batch }
120
+ return self
121
+ end
75
122
  columnar = as == :columns
76
123
  symbolize = @symbolize
77
124
  out_fields = select_fields(columns)
78
- filter = where && !where.empty? ? Filter.new(@schema, where) : nil
125
+ filter = (where && !where.empty?) ? Filter.new(@schema, where) : nil
79
126
  fields = filter ? out_fields | filter.fields : out_fields
80
127
  names = row_keys(out_fields, symbolize)
81
128
  nout = out_fields.size
@@ -157,15 +204,32 @@ module Herringbone
157
204
  # What a read with +where:+ / +from:+ would touch, without reading any data: an Array of
158
205
  # { row_group:, rows:, ranges: [[first_row, end_row), ...] } for the row groups that are
159
206
  # read. Row groups ruled out entirely are left out.
207
+ #
208
+ # @param where [Hash{String, Symbol => Object}, nil] column => condition, as in #each_batch
209
+ # @param from [Integer, nil] number of rows at the start of the file to skip
210
+ # @return [Array<Hash{Symbol => Object}>] +{row_group: Integer, rows: Integer,
211
+ # ranges: Array<Array(Integer, Integer)>}+ per row group that would be read
212
+ # @raise [ArgumentError] for a negative +from+ or an unknown column in +where+
160
213
  def scan_plan(where: nil, from: nil)
161
- filter = where && !where.empty? ? Filter.new(@schema, where) : nil
214
+ filter = (where && !where.empty?) ? Filter.new(@schema, where) : nil
162
215
  plan_rows(filter, from).map do |rg_index, ranges|
163
- { row_group: rg_index, rows: ranges.sum { |s, e| e - s }, ranges: ranges }
216
+ {row_group: rg_index, rows: ranges.sum { |s, e| e - s }, ranges: ranges}
164
217
  end
165
218
  end
166
219
 
167
220
  # Yields each row as a Hash of top-level field name => value. Takes the options of each_batch
168
221
  # except as:.
222
+ #
223
+ # @param columns [Array<String, Symbol>, String, Symbol, nil] top-level fields to return;
224
+ # nil returns all of them
225
+ # @param where [Hash{String, Symbol => Object}, nil] column => condition, see Filter
226
+ # @param from [Integer, nil] number of rows at the start of the file to skip
227
+ # @param limit [Integer, nil] maximum number of rows to yield
228
+ # @yield [row] once per row
229
+ # @yieldparam row [Hash{String, Symbol => Object}] top-level field name => value
230
+ # @yieldreturn [void]
231
+ # @return [Reader, Enumerator] self, or an Enumerator of rows when no block is given
232
+ # @raise [ArgumentError] for a negative +from+ / +limit+ or an unknown column
169
233
  def each_row(columns: nil, where: nil, from: nil, limit: nil, &block)
170
234
  return enum_for(:each_row, columns: columns, where: where, from: from, limit: limit) unless block
171
235
  each_batch(columns: columns, where: where, from: from, limit: limit) { |rows| rows.each(&block) }
@@ -173,9 +237,32 @@ module Herringbone
173
237
  end
174
238
 
175
239
  # Reads the whole file (or the selected rows) at once: an Array of row Hashes, or with
176
- # as: :columns a Hash of top-level field name => Array of values. Takes the options of each_batch.
240
+ # as: :columns a Hash of top-level field name => Array of values, or with as: :numo a Hash of
241
+ # top-level field name => Numo array. Takes the options of each_batch.
242
+ #
243
+ # @param columns [Array<String, Symbol>, String, Symbol, nil] top-level fields to return;
244
+ # nil returns all of them
245
+ # @param as [Symbol] +:rows+, +:columns+ or +:numo+, the shape of the result
246
+ # @param where [Hash{String, Symbol => Object}, nil] column => condition, see Filter
247
+ # @param from [Integer, nil] number of rows at the start of the file to skip
248
+ # @param limit [Integer, nil] maximum number of rows to return
249
+ # @return [Array<Hash>, Hash{String, Symbol => Array}, Hash{String, Symbol => Numo::NArray}]
250
+ # row Hashes (+:rows+), or field name => all values (+:columns+, +:numo+)
251
+ # @raise [ArgumentError] for an unknown +as+, negative +from+ / +limit+ or an unknown column
252
+ # @raise [UnsupportedError] with +as: :numo+ when no Numo gem can be loaded
177
253
  def read(columns: nil, as: :rows, where: nil, from: nil, limit: nil)
178
- if as == :columns
254
+ if as == :numo
255
+ raise ArgumentError, "limit: must not be negative" if limit&.negative?
256
+ out = nil
257
+ if limit&.zero?
258
+ validate_read_options(columns, where, from)
259
+ else
260
+ # One batch (num_rows is not trusted: some writers store 0), so the column types are
261
+ # decided from all the rows read
262
+ each_numo_batch(1 << 62, columns, where, from, limit) { |batch| out = batch }
263
+ end
264
+ out ||= numo_empty(columns)
265
+ elsif as == :columns
179
266
  out = row_keys(select_fields(columns), @symbolize).to_h { |name| [name, []] }
180
267
  each_batch(65_536, columns: columns, as: :columns, where: where, from: from, limit: limit) do |batch|
181
268
  batch.each { |name, values| out[name].concat(values) }
@@ -188,7 +275,13 @@ module Herringbone
188
275
  end
189
276
 
190
277
  # Internal (used by reads with where:/from:): [ColumnIndex or nil, OffsetIndex or nil] of a
191
- # leaf column in a row group
278
+ # leaf column in a row group. Cached per chunk; a missing or damaged index reads as nil.
279
+ #
280
+ # @param row_group_index [Integer] position of the row group in the footer
281
+ # @param column [Schema::Column, String, Array<String>] leaf column, or its dotted / Array path
282
+ # @return [Array(Format::ColumnIndex, Format::OffsetIndex)] either element may be nil
283
+ # @raise [ArgumentError] when +column+ does not name a leaf column
284
+ # @raise [IndexError] when the row group does not exist
192
285
  def page_index(row_group_index, column)
193
286
  column = @schema.column(column) unless column.is_a?(Schema::Column)
194
287
  raise ArgumentError, "No such leaf column" unless column
@@ -200,14 +293,146 @@ module Herringbone
200
293
  end
201
294
  end
202
295
 
296
+ # Short summary for the console, without the schema
297
+ #
298
+ # @return [String] row count, row group count and the writer's created_by
203
299
  def inspect
204
300
  "#<#{self.class.name} rows=#{num_rows} row_groups=#{row_groups.size} created_by=#{@file_metadata.created_by.inspect}>"
205
301
  end
206
302
 
207
303
  private
208
304
 
305
+ # as: :numo. Flat numeric/boolean output columns go through NumoCursors (no Ruby object per
306
+ # value); the other output columns, and every column a where: filter needs, are assembled as
307
+ # Ruby values like as: :columns and converted when a batch is complete.
308
+ #
309
+ # @param size [Integer] maximum number of rows per batch
310
+ # @param columns [Array<String, Symbol>, String, Symbol, nil] top-level fields to return
311
+ # @param where [Hash{String, Symbol => Object}, nil] column => condition, see Filter
312
+ # @param from [Integer, nil] number of rows at the start of the file to skip
313
+ # @param limit [Integer, nil] maximum number of rows to yield in total
314
+ # @yield [batch] once per batch
315
+ # @yieldparam batch [Hash{String, Symbol => Numo::NArray}] field name => values of the batch
316
+ # @yieldreturn [void]
317
+ # @return [void]
318
+ # @raise [UnsupportedError] when no Numo gem can be loaded
319
+ def each_numo_batch(size, columns, where, from, limit)
320
+ NumoColumns.load!
321
+ symbolize = @symbolize
322
+ out_fields = select_fields(columns)
323
+ filter = (where && !where.empty?) ? Filter.new(@schema, where) : nil
324
+ filter_fields = filter ? filter.fields : []
325
+ names = row_keys(out_fields, symbolize)
326
+ specs = out_fields.map { |f| NumoColumns.spec_for(f) }
327
+ fast = out_fields.each_index.select { |j| specs[j].fast? && !filter_fields.include?(out_fields[j]) }
328
+ ruby_fields = (out_fields.each_index.to_a - fast).map { |j| out_fields[j] } | filter_fields
329
+ # Where each output column comes from: index into the fast cursors or the Ruby fields
330
+ sources = out_fields.each_index.map { |j| (i = fast.index(j)) ? [:fast, i] : [:ruby, ruby_fields.index(out_fields[j])] }
331
+ converters = @schema.columns.map { |col| converter_for(col) }
332
+ assemblers = ruby_fields.map { |f| Assembler.new(f, symbolize) }
333
+ # Rows decoded per step: Ruby values are kept to slices of the batch, Numo arrays need not be
334
+ step = ruby_fields.empty? ? 1 << 20 : 65_536
335
+ left_to_yield = limit
336
+ pending = Array.new(out_fields.size) { [] }
337
+ pending_rows = 0
338
+ flush = lambda do
339
+ batch = sources.each_with_index.map do |(kind, _), j|
340
+ (kind == :fast) ? NumoColumns.finish_fixed(specs[j], pending[j]) : NumoColumns.finish_values(specs[j], pending[j])
341
+ end
342
+ pending = Array.new(out_fields.size) { [] }
343
+ pending_rows = 0
344
+ yield names.zip(batch).to_h
345
+ end
346
+
347
+ plan_rows(filter, from).each do |rg_index, ranges|
348
+ rg = row_groups[rg_index]
349
+ partial = ranges != [[0, rg.num_rows]]
350
+ open = lambda do |col|
351
+ reader = ColumnChunkReader.new(@io, rg.columns.fetch(col.index), col, converter: converters[col.index], lazy: true)
352
+ reader.locations = page_index(rg_index, col)[1]&.page_locations if partial
353
+ reader
354
+ end
355
+ ruby_cursors = ruby_fields.map { |f| f.leaves.map { |col| [col.index, ColumnCursor.new(open.call(col))] } }
356
+ fast_cursors = fast.map { |j| NumoCursor.new(open.call(out_fields[j].column), specs[j]) }
357
+ ranges.each do |first, stop|
358
+ ruby_cursors.each { |cs| cs.each { |_, cursor| cursor.seek(first) } }
359
+ fast_cursors.each { |cursor| cursor.seek(first) }
360
+ left = stop - first
361
+ while left.positive?
362
+ k = size - pending_rows
363
+ raise Error, "Internal error: batch needs #{k} rows" unless k.positive? # never loop without progress
364
+ k = step if k > step
365
+ k = left if left < k
366
+ data = assemblers.each_with_index.map do |asm, j|
367
+ asm.read_rows(k, ruby_cursors[j].to_h { |idx, cursor| [idx, cursor.take(k)] })
368
+ end
369
+ numo = fast_cursors.map { |cursor| cursor.take(k) }
370
+ left -= k
371
+ kept = k
372
+ if filter
373
+ keep = filter.matching_rows(data, ruby_fields, k, symbolize)
374
+ kept = keep.size
375
+ next if kept.zero?
376
+ if kept < k
377
+ data = data.map { |col| keep.map { |i| col[i] } }
378
+ index = Numo::Int64.cast(keep)
379
+ numo = numo.map { |values, valid| [values[index].dup, valid && valid[index].dup] }
380
+ end
381
+ end
382
+ if left_to_yield && kept > left_to_yield
383
+ kept = left_to_yield
384
+ data = data.map { |col| col.first(kept) }
385
+ numo = numo.map { |values, valid| [values[0...kept].dup, valid && valid[0...kept].dup] }
386
+ end
387
+ sources.each_with_index do |(kind, i), j|
388
+ pending[j] << ((kind == :fast) ? numo[i] : data[i])
389
+ end
390
+ pending_rows += kept
391
+ left_to_yield -= kept if left_to_yield
392
+ flush.call if pending_rows >= size || left_to_yield&.zero?
393
+ return if left_to_yield&.zero? # standard:disable Lint/NonLocalExitFromIterator
394
+ end
395
+ end
396
+ end
397
+ flush.call if pending_rows.positive?
398
+ end
399
+
400
+ # Checks the columns:, where: and from: options of a read that returns no rows (limit: 0),
401
+ # so it raises for bad options like any other read
402
+ #
403
+ # @param columns [Array<String, Symbol>, String, Symbol, nil] top-level fields to return
404
+ # @param where [Hash{String, Symbol => Object}, nil] column => condition, see Filter
405
+ # @param from [Integer, nil] number of rows at the start of the file to skip
406
+ # @return [Reader] self
407
+ # @raise [ArgumentError] for an unknown column, a bad condition or a negative +from+
408
+ def validate_read_options(columns, where, from)
409
+ select_fields(columns)
410
+ Filter.new(@schema, where) if where && !where.empty?
411
+ raise ArgumentError, "from: must not be negative" if Integer(from || 0).negative?
412
+ self
413
+ end
414
+
415
+ # read(as: :numo) of no rows: an empty array of each column's type
416
+ #
417
+ # @param columns [Array<String, Symbol>, String, Symbol, nil] top-level fields to return
418
+ # @return [Hash{String, Symbol => Numo::NArray}] field name => zero-length Numo array
419
+ def numo_empty(columns)
420
+ NumoColumns.load!
421
+ fields = select_fields(columns)
422
+ row_keys(fields, @symbolize).zip(fields.map { |f|
423
+ spec = NumoColumns.spec_for(f)
424
+ (spec.kind == :fixed) ? spec.klass.new(0) : Numo::RObject.new(0)
425
+ }).to_h
426
+ end
427
+
209
428
  # [[row_group_index, [[first_row, end_row), ...]], ...] to read, after ruling out row groups
210
429
  # (statistics, bloom filters) and pages (page index), and skipping the first +from+ rows
430
+ #
431
+ # @param filter [Filter, nil] the where: conditions, if any
432
+ # @param from [Integer, nil] number of rows at the start of the file to skip
433
+ # @return [Array<Array(Integer, Array<Array(Integer, Integer)>)>] row group index and its
434
+ # half-open row ranges, for row groups that have rows left to read
435
+ # @raise [ArgumentError] for a negative +from+
211
436
  def plan_rows(filter, from)
212
437
  from = Integer(from || 0)
213
438
  raise ArgumentError, "from: must not be negative" if from.negative?
@@ -228,6 +453,12 @@ module Herringbone
228
453
  plan
229
454
  end
230
455
 
456
+ # Decodes one Thrift struct stored elsewhere in the file (ColumnIndex, OffsetIndex)
457
+ #
458
+ # @param klass [Class] Format struct class to decode with (responds to .decode)
459
+ # @param offset [Integer, nil] file offset of the struct
460
+ # @param length [Integer, nil] byte length of the struct
461
+ # @return [Object, nil] the decoded +klass+ instance, or nil when absent, truncated or corrupt
231
462
  def read_struct(klass, offset, length)
232
463
  return nil unless offset && length&.positive?
233
464
  @io.seek(offset)
@@ -238,18 +469,34 @@ module Herringbone
238
469
  nil # a damaged index only means pages cannot be skipped
239
470
  end
240
471
 
472
+ # The top-level fields named by a +columns:+ option
473
+ #
474
+ # @param columns [Array<String, Symbol>, String, Symbol, nil] field names; nil selects all
475
+ # @return [Array<Schema::Field>] the fields, in the order requested
476
+ # @raise [ArgumentError] when a name is not a top-level field
241
477
  def select_fields(columns)
242
478
  return @schema.fields unless columns
243
479
  Array(columns).map { |c| @schema.field(c) or raise ArgumentError, "No such column #{c.inspect}" }
244
480
  end
245
481
 
482
+ # Hash keys for the given fields (frozen, deduplicated Strings, or Symbols)
483
+ #
484
+ # @param fields [Array<Schema::Field>] top-level fields being returned
485
+ # @param symbolize [Boolean] whether to key by Symbol instead of String
486
+ # @return [Array<String>, Array<Symbol>] one key per field
246
487
  def row_keys(fields, symbolize)
247
488
  fields.map { |f| symbolize ? f.name.to_sym : -f.name }
248
489
  end
249
490
 
491
+ # Turns column-wise batch data into row Hashes
492
+ #
493
+ # @param names [Array<String>, Array<Symbol>] Hash key per output field
494
+ # @param data [Array<Array>] values per output field, each holding +k+ entries
495
+ # @param k [Integer] number of rows in the batch
496
+ # @return [Array<Hash>] +k+ row Hashes
250
497
  def build_rows(names, data, k)
251
498
  return Array.new(k) { {} } if names.empty?
252
- return data.first.map { |v| { names.first => v } } if names.size == 1
499
+ return data.first.map { |v| {names.first => v} } if names.size == 1
253
500
  nf = names.size
254
501
  Array.new(k) do |i|
255
502
  row = {}
@@ -263,6 +510,9 @@ module Herringbone
263
510
  end
264
511
 
265
512
  # The column's value converter, with the time zone applied to timestamps
513
+ #
514
+ # @param column [Schema::Column] leaf column being read
515
+ # @return [Proc, nil] physical value => Ruby value, or nil when values are used as decoded
266
516
  def converter_for(column)
267
517
  base = column.converter
268
518
  zc = @zone_converter
@@ -272,15 +522,25 @@ module Herringbone
272
522
 
273
523
  # Timestamps that denote an instant (UTC-adjusted TIMESTAMP, INT96). Local timestamps
274
524
  # (isAdjustedToUTC = false) are wall-clock values and are left as they are.
525
+ #
526
+ # @param column [Schema::Column] leaf column to check
527
+ # @return [Boolean] true when the time zone applies to this column's values
275
528
  def instant_column?(column)
276
529
  return true if column.type == Format::Type::INT96
277
530
  kind, _unit, utc = Types.logical_of(column.node)
278
531
  kind == :timestamp && utc
279
532
  end
280
533
 
534
+ # A UTC offset string: "+02", "+0200", "+02:00" or "+02:00:00" (sign required)
281
535
  OFFSET_PATTERN = /\A([+-])(\d\d)(?::?(\d\d)(?::?(\d\d))?)?\z/
282
536
 
283
537
  # A lambda turning a UTC Time into the zone, or nil for UTC
538
+ #
539
+ # @param zone [String, Integer, Object, nil] the +time_zone:+ option: a UTC offset (String or
540
+ # seconds), a zone name (ActiveSupport or TZInfo), an object responding to #at or
541
+ # #utc_to_local, or nil
542
+ # @return [Proc, nil] Time => Time (or ActiveSupport::TimeWithZone), nil for UTC
543
+ # @raise [ArgumentError] for an unknown, unsupported or out-of-range zone
284
544
  def zone_converter(zone)
285
545
  case zone
286
546
  when nil then nil
@@ -292,7 +552,7 @@ module Herringbone
292
552
  return nil if zone.match?(/\A(?:utc|z)\z/i)
293
553
  if (m = OFFSET_PATTERN.match(zone))
294
554
  # As seconds, since Ruby 3.0 only parses "+HH:MM" offset strings
295
- return zone_converter((m[1] == "-" ? -1 : 1) * (m[2].to_i * 3600 + m[3].to_i * 60 + m[4].to_i))
555
+ return zone_converter(((m[1] == "-") ? -1 : 1) * (m[2].to_i * 3600 + m[3].to_i * 60 + m[4].to_i))
296
556
  end
297
557
  if defined?(::ActiveSupport::TimeZone) && (tz = ::ActiveSupport::TimeZone[zone])
298
558
  return ->(t) { tz.at(t) }
@@ -316,14 +576,20 @@ module Herringbone
316
576
  raise ArgumentError, "Invalid time_zone #{zone.inspect}: #{e.message}"
317
577
  end
318
578
 
579
+ # Reads and decodes the footer: the FileMetaData Thrift struct, its 4-byte little-endian
580
+ # length and the closing magic.
581
+ #
582
+ # @return [Format::FileMetaData] the decoded footer
583
+ # @raise [FormatError] when the file is too short, lacks the magic or the footer is corrupt
584
+ # @raise [UnsupportedError] for an encrypted file (+PARE+ magic)
319
585
  def read_footer
320
586
  @io.seek(0, IO::SEEK_END)
321
587
  size = @io.pos
322
588
  raise FormatError, "File too small to be Parquet (#{size} bytes)" if size < 12
323
589
  @io.seek(size - 8)
324
590
  tail = @io.read(8)
325
- raise FormatError, "Missing PAR1 footer magic" unless tail.byteslice(4, 4) == MAGIC
326
591
  raise UnsupportedError, "Encrypted Parquet files are not supported" if tail.byteslice(4, 4) == "PARE"
592
+ raise FormatError, "Missing PAR1 footer magic" unless tail.byteslice(4, 4) == MAGIC
327
593
  footer_len = tail.unpack1("V")
328
594
  raise FormatError, "Footer length #{footer_len} exceeds file size" if footer_len + 12 > size
329
595
  @io.seek(size - 8 - footer_len)
@@ -336,6 +602,8 @@ module Herringbone
336
602
  # Rebuilds nested values of one top-level field from the levels of its leaf columns
337
603
  # (the "record assembly" half of the Dremel algorithm).
338
604
  class Assembler
605
+ # @param field [Schema::Field] top-level field to assemble
606
+ # @param symbolize [Boolean] whether struct Hashes are keyed by Symbol instead of String
339
607
  def initialize(field, symbolize = false)
340
608
  @field = field
341
609
  @symbolize = symbolize
@@ -344,6 +612,12 @@ module Herringbone
344
612
 
345
613
  # Assembles +n+ values of the field from +chunks+ (leaf column index =>
346
614
  # [defs, reps, values] holding exactly those rows)
615
+ #
616
+ # @param n [Integer] number of rows in +chunks+
617
+ # @param chunks [Hash{Integer => Array(Array<Integer>, Array<Integer>, Array)}] leaf column
618
+ # index => [definition levels, repetition levels, values]; levels may be nil
619
+ # @return [Array] +n+ assembled values (nil, scalars, Arrays, Hashes)
620
+ # @raise [FormatError] when the levels do not add up to +n+ rows
347
621
  def read_rows(n, chunks)
348
622
  @defs = {}
349
623
  @reps = {}
@@ -365,7 +639,7 @@ module Herringbone
365
639
  max = f.column.max_definition_level
366
640
  return vals if vals.size == defs.size
367
641
  vi = -1
368
- return defs.map { |d| d == max ? vals[vi += 1] : nil }
642
+ return defs.map { |d| (d == max) ? vals[vi += 1] : nil }
369
643
  end
370
644
  return read_simple_list(f) if f.kind == :list && f.element.leaf? && f.element.column.max_repetition_level == 1
371
645
 
@@ -381,6 +655,9 @@ module Herringbone
381
655
  private
382
656
 
383
657
  # Fast path for a top-level list of primitives (the most common nested shape)
658
+ #
659
+ # @param field [Schema::Field] list field whose element is a leaf with repetition level 1
660
+ # @return [Array<Array, nil>] one list (or nil) per row
384
661
  def read_simple_list(field)
385
662
  col = field.element.column
386
663
  defs = @defs[col.index]
@@ -420,10 +697,19 @@ module Herringbone
420
697
  out
421
698
  end
422
699
 
700
+ # Hash key for a struct member, memoized
701
+ #
702
+ # @param field [Schema::Field] struct child
703
+ # @return [String, Symbol] frozen name, or Symbol when symbolizing
423
704
  def key_for(field)
424
705
  @keys[field] ||= @symbolize ? field.name.to_sym : -field.name
425
706
  end
426
707
 
708
+ # Assembles one value of +field+ at the current entry cursors, recursing into children
709
+ #
710
+ # @param field [Schema::Field] field to assemble
711
+ # @return [Object, nil] scalar, Hash (struct, map), Array (list) or nil
712
+ # @raise [FormatError] when the definition levels run out
427
713
  def read(field)
428
714
  c = field.first_leaf.index
429
715
  kind = field.kind
@@ -478,6 +764,10 @@ module Herringbone
478
764
  end
479
765
  end
480
766
 
767
+ # Moves every leaf of +field+ past one entry (a null or empty value takes a single entry)
768
+ #
769
+ # @param field [Schema::Field] field whose leaves to advance
770
+ # @return [void]
481
771
  def skip(field)
482
772
  field.leaves.each { |col| @ei[col.index] += 1 }
483
773
  end
@@ -489,3 +779,4 @@ require_relative "reader/page_stream"
489
779
  require_relative "reader/column_chunk_reader"
490
780
  require_relative "reader/column_cursor"
491
781
  require_relative "reader/scan"
782
+ require_relative "reader/numo"