herringbone 0.2.0 → 0.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -17,8 +17,9 @@ module Herringbone
17
17
  # end
18
18
  #
19
19
  # The output is any IO that responds to #write (a File, StringIO, Tempfile, socket, pipe...); it
20
- # is written sequentially and never seeked, rewound or closed by the writer. Herringbone does
21
- # not open files by path.
20
+ # is written sequentially and never seeked, rewound or closed by the writer; only #write is
21
+ # required, its return value is ignored, and #binmode and #flush are called when available.
22
+ # Herringbone does not open files by path.
22
23
  #
23
24
  # Options:
24
25
  # compression: :snappy (default), :zstd, :gzip, :lz4 (LZ4_RAW), :lz4_hadoop, :brotli, :none
@@ -40,16 +41,21 @@ module Herringbone
40
41
  #
41
42
  # Statistics and page indexes (ColumnIndex/OffsetIndex) are always written.
42
43
  class Writer
43
- MAGIC = "PAR1"
44
+ # Marker at the start and end of every Parquet file
45
+ MAGIC = "PAR1".b.freeze
46
+ # Shorthand for Format::Type (physical types)
44
47
  T = Format::Type
48
+ # Shorthand for Format::Encoding
45
49
  E = Format::Encoding
46
50
 
51
+ # Names accepted in the +encodings:+ option => Parquet encoding id
47
52
  ENCODING_NAMES = {
48
53
  plain: E::PLAIN, rle: E::RLE, delta_binary_packed: E::DELTA_BINARY_PACKED,
49
54
  delta_length_byte_array: E::DELTA_LENGTH_BYTE_ARRAY, delta_byte_array: E::DELTA_BYTE_ARRAY,
50
55
  byte_stream_split: E::BYTE_STREAM_SPLIT
51
56
  }.freeze
52
57
 
58
+ # Encoding id => physical types it may be used for, per the Parquet encodings spec
53
59
  VALID_ENCODINGS = {
54
60
  E::PLAIN => T::NAMES.keys,
55
61
  E::RLE => [T::BOOLEAN],
@@ -59,15 +65,41 @@ module Herringbone
59
65
  E::BYTE_STREAM_SPLIT => [T::INT32, T::INT64, T::FLOAT, T::DOUBLE, T::FIXED_LEN_BYTE_ARRAY]
60
66
  }.freeze
61
67
 
68
+ # Largest dictionary build_dictionary keeps; above it the chunk is written without a dictionary.
69
+ # Byte-array columns are dictionary-encoded by ByteValues, which applies its own
70
+ # ByteValues::MAX_DICTIONARY_BYTES as values arrive.
62
71
  MAX_DICTIONARY_BYTES = 1024 * 1024
63
72
  # The row group byte size is first estimated after this many rows, then after every row group
64
73
  ESTIMATE_AFTER_ROWS = 1000
65
74
 
75
+ # @return [Schema] schema the rows are written with
66
76
  attr_reader :schema
67
77
 
68
78
  # Opens a writer on +io+. With a block, the file is finished (footer written) when the block
69
79
  # returns and the block's value is returned; if the block raises, the writer is aborted and no
70
80
  # footer is written. Without a block, call #close to finish. The IO is never closed.
81
+ #
82
+ # @param io [IO, #write] destination
83
+ # @param schema [Schema] schema of the rows
84
+ # @param options [Hash{Symbol => Object}] see the class description and #initialize
85
+ # @option options [Symbol] :compression (:snappy) codec, see Herringbone.codecs
86
+ # @option options [Integer] :row_group_bytes (16MB) approximate buffered size that triggers a row group
87
+ # @option options [Integer, nil] :row_group_rows (nil) also flush a row group after this many rows
88
+ # @option options [Integer] :page_bytes (1MB) approximate uncompressed data page size
89
+ # @option options [Integer] :page_rows (20_000) maximum rows per data page
90
+ # @option options [Integer] :data_page_version (1) 1 or 2
91
+ # @option options [Boolean, Array<String>] :dictionary (true) dictionary-encode all eligible columns,
92
+ # none, or only the listed dotted column paths
93
+ # @option options [Hash{String => Symbol}] :encodings ({}) dotted column path => value encoding
94
+ # for non-dictionary pages
95
+ # @option options [Hash{String => String}] :metadata ({}) key/value metadata for the footer
96
+ # @option options [Boolean, Array<String>, Hash{String => Boolean, Hash}] :bloom_filters (nil)
97
+ # columns to write split block bloom filters for
98
+ # @yield [writer] the open writer
99
+ # @yieldparam writer [Writer] writer to append rows to
100
+ # @yieldreturn [Object] returned by open
101
+ # @return [Writer, Object] the writer without a block, the block's value with one
102
+ # @raise [ArgumentError] for an invalid IO, schema or option
71
103
  def self.open(io, schema, **options)
72
104
  writer = new(io, schema, **options)
73
105
  return writer unless block_given?
@@ -81,6 +113,27 @@ module Herringbone
81
113
  result
82
114
  end
83
115
 
116
+ # Validates the options and writes the leading magic bytes to +io+.
117
+ #
118
+ # @param io [IO, #write] destination; switched to binmode when it supports that
119
+ # @param schema [Schema] schema of the rows
120
+ # @param compression [Symbol, Integer] codec name (see Herringbone.codecs) or Format::Codec id
121
+ # @param row_group_bytes [Integer] approximate buffered size that triggers a row group
122
+ # @param row_group_rows [Integer, nil] also flush a row group after this many rows
123
+ # @param page_bytes [Integer] approximate uncompressed data page size
124
+ # @param page_rows [Integer] maximum level entries per data page (pages of repeated columns
125
+ # extend to the next row start)
126
+ # @param data_page_version [Integer] 1 or 2
127
+ # @param dictionary [Boolean, Array<String>] true for every column except BOOLEAN, FLOAT, DOUBLE
128
+ # and those listed in +encodings+; false for none; or the dotted paths of the columns to encode
129
+ # @param encodings [Hash{String, Symbol => Symbol, Integer}] dotted column path => value encoding
130
+ # (a name from ENCODING_NAMES or an encoding id) for non-dictionary pages
131
+ # @param metadata [Hash{#to_s => #to_s}] key/value metadata for the footer
132
+ # @param bloom_filters [Boolean, Array<String>, Hash{String => Boolean, Hash}, nil] see the class
133
+ # description
134
+ # @raise [ArgumentError] for an invalid IO, schema or option
135
+ # @raise [MissingCodecError] when the codec's optional gem cannot be loaded
136
+ # @raise [UnsupportedError] when the codec is not supported
84
137
  def initialize(io, schema, compression: :snappy, row_group_bytes: 16 * 1024 * 1024, row_group_rows: nil,
85
138
  page_bytes: 1024 * 1024, page_rows: 20_000, data_page_version: 1, dictionary: true, encodings: {},
86
139
  metadata: {}, bloom_filters: nil)
@@ -100,8 +153,10 @@ module Herringbone
100
153
  raise ArgumentError, "data_page_version must be 1 or 2" unless [1, 2].include?(@data_page_version)
101
154
  @dictionary = dictionary
102
155
  @encodings = encodings.to_h { |path, enc| [path.to_s, encoding_id(path, enc)] }
103
- unknown = @encodings.keys - schema.columns.map(&:dotted_path)
156
+ columns = schema.columns.to_h { |c| [c.dotted_path, c] }
157
+ unknown = @encodings.keys - columns.keys
104
158
  raise ArgumentError, "encodings: no such column #{unknown.join(", ")}" unless unknown.empty?
159
+ @encodings.each { |path, enc| check_encoding!(columns[path], enc) }
105
160
  @metadata = metadata
106
161
  @bloom_filters = bloom_filter_config(bloom_filters)
107
162
  @pending_bloom_filters = [] # [ColumnMetaData, BloomFilter] for the row group being written
@@ -116,7 +171,14 @@ module Herringbone
116
171
  end
117
172
 
118
173
  # Appends a row: a Hash keyed by top-level field names (Strings or Symbols), an Array of values
119
- # in schema order, or an object responding to #attributes (ActiveRecord) or #to_h (Struct, Data)
174
+ # in schema order, or an object responding to #attributes (ActiveRecord) or #to_h (Struct, Data).
175
+ # A row that fails to encode leaves nothing behind in the buffers. Flushes a row group when the
176
+ # buffered rows reach the size limits.
177
+ #
178
+ # @param row [Hash, Array, #attributes, #to_h] row to append
179
+ # @return [self]
180
+ # @raise [Error] when the writer is closed
181
+ # @raise [EncodeError] when a required field is nil or missing, or a value cannot be encoded
120
182
  def <<(row)
121
183
  raise Error, "Writer is closed" if @closed
122
184
  row = row_hash(row)
@@ -141,7 +203,7 @@ module Herringbone
141
203
  rescue EncodeError => e
142
204
  rollback_row(marks)
143
205
  raise EncodeError, "Row #{@total_rows + @buffered_rows}: #{e.message}"
144
- rescue StandardError
206
+ rescue
145
207
  rollback_row(marks)
146
208
  raise
147
209
  end
@@ -151,9 +213,13 @@ module Herringbone
151
213
  end
152
214
 
153
215
  # Number of rows written so far, including buffered ones
216
+ #
217
+ # @return [Integer]
154
218
  def rows_written = @total_rows + @buffered_rows
155
219
 
156
220
  # Writes any buffered rows as a row group
221
+ #
222
+ # @return [void]
157
223
  def flush_row_group
158
224
  return if @buffered_rows.zero?
159
225
  start = @pos
@@ -171,6 +237,10 @@ module Herringbone
171
237
  reset_buffers
172
238
  end
173
239
 
240
+ # Flushes buffered rows, then writes the page indexes and the footer and flushes the IO.
241
+ # Does nothing when already closed or aborted. The IO itself is not closed.
242
+ #
243
+ # @return [void]
174
244
  def close
175
245
  return if @closed
176
246
  flush_row_group
@@ -194,23 +264,47 @@ module Herringbone
194
264
 
195
265
  # Stops writing without finishing the file (no footer is written). Whatever was already written
196
266
  # to the IO stays there; discarding it is up to the caller.
267
+ #
268
+ # @return [void]
197
269
  def abort
198
270
  @closed = true
199
271
  end
200
272
 
201
273
  private
202
274
 
203
- # Pathname responds to #write too (writing a whole file by path), so it is rejected explicitly
275
+ # Pathname responds to #write too (writing a whole file by path), so it is rejected explicitly.
276
+ # A text-mode IO (a pipe, or a File opened with "w") transcodes what is written when
277
+ # Encoding.default_internal is set, as Rails does, and binary pages cannot be transcoded,
278
+ # so the IO is switched to binary mode.
279
+ #
280
+ # @param io [Object] candidate destination
281
+ # @return [IO, #write] +io+, in binary mode when it supports #binmode
282
+ # @raise [ArgumentError] when +io+ does not respond to #write, or is a String or Pathname
204
283
  def check_io!(io)
205
- return io if io.respond_to?(:write) && !io.is_a?(String) && !(defined?(Pathname) && io.is_a?(Pathname))
284
+ if io.respond_to?(:write) && !io.is_a?(String) && !(defined?(Pathname) && io.is_a?(Pathname))
285
+ io.binmode if io.respond_to?(:binmode)
286
+ return io
287
+ end
206
288
  raise ArgumentError, "Herringbone::Writer expects an IO that responds to #write " \
207
289
  "(e.g. File.open(path, \"wb\") or StringIO.new), got #{io.class}"
208
290
  end
209
291
 
210
292
  # Levels are kept as binary Strings (one byte per entry). Values are an Array for numeric and
211
293
  # boolean columns, and a compact ByteValues for BYTE_ARRAY / FIXED_LEN_BYTE_ARRAY columns.
294
+ #
295
+ # @!attribute defs
296
+ # @return [String, Array<Integer>] definition levels (unpacked to an Array while a chunk is written)
297
+ # @!attribute reps
298
+ # @return [String, Array<Integer>, nil] repetition levels, like +defs+; nil for non-repeated columns
299
+ # @!attribute values
300
+ # @return [Array, ByteValues] non-null physical values
212
301
  ColumnBuffer = Struct.new(:defs, :reps, :values)
213
302
 
303
+ # Starts a new row group: fresh buffers per column, and the per-field write plan
304
+ # (field, name, Symbol name, buffer and encoder for flat fields, or nils for shredded ones).
305
+ #
306
+ # @return [void]
307
+ # @raise [UnsupportedError] when a column has more than 255 definition or repetition levels
214
308
  def reset_buffers
215
309
  @buffers = @schema.columns.map do |col|
216
310
  if col.max_definition_level > 255 || col.max_repetition_level > 255
@@ -229,6 +323,8 @@ module Herringbone
229
323
  @buffered_rows = 0
230
324
  end
231
325
 
326
+ # @param col [Schema::Column] column to buffer
327
+ # @return [ByteValues, Array] empty value store for the column
232
328
  def new_values_store(col)
233
329
  case col.type
234
330
  when T::BYTE_ARRAY then ByteValues.new(dictionary: use_dictionary?(col))
@@ -239,6 +335,8 @@ module Herringbone
239
335
 
240
336
  # Flushes when the buffered values reach row_group_bytes (or row_group_rows rows). The bytes per
241
337
  # row are estimated from the buffered values after the first rows, then refreshed per row group.
338
+ #
339
+ # @return [void]
242
340
  def check_row_group_size
243
341
  @bytes_per_row ||= estimate_bytes_per_row
244
342
  limit = row_limit_for(@bytes_per_row)
@@ -251,11 +349,15 @@ module Herringbone
251
349
  end
252
350
  end
253
351
 
352
+ # @param bytes_per_row [Integer] estimated buffered bytes per row
353
+ # @return [Integer] rows to buffer before the next size check: what fits in row_group_bytes,
354
+ # at least ESTIMATE_AFTER_ROWS, at most row_group_rows
254
355
  def row_limit_for(bytes_per_row)
255
356
  by_bytes = [@row_group_bytes / [bytes_per_row, 1].max, ESTIMATE_AFTER_ROWS].max
256
357
  @row_group_rows ? [@row_group_rows, by_bytes].min : by_bytes
257
358
  end
258
359
 
360
+ # @return [Integer] memory held by the buffers divided by the buffered rows
259
361
  def estimate_bytes_per_row
260
362
  bytes = @schema.columns.sum do |col|
261
363
  buf = @buffers[col.index]
@@ -263,29 +365,58 @@ module Herringbone
263
365
  held = if values.is_a?(ByteValues)
264
366
  values.memory_bytes
265
367
  else
266
- values.size * (col.type == T::INT96 ? 48 : 8)
368
+ values.size * ((col.type == T::INT96) ? 48 : 8)
267
369
  end
268
370
  held + buf.defs.bytesize + (buf.reps ? buf.reps.bytesize : 0)
269
371
  end
270
372
  bytes / [@buffered_rows, 1].max
271
373
  end
272
374
 
375
+ # Writes to the IO and tracks the file offset, since the IO is never asked for its position
376
+ #
377
+ # @param bytes [String] binary data
378
+ # @return [void]
273
379
  def write_raw(bytes)
274
380
  @io.write(bytes)
275
381
  @pos += bytes.bytesize
276
382
  end
277
383
 
384
+ # @param path [String, Symbol] column path, for the error message
385
+ # @param enc [Symbol, String, Integer] encoding name from ENCODING_NAMES, or an encoding id
386
+ # @return [Integer] encoding id
387
+ # @raise [ArgumentError] for an unknown encoding name, or an id that is not in VALID_ENCODINGS
278
388
  def encoding_id(path, enc)
279
- return enc if enc.is_a?(Integer)
389
+ if enc.is_a?(Integer)
390
+ return enc if VALID_ENCODINGS.key?(enc)
391
+ raise ArgumentError, "Encoding #{E::NAMES.fetch(enc, enc)} cannot be requested for #{path}"
392
+ end
280
393
  ENCODING_NAMES.fetch(enc.to_s.downcase.to_sym) { raise ArgumentError, "Unknown encoding #{enc.inspect} for #{path}" }
281
394
  end
282
395
 
396
+ # @param col [Schema::Column] column the encoding is requested for
397
+ # @param enc [Integer] encoding id, a key of VALID_ENCODINGS
398
+ # @return [void]
399
+ # @raise [ArgumentError] when the encoding is not valid for the column's physical type
400
+ def check_encoding!(col, enc)
401
+ return if VALID_ENCODINGS.fetch(enc).include?(col.type)
402
+ raise ArgumentError, "Encoding #{E::NAMES[enc]} is not valid for #{T::NAMES[col.type]} column #{col.dotted_path}"
403
+ end
404
+
405
+ # Reads a struct member, by String or Symbol key
406
+ #
407
+ # @param hash [Hash, #to_h, nil] struct value
408
+ # @param name [String] member name
409
+ # @return [Object, nil] member value, nil when +hash+ is nil or has no such key
410
+ # @raise [EncodeError] when +hash+ cannot be converted to a Hash
283
411
  def lookup(hash, name)
284
412
  return nil if hash.nil?
285
413
  hash = as_hash(hash, name) unless hash.is_a?(Hash)
286
414
  hash.fetch(name) { hash[name.to_sym] }
287
415
  end
288
416
 
417
+ # @param row [Hash, Array, #attributes, #to_h] row given to #<<
418
+ # @return [Hash] the row keyed by top-level field names
419
+ # @raise [EncodeError] when an Array has the wrong size or the row cannot be converted to a Hash
289
420
  def row_hash(row)
290
421
  case row
291
422
  when Hash then row
@@ -300,6 +431,10 @@ module Herringbone
300
431
  end
301
432
  end
302
433
 
434
+ # @param value [Hash, #to_h] value to convert
435
+ # @param what [String] description of the value, for the error message
436
+ # @return [Hash]
437
+ # @raise [EncodeError] when +value+ does not respond to #to_h
303
438
  def as_hash(value, what)
304
439
  return value if value.is_a?(Hash)
305
440
  raise EncodeError, "Expected a Hash for #{what}, got #{value.class}" unless value.respond_to?(:to_h)
@@ -307,6 +442,10 @@ module Herringbone
307
442
  end
308
443
 
309
444
  # Removes the entries a failed row left behind, so the buffers stay aligned
445
+ #
446
+ # @param marks [Array<Array(Integer, Integer, Integer)>, nil] sizes of defs, reps and values of
447
+ # each nested buffer before the row, nil when there are none
448
+ # @return [void]
310
449
  def rollback_row(marks)
311
450
  @plan.each do |field, _, _, buf|
312
451
  next unless buf && buf.defs.bytesize > @buffered_rows
@@ -323,6 +462,14 @@ module Herringbone
323
462
  end
324
463
 
325
464
  # Record shredding: turns a nested value into (definition level, repetition level, value) entries
465
+ #
466
+ # @param field [Schema::Field] field the value belongs to
467
+ # @param value [Object, nil] Ruby value of the field
468
+ # @param parent_def [Integer] definition level recorded when +value+ is nil
469
+ # @param rep [Integer] repetition level of the first entry this value produces
470
+ # @return [void]
471
+ # @raise [EncodeError] when a required value is nil, a list or map value has the wrong type, a
472
+ # map key is nil, or a leaf value cannot be encoded
326
473
  def shred(field, value, parent_def, rep)
327
474
  if value.nil?
328
475
  raise EncodeError, "Field #{field.node.path.join(".")} is required but got nil" unless field.optional
@@ -379,6 +526,13 @@ module Herringbone
379
526
  end
380
527
  end
381
528
 
529
+ # Writes one column chunk of the current row group: an optional dictionary page, then data pages.
530
+ # Bloom filters and page indexes are queued, to be written after the row group and before the
531
+ # footer.
532
+ #
533
+ # @param col [Schema::Column] column being written
534
+ # @param buffer [ColumnBuffer] the column's buffered levels and values
535
+ # @return [Format::ColumnChunk] chunk with its ColumnMetaData, for the row group
382
536
  def write_column_chunk(col, buffer)
383
537
  type = col.type
384
538
  path = col.dotted_path
@@ -399,11 +553,7 @@ module Herringbone
399
553
  value_encoding = if dict_values
400
554
  E::RLE_DICTIONARY
401
555
  else
402
- enc = @encodings[path] || E::PLAIN
403
- unless VALID_ENCODINGS.fetch(enc).include?(type)
404
- raise ArgumentError, "Encoding #{E::NAMES[enc]} is not valid for #{T::NAMES[type]} column #{path}"
405
- end
406
- enc
556
+ @encodings[path] || E::PLAIN
407
557
  end
408
558
 
409
559
  chunk_start = @pos
@@ -482,8 +632,24 @@ module Herringbone
482
632
  chunk
483
633
  end
484
634
 
635
+ # What the page indexes need to know about a written data page
636
+ #
637
+ # @!attribute offset
638
+ # @return [Integer] file offset of the page header
639
+ # @!attribute size
640
+ # @return [Integer] page size in the file, header included
641
+ # @!attribute first_row
642
+ # @return [Integer] index of the page's first row within the row group
643
+ # @!attribute nulls
644
+ # @return [Integer] null entries in the page
645
+ # @!attribute non_null
646
+ # @return [Integer] non-null values in the page
647
+ # @!attribute range
648
+ # @return [Array(Object, Object), nil] [min, max] physical values, nil when there are none
485
649
  PageInfo = Struct.new(:offset, :size, :first_row, :nulls, :non_null, :range)
486
650
 
651
+ # @param pages [Array<PageInfo>] pages of a column chunk
652
+ # @return [Format::OffsetIndex]
487
653
  def offset_index_for(pages)
488
654
  Format::OffsetIndex.new(page_locations: pages.map do |p|
489
655
  Format::PageLocation.new(offset: p.offset, compressed_page_size: p.size, first_row_index: p.first_row)
@@ -491,6 +657,11 @@ module Herringbone
491
657
  end
492
658
 
493
659
  # nil when the column has no defined sort order, or a page's values have no min/max (all NaN)
660
+ #
661
+ # @param col [Schema::Column] column the pages belong to
662
+ # @param pages [Array<PageInfo>] pages of the column chunk
663
+ # @param order [Proc, Method, nil] sort key from #sort_key
664
+ # @return [Format::ColumnIndex, nil]
494
665
  def column_index_for(col, pages, order)
495
666
  return nil unless order
496
667
  return nil if pages.any? { |p| p.non_null.positive? && p.range.nil? }
@@ -503,13 +674,17 @@ module Herringbone
503
674
  )
504
675
  end
505
676
 
677
+ # @param ranges [Array<Array(Object, Object)>] [min, max] of each non-null page, in page order
678
+ # @param order [Proc, Method] sort key from #sort_key
679
+ # @return [Integer] Format::BoundaryOrder: ASCENDING when both mins and maxes never decrease
680
+ # (or there are fewer than two pages), DESCENDING when they never increase, else UNORDERED
506
681
  def boundary_order(ranges, order)
507
682
  return Format::BoundaryOrder::ASCENDING if ranges.size < 2
508
683
  cmp = ->(a, b) { order.equal?(IDENTITY) ? a <=> b : order.call(a) <=> order.call(b) }
509
684
  pairs = ranges.each_cons(2)
510
- if pairs.all? { |(a_min, a_max), (b_min, b_max)| cmp.(a_min, b_min) <= 0 && cmp.(a_max, b_max) <= 0 }
685
+ if pairs.all? { |(a_min, a_max), (b_min, b_max)| cmp.call(a_min, b_min) <= 0 && cmp.call(a_max, b_max) <= 0 }
511
686
  Format::BoundaryOrder::ASCENDING
512
- elsif pairs.all? { |(a_min, a_max), (b_min, b_max)| cmp.(a_min, b_min) >= 0 && cmp.(a_max, b_max) >= 0 }
687
+ elsif pairs.all? { |(a_min, a_max), (b_min, b_max)| cmp.call(a_min, b_min) >= 0 && cmp.call(a_max, b_max) >= 0 }
513
688
  Format::BoundaryOrder::DESCENDING
514
689
  else
515
690
  Format::BoundaryOrder::UNORDERED
@@ -517,6 +692,8 @@ module Herringbone
517
692
  end
518
693
 
519
694
  # Page indexes go after the last row group: all column indexes, then all offset indexes
695
+ #
696
+ # @return [void]
520
697
  def write_page_indexes
521
698
  @page_indexes.each do |chunk, column_index, _|
522
699
  next unless column_index
@@ -534,17 +711,25 @@ module Herringbone
534
711
  @page_indexes.clear
535
712
  end
536
713
 
714
+ # Per-column settings accepted in the +bloom_filters:+ option
537
715
  BLOOM_FILTER_OPTIONS = %i[ndv fpp max_bytes].freeze
538
716
 
539
717
  # { dotted_path => { ndv:, fpp:, max_bytes: } } from the bloom_filters: option
540
- def bloom_filter_config(option)
541
- return {} if option.nil? || option == false
718
+ #
719
+ # @param requested [Boolean, Array<String>, Hash{String => Boolean, Hash, nil}, nil] true for every
720
+ # column of a supported type, column paths, or column path => true / false / settings Hash.
721
+ # A path may also be given as an Array of names. Settings are +ndv+, +fpp+ and +max_bytes+.
722
+ # @return [Hash{String => Hash{Symbol => Numeric, nil}}] settings by dotted path; empty when disabled
723
+ # @raise [ArgumentError] for an unknown column, a column type without bloom filter support, or
724
+ # invalid settings
725
+ def bloom_filter_config(requested)
726
+ return {} if requested.nil? || requested == false
542
727
  columns = @schema.columns.to_h { |c| [c.dotted_path, c] }
543
- if option == true
728
+ if requested == true
544
729
  return columns.select { |_, c| BloomFilter::TYPES.include?(c.type) }.transform_values { bloom_filter_settings({}) }
545
730
  end
546
- option = Array(option).to_h { |path| [path, true] } unless option.is_a?(Hash)
547
- option.each_with_object({}) do |(path, settings), config|
731
+ requested = Array(requested).to_h { |path| [path, true] } unless requested.is_a?(Hash)
732
+ requested.each_with_object({}) do |(path, settings), config|
548
733
  path = path.is_a?(Array) ? path.join(".") : path.to_s
549
734
  col = columns[path] or raise ArgumentError, "bloom_filters: no such column #{path}"
550
735
  next if settings.nil? || settings == false
@@ -557,6 +742,11 @@ module Herringbone
557
742
  end
558
743
  end
559
744
 
745
+ # @param settings [Hash{Symbol => Object}] +ndv+, +fpp+ and +max_bytes+, all optional
746
+ # @param path [String, nil] column path, for error messages
747
+ # @return [Hash{Symbol => Numeric, nil}] +ndv+ (nil to count distinct values), +fpp+ and
748
+ # +max_bytes+ with defaults filled in
749
+ # @raise [ArgumentError] for unknown keys, a non-positive ndv, or an fpp outside (0, 1)
560
750
  def bloom_filter_settings(settings, path = nil)
561
751
  unknown = settings.keys - BLOOM_FILTER_OPTIONS
562
752
  raise ArgumentError, "bloom_filters: unknown option #{unknown.join(", ")} for #{path}" unless unknown.empty?
@@ -564,12 +754,18 @@ module Herringbone
564
754
  raise ArgumentError, "bloom_filters: ndv must be positive for #{path}" if ndv && !ndv.positive?
565
755
  fpp = Float(settings.fetch(:fpp, BloomFilter::DEFAULT_FPP))
566
756
  raise ArgumentError, "bloom_filters: fpp must be between 0 and 1 for #{path}" unless fpp > 0 && fpp < 1
567
- { ndv: ndv, fpp: fpp, max_bytes: Integer(settings.fetch(:max_bytes, BloomFilter::DEFAULT_MAX_BYTES)) }
757
+ {ndv: ndv, fpp: fpp, max_bytes: Integer(settings.fetch(:max_bytes, BloomFilter::DEFAULT_MAX_BYTES))}
568
758
  end
569
759
 
570
760
  # A filter holding the chunk's values: +dict_values+ (already distinct) for dictionary-encoded
571
761
  # chunks, +values+ otherwise. Each distinct value is hashed once, and the filter is sized from
572
762
  # the configured ndv or from the number of distinct values.
763
+ #
764
+ # @param col [Schema::Column] column the filter is for
765
+ # @param settings [Hash{Symbol => Numeric, nil}] from #bloom_filter_settings
766
+ # @param dict_values [Array, nil] dictionary of the chunk, when dictionary-encoded
767
+ # @param values [Array, nil] the chunk's non-null physical values, used without a dictionary
768
+ # @return [BloomFilter]
573
769
  def build_bloom_filter(col, settings, dict_values, values)
574
770
  hashes = if dict_values
575
771
  BloomFilter.hash_physical_all(dict_values, col.type)
@@ -583,6 +779,8 @@ module Herringbone
583
779
  end
584
780
 
585
781
  # Bloom filters go right after the row group's column chunks, in column order
782
+ #
783
+ # @return [void]
586
784
  def write_bloom_filters
587
785
  @pending_bloom_filters.each do |meta, filter|
588
786
  bytes = filter.encode
@@ -593,6 +791,9 @@ module Herringbone
593
791
  @pending_bloom_filters.clear
594
792
  end
595
793
 
794
+ # @param col [Schema::Column] column to check
795
+ # @return [Boolean] whether the +dictionary:+ option asks for this column to be dictionary-encoded
796
+ # (never for BOOLEAN)
596
797
  def use_dictionary?(col)
597
798
  return false if col.type == T::BOOLEAN
598
799
  case @dictionary
@@ -602,13 +803,20 @@ module Herringbone
602
803
  end
603
804
  end
604
805
 
605
- # Returns [dictionary_values, indices] or nil when a dictionary is not worthwhile
806
+ # Returns [dictionary_values, indices] or nil when a dictionary is not worthwhile: more than about
807
+ # half the values are distinct, or the dictionary exceeds MAX_DICTIONARY_BYTES
808
+ #
809
+ # @param values [Array] non-null physical values of the chunk
810
+ # @param type [Integer] physical type
811
+ # @param type_length [Integer, nil] FIXED_LEN_BYTE_ARRAY width
812
+ # @return [Array(Array, Array<Integer>), nil]
606
813
  def build_dictionary(values, type, type_length)
607
814
  # Floats are keyed by bit pattern so that -0.0 and 0.0 (and NaNs) stay distinct
608
815
  if type == T::FLOAT || type == T::DOUBLE
609
816
  keys = values.pack("G*").unpack("Q>*")
610
817
  uniq = keys.uniq
611
818
  return nil if uniq.size > values.size / 2 + 1 && values.size > 16
819
+ return nil if uniq.size * 8 > MAX_DICTIONARY_BYTES
612
820
  index = uniq.each_with_index.to_h
613
821
  return [uniq.pack("Q>*").unpack("G*"), keys.map(&index)]
614
822
  end
@@ -626,11 +834,15 @@ module Herringbone
626
834
 
627
835
  # Splits a column buffer into pages of roughly @page_bytes bytes. Repeated columns
628
836
  # are only cut where a new row starts.
837
+ #
838
+ # @param buf [ColumnBuffer] column buffer with levels as Arrays
839
+ # @param value_bytes [Integer] estimated encoded size of all the chunk's values
840
+ # @return [Array<Array(Integer, Integer)>] [from, to) ranges of level entries, one per page
629
841
  def page_ranges(buf, value_bytes)
630
842
  n = buf.defs.size
631
843
  bytes = n + value_bytes
632
844
  pages = (bytes + @page_bytes - 1) / @page_bytes
633
- per = pages <= 1 ? n : (n + pages - 1) / pages
845
+ per = (pages <= 1) ? n : (n + pages - 1) / pages
634
846
  per = @page_rows if per > @page_rows
635
847
  return [[0, n]] if per >= n
636
848
  reps = buf.reps
@@ -646,6 +858,8 @@ module Herringbone
646
858
  ranges
647
859
  end
648
860
 
861
+ # @param col [Schema::Column] column to check
862
+ # @return [Integer, nil] PLAIN-encoded bytes per value, nil for BYTE_ARRAY (variable width)
649
863
  def value_width(col)
650
864
  case col.type
651
865
  when T::BOOLEAN then 1
@@ -656,6 +870,15 @@ module Herringbone
656
870
  end
657
871
  end
658
872
 
873
+ # Encodes a page's values. RLE_DICTIONARY values are the dictionary indices, prefixed with the
874
+ # bit width byte; RLE is only used for BOOLEAN and carries the 4-byte length prefix.
875
+ #
876
+ # @param values [Array] physical values, or dictionary indices
877
+ # @param encoding [Integer] encoding id
878
+ # @param type [Integer] physical type
879
+ # @param type_length [Integer, nil] FIXED_LEN_BYTE_ARRAY width
880
+ # @param dict_size [Integer, nil] number of dictionary entries, for RLE_DICTIONARY
881
+ # @return [String] encoded binary page values
659
882
  def encode_values(values, encoding, type, type_length, dict_size)
660
883
  case encoding
661
884
  when E::PLAIN then Encodings::Plain.encode(values, type, type_length)
@@ -666,27 +889,41 @@ module Herringbone
666
889
  body = Encodings::RLE.encode_hybrid(values.map { |v| v ? 1 : 0 }, 1)
667
890
  [body.bytesize].pack("V") << body
668
891
  when E::DELTA_BINARY_PACKED
669
- Encodings::Delta.encode_binary_packed(values, type == T::INT32 ? 32 : 64)
892
+ Encodings::Delta.encode_binary_packed(values, (type == T::INT32) ? 32 : 64)
670
893
  when E::DELTA_LENGTH_BYTE_ARRAY then Encodings::Delta.encode_length_byte_array(values)
671
894
  when E::DELTA_BYTE_ARRAY then Encodings::Delta.encode_byte_array(values)
672
895
  when E::BYTE_STREAM_SPLIT
673
- width = { T::INT32 => 4, T::FLOAT => 4, T::INT64 => 8, T::DOUBLE => 8 }[type] || type_length
896
+ width = {T::INT32 => 4, T::FLOAT => 4, T::INT64 => 8, T::DOUBLE => 8}[type] || type_length
674
897
  Encodings::ByteStreamSplit.encode(Encodings::Plain.encode(values, type, type_length), width)
675
898
  end
676
899
  end
677
900
 
678
901
  # Writes a page, returning the uncompressed size including the header
902
+ #
903
+ # @param header [Format::PageHeader] header; sizes and CRC32 are filled in here
904
+ # @param body [String] uncompressed page body
905
+ # @param compressed [String, nil] bytes to write as the page body, when already prepared (v2 data
906
+ # pages, whose levels stay uncompressed); +body+ is compressed otherwise
907
+ # @return [Integer]
679
908
  def write_page(header, body, compressed = nil)
680
909
  compressed ||= Compression.compress(@codec, body)
681
910
  header.uncompressed_page_size ||= body.bytesize
682
911
  header.compressed_page_size = compressed.bytesize
683
- header.crc = Zlib.crc32(compressed).then { |c| c >= 0x8000_0000 ? c - 0x1_0000_0000 : c }
912
+ header.crc = Zlib.crc32(compressed).then { |c| (c >= 0x8000_0000) ? c - 0x1_0000_0000 : c }
684
913
  encoded = header.encode
685
914
  write_raw(encoded)
686
915
  write_raw(compressed)
687
916
  encoded.bytesize + header.uncompressed_page_size
688
917
  end
689
918
 
919
+ # A DATA_PAGE: length-prefixed repetition and definition levels, then the values, all compressed
920
+ #
921
+ # @param n [Integer] number of level entries (values including nulls)
922
+ # @param rep_bytes [String] RLE-encoded repetition levels, empty when the column has none
923
+ # @param def_bytes [String] RLE-encoded definition levels, empty when the column has none
924
+ # @param encoded [String] encoded values
925
+ # @param encoding [Integer] encoding id of the values
926
+ # @return [Integer] uncompressed size including the header
690
927
  def write_data_page_v1(n, rep_bytes, def_bytes, encoded, encoding)
691
928
  body = String.new(encoding: Encoding::BINARY)
692
929
  body << [rep_bytes.bytesize].pack("V") << rep_bytes unless rep_bytes.empty?
@@ -702,6 +939,16 @@ module Herringbone
702
939
  write_page(header, body)
703
940
  end
704
941
 
942
+ # A DATA_PAGE_V2: levels without length prefixes and uncompressed, then the compressed values
943
+ #
944
+ # @param n [Integer] number of level entries (values including nulls)
945
+ # @param nulls [Integer] null entries
946
+ # @param rows [Integer] rows in the page
947
+ # @param rep_bytes [String] RLE-encoded repetition levels, empty when the column has none
948
+ # @param def_bytes [String] RLE-encoded definition levels, empty when the column has none
949
+ # @param encoded [String] encoded values
950
+ # @param encoding [Integer] encoding id of the values
951
+ # @return [Integer] uncompressed size including the header
705
952
  def write_data_page_v2(n, nulls, rows, rep_bytes, def_bytes, encoded, encoding)
706
953
  compressed = Compression.compress(@codec, encoded)
707
954
  header = Format::PageHeader.new(
@@ -717,9 +964,20 @@ module Herringbone
717
964
  write_page(header, "".b, rep_bytes + def_bytes + compressed)
718
965
  end
719
966
 
967
+ # Statistics and column index bounds longer than this are truncated, see #truncate_min
720
968
  STAT_TRUNCATE_BYTES = 64
969
+ # Sort key for types whose Ruby ordering already matches Parquet's; compared by identity to
970
+ # skip calling it
721
971
  IDENTITY = ->(v) { v }
722
972
 
973
+ # Chunk statistics: null count, plus min/max (flagged exact unless truncated) when the column
974
+ # has a sort order and non-NaN values
975
+ #
976
+ # @param col [Schema::Column] column the chunk belongs to
977
+ # @param defs [Array<Integer>] definition levels of the chunk
978
+ # @param values [Array] distinct or all non-null physical values of the chunk
979
+ # @param order [Proc, Method, nil] sort key from #sort_key
980
+ # @return [Format::Statistics]
723
981
  def statistics_for(col, defs, values, order)
724
982
  max_def = col.max_definition_level
725
983
  nulls = max_def.zero? ? 0 : defs.size - defs.count(max_def)
@@ -738,13 +996,16 @@ module Herringbone
738
996
 
739
997
  # A key giving the Parquet sort order of the column's physical values (IDENTITY when Ruby's own
740
998
  # comparison already matches), or nil when the order is undefined (INT96)
999
+ #
1000
+ # @param col [Schema::Column] column to order
1001
+ # @return [Proc, Method, nil]
741
1002
  def sort_key(col)
742
1003
  kind, _, signed = Types.logical_of(col.node)
743
1004
  case col.type
744
1005
  when T::BOOLEAN then ->(v) { v ? 1 : 0 }
745
1006
  when T::INT32, T::INT64
746
1007
  return IDENTITY unless kind == :integer && !signed
747
- mask = col.type == T::INT32 ? 0xFFFF_FFFF : 0xFFFF_FFFF_FFFF_FFFF
1008
+ mask = (col.type == T::INT32) ? 0xFFFF_FFFF : 0xFFFF_FFFF_FFFF_FFFF
748
1009
  ->(v) { v & mask }
749
1010
  when T::FLOAT, T::DOUBLE then IDENTITY
750
1011
  when T::BYTE_ARRAY, T::FIXED_LEN_BYTE_ARRAY
@@ -756,7 +1017,13 @@ module Herringbone
756
1017
  end
757
1018
  end
758
1019
 
759
- # [min, max] of +values+ in column order, ignoring NaNs; nil if there is nothing to compare
1020
+ # [min, max] of +values+ in column order, ignoring NaNs; nil if there is nothing to compare.
1021
+ # A zero float bound is normalized to -0.0 (min) / 0.0 (max), as the spec asks.
1022
+ #
1023
+ # @param col [Schema::Column] column the values belong to
1024
+ # @param values [Array] physical values
1025
+ # @param order [Proc, Method] sort key from #sort_key
1026
+ # @return [Array(Object, Object), nil]
760
1027
  def value_range(col, values, order)
761
1028
  floats = col.type == T::FLOAT || col.type == T::DOUBLE
762
1029
  if floats
@@ -773,6 +1040,9 @@ module Herringbone
773
1040
  [min, max]
774
1041
  end
775
1042
 
1043
+ # @param col [Schema::Column] column the value belongs to
1044
+ # @param value [Object] physical value
1045
+ # @return [String] the value PLAIN-encoded, as statistics store it (byte arrays without length)
776
1046
  def stat_bytes(col, value)
777
1047
  case col.type
778
1048
  when T::BOOLEAN then value ? "\x01".b : "\x00".b
@@ -786,10 +1056,17 @@ module Herringbone
786
1056
 
787
1057
  # Long byte-array bounds are truncated: a prefix is still a lower bound for the minimum, and
788
1058
  # a prefix with its last byte incremented is an upper bound for the maximum
1059
+ #
1060
+ # @param bytes [String] encoded minimum
1061
+ # @return [String] at most STAT_TRUNCATE_BYTES bytes
789
1062
  def truncate_min(bytes)
790
- bytes.bytesize > STAT_TRUNCATE_BYTES ? bytes.byteslice(0, STAT_TRUNCATE_BYTES) : bytes
1063
+ (bytes.bytesize > STAT_TRUNCATE_BYTES) ? bytes.byteslice(0, STAT_TRUNCATE_BYTES) : bytes
791
1064
  end
792
1065
 
1066
+ # @param bytes [String] encoded maximum
1067
+ # @return [String] at most STAT_TRUNCATE_BYTES bytes, with the last byte that is not 0xFF
1068
+ # incremented (trailing 0xFF bytes dropped); +bytes+ unchanged when it fits, or when the
1069
+ # whole prefix is 0xFF
793
1070
  def truncate_max(bytes)
794
1071
  return bytes if bytes.bytesize <= STAT_TRUNCATE_BYTES
795
1072
  prefix = bytes.byteslice(0, STAT_TRUNCATE_BYTES).bytes