herringbone 0.1.0 → 0.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -8,27 +8,37 @@ module Herringbone
8
8
  # string :name
9
9
  # list :tags, :string
10
10
  # end
11
- # Herringbone::Writer.open("out.parquet", schema) do |w|
12
- # w << { "id" => 1, "name" => "one", "tags" => ["a", "b"] }
13
- # w << [2, "two", []] # Arrays are taken in schema order
14
- # w << order # objects responding to #attributes (ActiveRecord) or #to_h
11
+ # File.open("out.parquet", "wb") do |file|
12
+ # Herringbone::Writer.open(file, schema) do |w|
13
+ # w << { "id" => 1, "name" => "one", "tags" => ["a", "b"] }
14
+ # w << [2, "two", []] # Arrays are taken in schema order
15
+ # w << order # objects responding to #attributes (ActiveRecord) or #to_h
16
+ # end
15
17
  # end
16
18
  #
17
- # +schema+ is a Herringbone::Schema or a Hash spec (see Schema.define). The target is a path or an IO;
18
- # paths are written to a temporary file next to the target and renamed into place on close, so a
19
- # failed write never leaves a truncated file behind.
19
+ # The output is any IO that responds to #write (a File, StringIO, Tempfile, socket, pipe...); it
20
+ # is written sequentially and never seeked, rewound or closed by the writer. Herringbone does
21
+ # not open files by path.
20
22
  #
21
23
  # Options:
22
- # compression: :zstd (default), :snappy, :gzip, :lz4 (LZ4_RAW), :brotli, :none
24
+ # compression: :snappy (default), :zstd, :gzip, :lz4 (LZ4_RAW), :lz4_hadoop, :brotli, :none
25
+ # (:zstd and :brotli need the zstd-ruby / brotli gems)
23
26
  # row_group_bytes: flush a row group once the buffered values take roughly this much memory
24
27
  # (default 16MB). This bounds memory use while writing.
25
- # row_group_size: also flush after this many rows (default: no row limit)
26
- # page_size: approximate uncompressed data page size in bytes (default 1MB)
28
+ # row_group_rows: also flush after this many rows (default: no row limit)
29
+ # page_bytes: approximate uncompressed data page size (default 1MB)
30
+ # page_rows: at most this many rows per data page (default 20_000), which keeps the page
31
+ # index selective
27
32
  # data_page_version: 1 (default) or 2
28
33
  # dictionary: true/false, or an Array of column paths to dictionary-encode
29
34
  # encodings: { "path.to.column" => :delta_binary_packed, ... } for non-dictionary pages
30
- # statistics: write min/max/null_count statistics (default true)
31
35
  # metadata: Hash of String => String key/value metadata for the footer
36
+ # bloom_filters: write split block bloom filters: true (every column that supports them),
37
+ # an Array of column paths, or { "path" => true | { ndv:, fpp:, max_bytes: } }.
38
+ # Without ndv: the distinct values of each row group are counted. fpp defaults
39
+ # to 0.01 and max_bytes to 1MB. Filters are written after each row group.
40
+ #
41
+ # Statistics and page indexes (ColumnIndex/OffsetIndex) are always written.
32
42
  class Writer
33
43
  MAGIC = "PAR1"
34
44
  T = Format::Type
@@ -55,14 +65,15 @@ module Herringbone
55
65
 
56
66
  attr_reader :schema
57
67
 
58
- # Opens a writer on a path or an IO. With a block, the file is finished when the block returns
59
- # (or discarded if it raises) and the block's value is returned; without one, call #close.
60
- def self.open(target, schema, **options)
61
- writer = new(target, schema, **options)
68
+ # Opens a writer on +io+. With a block, the file is finished (footer written) when the block
69
+ # returns and the block's value is returned; if the block raises, the writer is aborted and no
70
+ # footer is written. Without a block, call #close to finish. The IO is never closed.
71
+ def self.open(io, schema, **options)
72
+ writer = new(io, schema, **options)
62
73
  return writer unless block_given?
63
74
  begin
64
75
  result = yield writer
65
- rescue Exception # rubocop:disable Lint/RescueException -- also discard on Interrupt
76
+ rescue Exception # rubocop:disable Lint/RescueException -- also abort on Interrupt
66
77
  writer.abort
67
78
  raise
68
79
  end
@@ -70,29 +81,36 @@ module Herringbone
70
81
  result
71
82
  end
72
83
 
73
- def initialize(target, schema, compression: :zstd, row_group_bytes: 16 * 1024 * 1024, row_group_size: nil,
74
- page_size: 1024 * 1024, data_page_version: 1, dictionary: true, encodings: {}, statistics: true, metadata: {})
75
- @schema = Schema.coerce(schema)
76
- schema = @schema
84
+ def initialize(io, schema, compression: :snappy, row_group_bytes: 16 * 1024 * 1024, row_group_rows: nil,
85
+ page_bytes: 1024 * 1024, page_rows: 20_000, data_page_version: 1, dictionary: true, encodings: {},
86
+ metadata: {}, bloom_filters: nil)
87
+ raise ArgumentError, "Expected a Herringbone::Schema, got #{schema.class}" unless schema.is_a?(Schema)
88
+ @schema = schema
77
89
  @codec = Compression.codec_id(compression)
90
+ # Fail before creating any file if the codec's library is missing
91
+ Compression.ensure_available!(@codec)
78
92
  @row_group_bytes = Integer(row_group_bytes)
79
- @row_group_size = row_group_size && Integer(row_group_size)
80
- @row_limit = @row_group_size || ESTIMATE_AFTER_ROWS
81
- @page_size = Integer(page_size)
93
+ @row_group_rows = row_group_rows && Integer(row_group_rows)
94
+ @row_limit = @row_group_rows || ESTIMATE_AFTER_ROWS
95
+ @page_bytes = Integer(page_bytes)
96
+ @page_rows = Integer(page_rows)
97
+ raise ArgumentError, "page_rows must be positive" unless @page_rows.positive?
98
+ @page_indexes = [] # [ColumnChunk, ColumnIndex or nil, OffsetIndex] per written column chunk
82
99
  @data_page_version = Integer(data_page_version)
83
100
  raise ArgumentError, "data_page_version must be 1 or 2" unless [1, 2].include?(@data_page_version)
84
101
  @dictionary = dictionary
85
102
  @encodings = encodings.to_h { |path, enc| [path.to_s, encoding_id(path, enc)] }
86
103
  unknown = @encodings.keys - schema.columns.map(&:dotted_path)
87
104
  raise ArgumentError, "encodings: no such column #{unknown.join(", ")}" unless unknown.empty?
88
- @statistics = statistics
89
105
  @metadata = metadata
106
+ @bloom_filters = bloom_filter_config(bloom_filters)
107
+ @pending_bloom_filters = [] # [ColumnMetaData, BloomFilter] for the row group being written
90
108
  @row_groups = []
91
109
  @total_rows = 0
92
110
  @pos = 0
93
111
  @closed = false
94
112
  @bytes_per_row = nil
95
- open_target(target)
113
+ @io = check_io!(io)
96
114
  write_raw(MAGIC)
97
115
  reset_buffers
98
116
  end
@@ -131,12 +149,6 @@ module Herringbone
131
149
  check_row_group_size if @buffered_rows >= @row_limit
132
150
  self
133
151
  end
134
- alias_method :write, :<<
135
-
136
- def write_rows(rows)
137
- rows.each { |row| self << row }
138
- self
139
- end
140
152
 
141
153
  # Number of rows written so far, including buffered ones
142
154
  def rows_written = @total_rows + @buffered_rows
@@ -154,6 +166,7 @@ module Herringbone
154
166
  total_compressed_size: @pos - start,
155
167
  ordinal: @row_groups.size
156
168
  )
169
+ write_bloom_filters
157
170
  @total_rows += @buffered_rows
158
171
  reset_buffers
159
172
  end
@@ -161,13 +174,14 @@ module Herringbone
161
174
  def close
162
175
  return if @closed
163
176
  flush_row_group
177
+ write_page_indexes
164
178
  meta = Format::FileMetaData.new(
165
179
  version: 2,
166
180
  schema: @schema.to_elements,
167
181
  num_rows: @total_rows,
168
182
  row_groups: @row_groups,
169
183
  key_value_metadata: @metadata.empty? ? nil : @metadata.map { |k, v| Format::KeyValue.new(key: k.to_s, value: v&.to_s) },
170
- created_by: "herringbone version #{VERSION}",
184
+ created_by: "herringbone-ruby #{VERSION}",
171
185
  column_orders: @schema.columns.map { Format::ColumnOrder.new(type_order: Format::TypeDefinedOrder.new) }
172
186
  )
173
187
  footer = meta.encode
@@ -176,20 +190,23 @@ module Herringbone
176
190
  write_raw(MAGIC)
177
191
  @io.flush if @io.respond_to?(:flush)
178
192
  @closed = true
179
- finish_target
180
193
  end
181
194
 
182
- # Stops writing without producing a file: a temporary file is deleted, an IO is left as is
195
+ # Stops writing without finishing the file (no footer is written). Whatever was already written
196
+ # to the IO stays there; discarding it is up to the caller.
183
197
  def abort
184
- return if @closed
185
198
  @closed = true
186
- return unless @temp_path
187
- @io.close unless @io.closed?
188
- File.unlink(@temp_path) if File.exist?(@temp_path)
189
199
  end
190
200
 
191
201
  private
192
202
 
203
+ # Pathname responds to #write too (writing a whole file by path), so it is rejected explicitly
204
+ def check_io!(io)
205
+ return io if io.respond_to?(:write) && !io.is_a?(String) && !(defined?(Pathname) && io.is_a?(Pathname))
206
+ raise ArgumentError, "Herringbone::Writer expects an IO that responds to #write " \
207
+ "(e.g. File.open(path, \"wb\") or StringIO.new), got #{io.class}"
208
+ end
209
+
193
210
  # Levels are kept as binary Strings (one byte per entry). Values are an Array for numeric and
194
211
  # boolean columns, and a compact ByteValues for BYTE_ARRAY / FIXED_LEN_BYTE_ARRAY columns.
195
212
  ColumnBuffer = Struct.new(:defs, :reps, :values)
@@ -220,24 +237,7 @@ module Herringbone
220
237
  end
221
238
  end
222
239
 
223
- def open_target(target)
224
- if target.respond_to?(:write)
225
- @io = target
226
- else
227
- @path = target.to_s
228
- dir = File.dirname(@path)
229
- @temp_path = File.join(dir, ".#{File.basename(@path)}.#{Process.pid}.#{rand(1 << 32).to_s(36)}.tmp")
230
- @io = File.open(@temp_path, "wb")
231
- end
232
- end
233
-
234
- def finish_target
235
- return unless @temp_path
236
- @io.close
237
- File.rename(@temp_path, @path)
238
- end
239
-
240
- # Flushes when the buffered values reach row_group_bytes (or row_group_size rows). The bytes per
240
+ # Flushes when the buffered values reach row_group_bytes (or row_group_rows rows). The bytes per
241
241
  # row are estimated from the buffered values after the first rows, then refreshed per row group.
242
242
  def check_row_group_size
243
243
  @bytes_per_row ||= estimate_bytes_per_row
@@ -253,7 +253,7 @@ module Herringbone
253
253
 
254
254
  def row_limit_for(bytes_per_row)
255
255
  by_bytes = [@row_group_bytes / [bytes_per_row, 1].max, ESTIMATE_AFTER_ROWS].max
256
- @row_group_size ? [@row_group_size, by_bytes].min : by_bytes
256
+ @row_group_rows ? [@row_group_rows, by_bytes].min : by_bytes
257
257
  end
258
258
 
259
259
  def estimate_bytes_per_row
@@ -430,6 +430,9 @@ module Herringbone
430
430
  else
431
431
  values.sum(&:bytesize) + 4 * values.size
432
432
  end
433
+ order = sort_key(col)
434
+ pages = []
435
+ first_row = 0
433
436
  page_ranges(buf, value_bytes).each do |from, to|
434
437
  n = to - from
435
438
  defs = buf.defs[from, n]
@@ -437,6 +440,12 @@ module Herringbone
437
440
  non_null = max_def.zero? ? n : defs.count(max_def)
438
441
  page_values = (indices || values)[value_index, non_null]
439
442
  value_index += non_null
443
+ page_offset = @pos
444
+ range = nil
445
+ if order && non_null.positive?
446
+ in_page = dict_values ? page_values.uniq.map { |i| dict_values[i] } : page_values
447
+ range = value_range(col, in_page, order)
448
+ end
440
449
  encoded = encode_values(page_values, value_encoding, type, col.type_length, dict_values&.size)
441
450
  rep_bytes = max_rep.positive? ? Encodings::RLE.encode_hybrid(reps, max_rep.bit_length) : "".b
442
451
  def_bytes = max_def.positive? ? Encodings::RLE.encode_hybrid(defs, max_def.bit_length) : "".b
@@ -446,6 +455,8 @@ module Herringbone
446
455
  num_rows = max_rep.zero? ? n : reps.count(0)
447
456
  write_data_page_v2(n, n - non_null, num_rows, rep_bytes, def_bytes, encoded, value_encoding)
448
457
  end
458
+ pages << PageInfo.new(page_offset, @pos - page_offset, first_row, n - non_null, non_null, range)
459
+ first_row += max_rep.zero? ? n : reps.count(0)
449
460
  end
450
461
 
451
462
  encodings = [E::RLE]
@@ -461,9 +472,125 @@ module Herringbone
461
472
  total_compressed_size: @pos - chunk_start,
462
473
  data_page_offset: data_offset,
463
474
  dictionary_page_offset: dictionary_offset,
464
- statistics: @statistics ? statistics_for(col, buf.defs, values || indices.uniq.map { |i| dict_values[i] }) : nil
475
+ statistics: statistics_for(col, buf.defs, values || indices.uniq.map { |i| dict_values[i] }, order)
465
476
  )
466
- Format::ColumnChunk.new(file_offset: chunk_start, meta_data: meta)
477
+ chunk = Format::ColumnChunk.new(file_offset: chunk_start, meta_data: meta)
478
+ if (bloom = @bloom_filters[path])
479
+ @pending_bloom_filters << [meta, build_bloom_filter(col, bloom, dict_values, values)]
480
+ end
481
+ @page_indexes << [chunk, column_index_for(col, pages, order), offset_index_for(pages)]
482
+ chunk
483
+ end
484
+
485
+ PageInfo = Struct.new(:offset, :size, :first_row, :nulls, :non_null, :range)
486
+
487
+ def offset_index_for(pages)
488
+ Format::OffsetIndex.new(page_locations: pages.map do |p|
489
+ Format::PageLocation.new(offset: p.offset, compressed_page_size: p.size, first_row_index: p.first_row)
490
+ end)
491
+ end
492
+
493
+ # nil when the column has no defined sort order, or a page's values have no min/max (all NaN)
494
+ def column_index_for(col, pages, order)
495
+ return nil unless order
496
+ return nil if pages.any? { |p| p.non_null.positive? && p.range.nil? }
497
+ Format::ColumnIndex.new(
498
+ null_pages: pages.map { |p| p.non_null.zero? },
499
+ min_values: pages.map { |p| p.range ? truncate_min(stat_bytes(col, p.range[0])) : "".b },
500
+ max_values: pages.map { |p| p.range ? truncate_max(stat_bytes(col, p.range[1])) : "".b },
501
+ boundary_order: boundary_order(pages.filter_map(&:range), order),
502
+ null_counts: pages.map(&:nulls)
503
+ )
504
+ end
505
+
506
+ def boundary_order(ranges, order)
507
+ return Format::BoundaryOrder::ASCENDING if ranges.size < 2
508
+ cmp = ->(a, b) { order.equal?(IDENTITY) ? a <=> b : order.call(a) <=> order.call(b) }
509
+ pairs = ranges.each_cons(2)
510
+ if pairs.all? { |(a_min, a_max), (b_min, b_max)| cmp.(a_min, b_min) <= 0 && cmp.(a_max, b_max) <= 0 }
511
+ Format::BoundaryOrder::ASCENDING
512
+ elsif pairs.all? { |(a_min, a_max), (b_min, b_max)| cmp.(a_min, b_min) >= 0 && cmp.(a_max, b_max) >= 0 }
513
+ Format::BoundaryOrder::DESCENDING
514
+ else
515
+ Format::BoundaryOrder::UNORDERED
516
+ end
517
+ end
518
+
519
+ # Page indexes go after the last row group: all column indexes, then all offset indexes
520
+ def write_page_indexes
521
+ @page_indexes.each do |chunk, column_index, _|
522
+ next unless column_index
523
+ bytes = column_index.encode
524
+ chunk.column_index_offset = @pos
525
+ chunk.column_index_length = bytes.bytesize
526
+ write_raw(bytes)
527
+ end
528
+ @page_indexes.each do |chunk, _, offset_index|
529
+ bytes = offset_index.encode
530
+ chunk.offset_index_offset = @pos
531
+ chunk.offset_index_length = bytes.bytesize
532
+ write_raw(bytes)
533
+ end
534
+ @page_indexes.clear
535
+ end
536
+
537
+ BLOOM_FILTER_OPTIONS = %i[ndv fpp max_bytes].freeze
538
+
539
+ # { dotted_path => { ndv:, fpp:, max_bytes: } } from the bloom_filters: option
540
+ def bloom_filter_config(option)
541
+ return {} if option.nil? || option == false
542
+ columns = @schema.columns.to_h { |c| [c.dotted_path, c] }
543
+ if option == true
544
+ return columns.select { |_, c| BloomFilter::TYPES.include?(c.type) }.transform_values { bloom_filter_settings({}) }
545
+ end
546
+ option = Array(option).to_h { |path| [path, true] } unless option.is_a?(Hash)
547
+ option.each_with_object({}) do |(path, settings), config|
548
+ path = path.is_a?(Array) ? path.join(".") : path.to_s
549
+ col = columns[path] or raise ArgumentError, "bloom_filters: no such column #{path}"
550
+ next if settings.nil? || settings == false
551
+ unless BloomFilter::TYPES.include?(col.type)
552
+ raise ArgumentError, "bloom_filters: not supported for #{T::NAMES[col.type]} column #{path}"
553
+ end
554
+ settings = {} if settings == true
555
+ raise ArgumentError, "bloom_filters: expected true or a Hash for #{path}, got #{settings.inspect}" unless settings.is_a?(Hash)
556
+ config[path] = bloom_filter_settings(settings.transform_keys(&:to_sym), path)
557
+ end
558
+ end
559
+
560
+ def bloom_filter_settings(settings, path = nil)
561
+ unknown = settings.keys - BLOOM_FILTER_OPTIONS
562
+ raise ArgumentError, "bloom_filters: unknown option #{unknown.join(", ")} for #{path}" unless unknown.empty?
563
+ ndv = settings[:ndv] && Integer(settings[:ndv])
564
+ raise ArgumentError, "bloom_filters: ndv must be positive for #{path}" if ndv && !ndv.positive?
565
+ fpp = Float(settings.fetch(:fpp, BloomFilter::DEFAULT_FPP))
566
+ raise ArgumentError, "bloom_filters: fpp must be between 0 and 1 for #{path}" unless fpp > 0 && fpp < 1
567
+ { ndv: ndv, fpp: fpp, max_bytes: Integer(settings.fetch(:max_bytes, BloomFilter::DEFAULT_MAX_BYTES)) }
568
+ end
569
+
570
+ # A filter holding the chunk's values: +dict_values+ (already distinct) for dictionary-encoded
571
+ # chunks, +values+ otherwise. Each distinct value is hashed once, and the filter is sized from
572
+ # the configured ndv or from the number of distinct values.
573
+ def build_bloom_filter(col, settings, dict_values, values)
574
+ hashes = if dict_values
575
+ BloomFilter.hash_physical_all(dict_values, col.type)
576
+ else
577
+ BloomFilter.hash_physical_all(values, col.type, distinct: true)
578
+ end
579
+ size = BloomFilter.optimal_num_bytes(settings[:ndv] || hashes.size, settings[:fpp], max_bytes: settings[:max_bytes])
580
+ filter = BloomFilter.new(size, column: col)
581
+ filter.insert_hashes(hashes)
582
+ filter
583
+ end
584
+
585
+ # Bloom filters go right after the row group's column chunks, in column order
586
+ def write_bloom_filters
587
+ @pending_bloom_filters.each do |meta, filter|
588
+ bytes = filter.encode
589
+ meta.bloom_filter_offset = @pos
590
+ meta.bloom_filter_length = bytes.bytesize
591
+ write_raw(bytes)
592
+ end
593
+ @pending_bloom_filters.clear
467
594
  end
468
595
 
469
596
  def use_dictionary?(col)
@@ -497,14 +624,15 @@ module Herringbone
497
624
  [dict, values.map(&index)]
498
625
  end
499
626
 
500
- # Splits a column buffer into pages of roughly @page_size bytes. Repeated columns
627
+ # Splits a column buffer into pages of roughly @page_bytes bytes. Repeated columns
501
628
  # are only cut where a new row starts.
502
629
  def page_ranges(buf, value_bytes)
503
630
  n = buf.defs.size
504
631
  bytes = n + value_bytes
505
- pages = (bytes + @page_size - 1) / @page_size
506
- return [[0, n]] if pages <= 1
507
- per = (n + pages - 1) / pages
632
+ pages = (bytes + @page_bytes - 1) / @page_bytes
633
+ per = pages <= 1 ? n : (n + pages - 1) / pages
634
+ per = @page_rows if per > @page_rows
635
+ return [[0, n]] if per >= n
508
636
  reps = buf.reps
509
637
  ranges = []
510
638
  start = 0
@@ -589,46 +717,86 @@ module Herringbone
589
717
  write_page(header, "".b, rep_bytes + def_bytes + compressed)
590
718
  end
591
719
 
592
- def statistics_for(col, defs, values)
720
+ STAT_TRUNCATE_BYTES = 64
721
+ IDENTITY = ->(v) { v }
722
+
723
+ def statistics_for(col, defs, values, order)
593
724
  max_def = col.max_definition_level
594
725
  nulls = max_def.zero? ? 0 : defs.size - defs.count(max_def)
595
726
  stats = Format::Statistics.new(null_count: nulls)
596
- min, max = min_max(col, values)
597
- if min
598
- stats.min_value = min
599
- stats.max_value = max
600
- stats.is_min_value_exact = true
601
- stats.is_max_value_exact = true
727
+ range = order && value_range(col, values, order)
728
+ if range
729
+ min = stat_bytes(col, range[0])
730
+ max = stat_bytes(col, range[1])
731
+ stats.min_value = truncate_min(min)
732
+ stats.max_value = truncate_max(max)
733
+ stats.is_min_value_exact = stats.min_value.bytesize == min.bytesize
734
+ stats.is_max_value_exact = stats.max_value.bytesize == max.bytesize
602
735
  end
603
736
  stats
604
737
  end
605
738
 
606
- # Min/max for types whose Parquet sort order matches Ruby's comparison of the physical values
607
- def min_max(col, values)
608
- return nil if values.empty?
739
+ # A key giving the Parquet sort order of the column's physical values (IDENTITY when Ruby's own
740
+ # comparison already matches), or nil when the order is undefined (INT96)
741
+ def sort_key(col)
609
742
  kind, _, signed = Types.logical_of(col.node)
610
743
  case col.type
611
- when T::BOOLEAN
612
- [values.include?(false) ? "\x00".b : "\x01".b, values.include?(true) ? "\x01".b : "\x00".b]
744
+ when T::BOOLEAN then ->(v) { v ? 1 : 0 }
613
745
  when T::INT32, T::INT64
614
- return nil if kind == :integer && !signed
615
- fmt = col.type == T::INT32 ? "l<" : "q<"
616
- min, max = values.minmax
617
- [[min].pack(fmt), [max].pack(fmt)]
618
- when T::FLOAT, T::DOUBLE
619
- finite = values.reject(&:nan?)
620
- return nil if finite.empty?
621
- min, max = finite.minmax
746
+ return IDENTITY unless kind == :integer && !signed
747
+ mask = col.type == T::INT32 ? 0xFFFF_FFFF : 0xFFFF_FFFF_FFFF_FFFF
748
+ ->(v) { v & mask }
749
+ when T::FLOAT, T::DOUBLE then IDENTITY
750
+ when T::BYTE_ARRAY, T::FIXED_LEN_BYTE_ARRAY
751
+ case kind
752
+ when :decimal then Types.method(:be_to_int)
753
+ when :float16 then ->(v) { Types.half_to_float(v.unpack1("v")) }
754
+ else IDENTITY # unsigned lexicographic, which is how Ruby compares binary Strings
755
+ end
756
+ end
757
+ end
758
+
759
+ # [min, max] of +values+ in column order, ignoring NaNs; nil if there is nothing to compare
760
+ def value_range(col, values, order)
761
+ floats = col.type == T::FLOAT || col.type == T::DOUBLE
762
+ if floats
763
+ values = values.reject(&:nan?)
764
+ elsif Types.logical_of(col.node).first == :float16
765
+ values = values.reject { |v| order.call(v).nan? }
766
+ end
767
+ return nil if values.empty?
768
+ min, max = order.equal?(IDENTITY) ? values.minmax : values.minmax_by(&order)
769
+ if floats
622
770
  min = -0.0 if min.zero?
623
771
  max = 0.0 if max.zero?
624
- fmt = col.type == T::FLOAT ? "e" : "E"
625
- [[min].pack(fmt), [max].pack(fmt)]
626
- when T::BYTE_ARRAY
627
- return nil if kind == :decimal
628
- min, max = values.minmax
629
- return nil if min.bytesize > 1024 || max.bytesize > 1024
630
- [min, max]
631
772
  end
773
+ [min, max]
774
+ end
775
+
776
+ def stat_bytes(col, value)
777
+ case col.type
778
+ when T::BOOLEAN then value ? "\x01".b : "\x00".b
779
+ when T::INT32 then [value].pack("l<")
780
+ when T::INT64 then [value].pack("q<")
781
+ when T::FLOAT then [value].pack("e")
782
+ when T::DOUBLE then [value].pack("E")
783
+ else value.b
784
+ end
785
+ end
786
+
787
+ # Long byte-array bounds are truncated: a prefix is still a lower bound for the minimum, and
788
+ # a prefix with its last byte incremented is an upper bound for the maximum
789
+ def truncate_min(bytes)
790
+ bytes.bytesize > STAT_TRUNCATE_BYTES ? bytes.byteslice(0, STAT_TRUNCATE_BYTES) : bytes
791
+ end
792
+
793
+ def truncate_max(bytes)
794
+ return bytes if bytes.bytesize <= STAT_TRUNCATE_BYTES
795
+ prefix = bytes.byteslice(0, STAT_TRUNCATE_BYTES).bytes
796
+ prefix.pop while prefix.last == 0xFF
797
+ return bytes if prefix.empty?
798
+ prefix[-1] += 1
799
+ prefix.pack("C*")
632
800
  end
633
801
  end
634
802
  end