herringbone 0.1.0 → 0.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -8,38 +8,54 @@ module Herringbone
8
8
  # string :name
9
9
  # list :tags, :string
10
10
  # end
11
- # Herringbone::Writer.open("out.parquet", schema) do |w|
12
- # w << { "id" => 1, "name" => "one", "tags" => ["a", "b"] }
13
- # w << [2, "two", []] # Arrays are taken in schema order
14
- # w << order # objects responding to #attributes (ActiveRecord) or #to_h
11
+ # File.open("out.parquet", "wb") do |file|
12
+ # Herringbone::Writer.open(file, schema) do |w|
13
+ # w << { "id" => 1, "name" => "one", "tags" => ["a", "b"] }
14
+ # w << [2, "two", []] # Arrays are taken in schema order
15
+ # w << order # objects responding to #attributes (ActiveRecord) or #to_h
16
+ # end
15
17
  # end
16
18
  #
17
- # +schema+ is a Herringbone::Schema or a Hash spec (see Schema.define). The target is a path or an IO;
18
- # paths are written to a temporary file next to the target and renamed into place on close, so a
19
- # failed write never leaves a truncated file behind.
19
+ # The output is any IO that responds to #write (a File, StringIO, Tempfile, socket, pipe...); it
20
+ # is written sequentially and never seeked, rewound or closed by the writer; only #write is
21
+ # required, its return value is ignored, and #binmode and #flush are called when available.
22
+ # Herringbone does not open files by path.
20
23
  #
21
24
  # Options:
22
- # compression: :zstd (default), :snappy, :gzip, :lz4 (LZ4_RAW), :brotli, :none
25
+ # compression: :snappy (default), :zstd, :gzip, :lz4 (LZ4_RAW), :lz4_hadoop, :brotli, :none
26
+ # (:zstd and :brotli need the zstd-ruby / brotli gems)
23
27
  # row_group_bytes: flush a row group once the buffered values take roughly this much memory
24
28
  # (default 16MB). This bounds memory use while writing.
25
- # row_group_size: also flush after this many rows (default: no row limit)
26
- # page_size: approximate uncompressed data page size in bytes (default 1MB)
29
+ # row_group_rows: also flush after this many rows (default: no row limit)
30
+ # page_bytes: approximate uncompressed data page size (default 1MB)
31
+ # page_rows: at most this many rows per data page (default 20_000), which keeps the page
32
+ # index selective
27
33
  # data_page_version: 1 (default) or 2
28
34
  # dictionary: true/false, or an Array of column paths to dictionary-encode
29
35
  # encodings: { "path.to.column" => :delta_binary_packed, ... } for non-dictionary pages
30
- # statistics: write min/max/null_count statistics (default true)
31
36
  # metadata: Hash of String => String key/value metadata for the footer
37
+ # bloom_filters: write split block bloom filters: true (every column that supports them),
38
+ # an Array of column paths, or { "path" => true | { ndv:, fpp:, max_bytes: } }.
39
+ # Without ndv: the distinct values of each row group are counted. fpp defaults
40
+ # to 0.01 and max_bytes to 1MB. Filters are written after each row group.
41
+ #
42
+ # Statistics and page indexes (ColumnIndex/OffsetIndex) are always written.
32
43
  class Writer
33
- MAGIC = "PAR1"
44
+ # Marker at the start and end of every Parquet file
45
+ MAGIC = "PAR1".b.freeze
46
+ # Shorthand for Format::Type (physical types)
34
47
  T = Format::Type
48
+ # Shorthand for Format::Encoding
35
49
  E = Format::Encoding
36
50
 
51
+ # Names accepted in the +encodings:+ option => Parquet encoding id
37
52
  ENCODING_NAMES = {
38
53
  plain: E::PLAIN, rle: E::RLE, delta_binary_packed: E::DELTA_BINARY_PACKED,
39
54
  delta_length_byte_array: E::DELTA_LENGTH_BYTE_ARRAY, delta_byte_array: E::DELTA_BYTE_ARRAY,
40
55
  byte_stream_split: E::BYTE_STREAM_SPLIT
41
56
  }.freeze
42
57
 
58
+ # Encoding id => physical types it may be used for, per the Parquet encodings spec
43
59
  VALID_ENCODINGS = {
44
60
  E::PLAIN => T::NAMES.keys,
45
61
  E::RLE => [T::BOOLEAN],
@@ -49,20 +65,47 @@ module Herringbone
49
65
  E::BYTE_STREAM_SPLIT => [T::INT32, T::INT64, T::FLOAT, T::DOUBLE, T::FIXED_LEN_BYTE_ARRAY]
50
66
  }.freeze
51
67
 
68
+ # Largest dictionary build_dictionary keeps; above it the chunk is written without a dictionary.
69
+ # Byte-array columns are dictionary-encoded by ByteValues, which applies its own
70
+ # ByteValues::MAX_DICTIONARY_BYTES as values arrive.
52
71
  MAX_DICTIONARY_BYTES = 1024 * 1024
53
72
  # The row group byte size is first estimated after this many rows, then after every row group
54
73
  ESTIMATE_AFTER_ROWS = 1000
55
74
 
75
+ # @return [Schema] schema the rows are written with
56
76
  attr_reader :schema
57
77
 
58
- # Opens a writer on a path or an IO. With a block, the file is finished when the block returns
59
- # (or discarded if it raises) and the block's value is returned; without one, call #close.
60
- def self.open(target, schema, **options)
61
- writer = new(target, schema, **options)
78
+ # Opens a writer on +io+. With a block, the file is finished (footer written) when the block
79
+ # returns and the block's value is returned; if the block raises, the writer is aborted and no
80
+ # footer is written. Without a block, call #close to finish. The IO is never closed.
81
+ #
82
+ # @param io [IO, #write] destination
83
+ # @param schema [Schema] schema of the rows
84
+ # @param options [Hash{Symbol => Object}] see the class description and #initialize
85
+ # @option options [Symbol] :compression (:snappy) codec, see Herringbone.codecs
86
+ # @option options [Integer] :row_group_bytes (16MB) approximate buffered size that triggers a row group
87
+ # @option options [Integer, nil] :row_group_rows (nil) also flush a row group after this many rows
88
+ # @option options [Integer] :page_bytes (1MB) approximate uncompressed data page size
89
+ # @option options [Integer] :page_rows (20_000) maximum rows per data page
90
+ # @option options [Integer] :data_page_version (1) 1 or 2
91
+ # @option options [Boolean, Array<String>] :dictionary (true) dictionary-encode all eligible columns,
92
+ # none, or only the listed dotted column paths
93
+ # @option options [Hash{String => Symbol}] :encodings ({}) dotted column path => value encoding
94
+ # for non-dictionary pages
95
+ # @option options [Hash{String => String}] :metadata ({}) key/value metadata for the footer
96
+ # @option options [Boolean, Array<String>, Hash{String => Boolean, Hash}] :bloom_filters (nil)
97
+ # columns to write split block bloom filters for
98
+ # @yield [writer] the open writer
99
+ # @yieldparam writer [Writer] writer to append rows to
100
+ # @yieldreturn [Object] returned by open
101
+ # @return [Writer, Object] the writer without a block, the block's value with one
102
+ # @raise [ArgumentError] for an invalid IO, schema or option
103
+ def self.open(io, schema, **options)
104
+ writer = new(io, schema, **options)
62
105
  return writer unless block_given?
63
106
  begin
64
107
  result = yield writer
65
- rescue Exception # rubocop:disable Lint/RescueException -- also discard on Interrupt
108
+ rescue Exception # rubocop:disable Lint/RescueException -- also abort on Interrupt
66
109
  writer.abort
67
110
  raise
68
111
  end
@@ -70,35 +113,72 @@ module Herringbone
70
113
  result
71
114
  end
72
115
 
73
- def initialize(target, schema, compression: :zstd, row_group_bytes: 16 * 1024 * 1024, row_group_size: nil,
74
- page_size: 1024 * 1024, data_page_version: 1, dictionary: true, encodings: {}, statistics: true, metadata: {})
75
- @schema = Schema.coerce(schema)
76
- schema = @schema
116
+ # Validates the options and writes the leading magic bytes to +io+.
117
+ #
118
+ # @param io [IO, #write] destination; switched to binmode when it supports that
119
+ # @param schema [Schema] schema of the rows
120
+ # @param compression [Symbol, Integer] codec name (see Herringbone.codecs) or Format::Codec id
121
+ # @param row_group_bytes [Integer] approximate buffered size that triggers a row group
122
+ # @param row_group_rows [Integer, nil] also flush a row group after this many rows
123
+ # @param page_bytes [Integer] approximate uncompressed data page size
124
+ # @param page_rows [Integer] maximum level entries per data page (pages of repeated columns
125
+ # extend to the next row start)
126
+ # @param data_page_version [Integer] 1 or 2
127
+ # @param dictionary [Boolean, Array<String>] true for every column except BOOLEAN, FLOAT, DOUBLE
128
+ # and those listed in +encodings+; false for none; or the dotted paths of the columns to encode
129
+ # @param encodings [Hash{String, Symbol => Symbol, Integer}] dotted column path => value encoding
130
+ # (a name from ENCODING_NAMES or an encoding id) for non-dictionary pages
131
+ # @param metadata [Hash{#to_s => #to_s}] key/value metadata for the footer
132
+ # @param bloom_filters [Boolean, Array<String>, Hash{String => Boolean, Hash}, nil] see the class
133
+ # description
134
+ # @raise [ArgumentError] for an invalid IO, schema or option
135
+ # @raise [MissingCodecError] when the codec's optional gem cannot be loaded
136
+ # @raise [UnsupportedError] when the codec is not supported
137
+ def initialize(io, schema, compression: :snappy, row_group_bytes: 16 * 1024 * 1024, row_group_rows: nil,
138
+ page_bytes: 1024 * 1024, page_rows: 20_000, data_page_version: 1, dictionary: true, encodings: {},
139
+ metadata: {}, bloom_filters: nil)
140
+ raise ArgumentError, "Expected a Herringbone::Schema, got #{schema.class}" unless schema.is_a?(Schema)
141
+ @schema = schema
77
142
  @codec = Compression.codec_id(compression)
143
+ # Fail before creating any file if the codec's library is missing
144
+ Compression.ensure_available!(@codec)
78
145
  @row_group_bytes = Integer(row_group_bytes)
79
- @row_group_size = row_group_size && Integer(row_group_size)
80
- @row_limit = @row_group_size || ESTIMATE_AFTER_ROWS
81
- @page_size = Integer(page_size)
146
+ @row_group_rows = row_group_rows && Integer(row_group_rows)
147
+ @row_limit = @row_group_rows || ESTIMATE_AFTER_ROWS
148
+ @page_bytes = Integer(page_bytes)
149
+ @page_rows = Integer(page_rows)
150
+ raise ArgumentError, "page_rows must be positive" unless @page_rows.positive?
151
+ @page_indexes = [] # [ColumnChunk, ColumnIndex or nil, OffsetIndex] per written column chunk
82
152
  @data_page_version = Integer(data_page_version)
83
153
  raise ArgumentError, "data_page_version must be 1 or 2" unless [1, 2].include?(@data_page_version)
84
154
  @dictionary = dictionary
85
155
  @encodings = encodings.to_h { |path, enc| [path.to_s, encoding_id(path, enc)] }
86
- unknown = @encodings.keys - schema.columns.map(&:dotted_path)
156
+ columns = schema.columns.to_h { |c| [c.dotted_path, c] }
157
+ unknown = @encodings.keys - columns.keys
87
158
  raise ArgumentError, "encodings: no such column #{unknown.join(", ")}" unless unknown.empty?
88
- @statistics = statistics
159
+ @encodings.each { |path, enc| check_encoding!(columns[path], enc) }
89
160
  @metadata = metadata
161
+ @bloom_filters = bloom_filter_config(bloom_filters)
162
+ @pending_bloom_filters = [] # [ColumnMetaData, BloomFilter] for the row group being written
90
163
  @row_groups = []
91
164
  @total_rows = 0
92
165
  @pos = 0
93
166
  @closed = false
94
167
  @bytes_per_row = nil
95
- open_target(target)
168
+ @io = check_io!(io)
96
169
  write_raw(MAGIC)
97
170
  reset_buffers
98
171
  end
99
172
 
100
173
  # Appends a row: a Hash keyed by top-level field names (Strings or Symbols), an Array of values
101
- # in schema order, or an object responding to #attributes (ActiveRecord) or #to_h (Struct, Data)
174
+ # in schema order, or an object responding to #attributes (ActiveRecord) or #to_h (Struct, Data).
175
+ # A row that fails to encode leaves nothing behind in the buffers. Flushes a row group when the
176
+ # buffered rows reach the size limits.
177
+ #
178
+ # @param row [Hash, Array, #attributes, #to_h] row to append
179
+ # @return [self]
180
+ # @raise [Error] when the writer is closed
181
+ # @raise [EncodeError] when a required field is nil or missing, or a value cannot be encoded
102
182
  def <<(row)
103
183
  raise Error, "Writer is closed" if @closed
104
184
  row = row_hash(row)
@@ -123,7 +203,7 @@ module Herringbone
123
203
  rescue EncodeError => e
124
204
  rollback_row(marks)
125
205
  raise EncodeError, "Row #{@total_rows + @buffered_rows}: #{e.message}"
126
- rescue StandardError
206
+ rescue
127
207
  rollback_row(marks)
128
208
  raise
129
209
  end
@@ -131,17 +211,15 @@ module Herringbone
131
211
  check_row_group_size if @buffered_rows >= @row_limit
132
212
  self
133
213
  end
134
- alias_method :write, :<<
135
-
136
- def write_rows(rows)
137
- rows.each { |row| self << row }
138
- self
139
- end
140
214
 
141
215
  # Number of rows written so far, including buffered ones
216
+ #
217
+ # @return [Integer]
142
218
  def rows_written = @total_rows + @buffered_rows
143
219
 
144
220
  # Writes any buffered rows as a row group
221
+ #
222
+ # @return [void]
145
223
  def flush_row_group
146
224
  return if @buffered_rows.zero?
147
225
  start = @pos
@@ -154,20 +232,26 @@ module Herringbone
154
232
  total_compressed_size: @pos - start,
155
233
  ordinal: @row_groups.size
156
234
  )
235
+ write_bloom_filters
157
236
  @total_rows += @buffered_rows
158
237
  reset_buffers
159
238
  end
160
239
 
240
+ # Flushes buffered rows, then writes the page indexes and the footer and flushes the IO.
241
+ # Does nothing when already closed or aborted. The IO itself is not closed.
242
+ #
243
+ # @return [void]
161
244
  def close
162
245
  return if @closed
163
246
  flush_row_group
247
+ write_page_indexes
164
248
  meta = Format::FileMetaData.new(
165
249
  version: 2,
166
250
  schema: @schema.to_elements,
167
251
  num_rows: @total_rows,
168
252
  row_groups: @row_groups,
169
253
  key_value_metadata: @metadata.empty? ? nil : @metadata.map { |k, v| Format::KeyValue.new(key: k.to_s, value: v&.to_s) },
170
- created_by: "herringbone version #{VERSION}",
254
+ created_by: "herringbone-ruby #{VERSION}",
171
255
  column_orders: @schema.columns.map { Format::ColumnOrder.new(type_order: Format::TypeDefinedOrder.new) }
172
256
  )
173
257
  footer = meta.encode
@@ -176,24 +260,51 @@ module Herringbone
176
260
  write_raw(MAGIC)
177
261
  @io.flush if @io.respond_to?(:flush)
178
262
  @closed = true
179
- finish_target
180
263
  end
181
264
 
182
- # Stops writing without producing a file: a temporary file is deleted, an IO is left as is
265
+ # Stops writing without finishing the file (no footer is written). Whatever was already written
266
+ # to the IO stays there; discarding it is up to the caller.
267
+ #
268
+ # @return [void]
183
269
  def abort
184
- return if @closed
185
270
  @closed = true
186
- return unless @temp_path
187
- @io.close unless @io.closed?
188
- File.unlink(@temp_path) if File.exist?(@temp_path)
189
271
  end
190
272
 
191
273
  private
192
274
 
275
+ # Pathname responds to #write too (writing a whole file by path), so it is rejected explicitly.
276
+ # A text-mode IO (a pipe, or a File opened with "w") transcodes what is written when
277
+ # Encoding.default_internal is set, as Rails does, and binary pages cannot be transcoded,
278
+ # so the IO is switched to binary mode.
279
+ #
280
+ # @param io [Object] candidate destination
281
+ # @return [IO, #write] +io+, in binary mode when it supports #binmode
282
+ # @raise [ArgumentError] when +io+ does not respond to #write, or is a String or Pathname
283
+ def check_io!(io)
284
+ if io.respond_to?(:write) && !io.is_a?(String) && !(defined?(Pathname) && io.is_a?(Pathname))
285
+ io.binmode if io.respond_to?(:binmode)
286
+ return io
287
+ end
288
+ raise ArgumentError, "Herringbone::Writer expects an IO that responds to #write " \
289
+ "(e.g. File.open(path, \"wb\") or StringIO.new), got #{io.class}"
290
+ end
291
+
193
292
  # Levels are kept as binary Strings (one byte per entry). Values are an Array for numeric and
194
293
  # boolean columns, and a compact ByteValues for BYTE_ARRAY / FIXED_LEN_BYTE_ARRAY columns.
294
+ #
295
+ # @!attribute defs
296
+ # @return [String, Array<Integer>] definition levels (unpacked to an Array while a chunk is written)
297
+ # @!attribute reps
298
+ # @return [String, Array<Integer>, nil] repetition levels, like +defs+; nil for non-repeated columns
299
+ # @!attribute values
300
+ # @return [Array, ByteValues] non-null physical values
195
301
  ColumnBuffer = Struct.new(:defs, :reps, :values)
196
302
 
303
+ # Starts a new row group: fresh buffers per column, and the per-field write plan
304
+ # (field, name, Symbol name, buffer and encoder for flat fields, or nils for shredded ones).
305
+ #
306
+ # @return [void]
307
+ # @raise [UnsupportedError] when a column has more than 255 definition or repetition levels
197
308
  def reset_buffers
198
309
  @buffers = @schema.columns.map do |col|
199
310
  if col.max_definition_level > 255 || col.max_repetition_level > 255
@@ -212,6 +323,8 @@ module Herringbone
212
323
  @buffered_rows = 0
213
324
  end
214
325
 
326
+ # @param col [Schema::Column] column to buffer
327
+ # @return [ByteValues, Array] empty value store for the column
215
328
  def new_values_store(col)
216
329
  case col.type
217
330
  when T::BYTE_ARRAY then ByteValues.new(dictionary: use_dictionary?(col))
@@ -220,25 +333,10 @@ module Herringbone
220
333
  end
221
334
  end
222
335
 
223
- def open_target(target)
224
- if target.respond_to?(:write)
225
- @io = target
226
- else
227
- @path = target.to_s
228
- dir = File.dirname(@path)
229
- @temp_path = File.join(dir, ".#{File.basename(@path)}.#{Process.pid}.#{rand(1 << 32).to_s(36)}.tmp")
230
- @io = File.open(@temp_path, "wb")
231
- end
232
- end
233
-
234
- def finish_target
235
- return unless @temp_path
236
- @io.close
237
- File.rename(@temp_path, @path)
238
- end
239
-
240
- # Flushes when the buffered values reach row_group_bytes (or row_group_size rows). The bytes per
336
+ # Flushes when the buffered values reach row_group_bytes (or row_group_rows rows). The bytes per
241
337
  # row are estimated from the buffered values after the first rows, then refreshed per row group.
338
+ #
339
+ # @return [void]
242
340
  def check_row_group_size
243
341
  @bytes_per_row ||= estimate_bytes_per_row
244
342
  limit = row_limit_for(@bytes_per_row)
@@ -251,11 +349,15 @@ module Herringbone
251
349
  end
252
350
  end
253
351
 
352
+ # @param bytes_per_row [Integer] estimated buffered bytes per row
353
+ # @return [Integer] rows to buffer before the next size check: what fits in row_group_bytes,
354
+ # at least ESTIMATE_AFTER_ROWS, at most row_group_rows
254
355
  def row_limit_for(bytes_per_row)
255
356
  by_bytes = [@row_group_bytes / [bytes_per_row, 1].max, ESTIMATE_AFTER_ROWS].max
256
- @row_group_size ? [@row_group_size, by_bytes].min : by_bytes
357
+ @row_group_rows ? [@row_group_rows, by_bytes].min : by_bytes
257
358
  end
258
359
 
360
+ # @return [Integer] memory held by the buffers divided by the buffered rows
259
361
  def estimate_bytes_per_row
260
362
  bytes = @schema.columns.sum do |col|
261
363
  buf = @buffers[col.index]
@@ -263,29 +365,58 @@ module Herringbone
263
365
  held = if values.is_a?(ByteValues)
264
366
  values.memory_bytes
265
367
  else
266
- values.size * (col.type == T::INT96 ? 48 : 8)
368
+ values.size * ((col.type == T::INT96) ? 48 : 8)
267
369
  end
268
370
  held + buf.defs.bytesize + (buf.reps ? buf.reps.bytesize : 0)
269
371
  end
270
372
  bytes / [@buffered_rows, 1].max
271
373
  end
272
374
 
375
+ # Writes to the IO and tracks the file offset, since the IO is never asked for its position
376
+ #
377
+ # @param bytes [String] binary data
378
+ # @return [void]
273
379
  def write_raw(bytes)
274
380
  @io.write(bytes)
275
381
  @pos += bytes.bytesize
276
382
  end
277
383
 
384
+ # @param path [String, Symbol] column path, for the error message
385
+ # @param enc [Symbol, String, Integer] encoding name from ENCODING_NAMES, or an encoding id
386
+ # @return [Integer] encoding id
387
+ # @raise [ArgumentError] for an unknown encoding name, or an id that is not in VALID_ENCODINGS
278
388
  def encoding_id(path, enc)
279
- return enc if enc.is_a?(Integer)
389
+ if enc.is_a?(Integer)
390
+ return enc if VALID_ENCODINGS.key?(enc)
391
+ raise ArgumentError, "Encoding #{E::NAMES.fetch(enc, enc)} cannot be requested for #{path}"
392
+ end
280
393
  ENCODING_NAMES.fetch(enc.to_s.downcase.to_sym) { raise ArgumentError, "Unknown encoding #{enc.inspect} for #{path}" }
281
394
  end
282
395
 
396
+ # @param col [Schema::Column] column the encoding is requested for
397
+ # @param enc [Integer] encoding id, a key of VALID_ENCODINGS
398
+ # @return [void]
399
+ # @raise [ArgumentError] when the encoding is not valid for the column's physical type
400
+ def check_encoding!(col, enc)
401
+ return if VALID_ENCODINGS.fetch(enc).include?(col.type)
402
+ raise ArgumentError, "Encoding #{E::NAMES[enc]} is not valid for #{T::NAMES[col.type]} column #{col.dotted_path}"
403
+ end
404
+
405
+ # Reads a struct member, by String or Symbol key
406
+ #
407
+ # @param hash [Hash, #to_h, nil] struct value
408
+ # @param name [String] member name
409
+ # @return [Object, nil] member value, nil when +hash+ is nil or has no such key
410
+ # @raise [EncodeError] when +hash+ cannot be converted to a Hash
283
411
  def lookup(hash, name)
284
412
  return nil if hash.nil?
285
413
  hash = as_hash(hash, name) unless hash.is_a?(Hash)
286
414
  hash.fetch(name) { hash[name.to_sym] }
287
415
  end
288
416
 
417
+ # @param row [Hash, Array, #attributes, #to_h] row given to #<<
418
+ # @return [Hash] the row keyed by top-level field names
419
+ # @raise [EncodeError] when an Array has the wrong size or the row cannot be converted to a Hash
289
420
  def row_hash(row)
290
421
  case row
291
422
  when Hash then row
@@ -300,6 +431,10 @@ module Herringbone
300
431
  end
301
432
  end
302
433
 
434
+ # @param value [Hash, #to_h] value to convert
435
+ # @param what [String] description of the value, for the error message
436
+ # @return [Hash]
437
+ # @raise [EncodeError] when +value+ does not respond to #to_h
303
438
  def as_hash(value, what)
304
439
  return value if value.is_a?(Hash)
305
440
  raise EncodeError, "Expected a Hash for #{what}, got #{value.class}" unless value.respond_to?(:to_h)
@@ -307,6 +442,10 @@ module Herringbone
307
442
  end
308
443
 
309
444
  # Removes the entries a failed row left behind, so the buffers stay aligned
445
+ #
446
+ # @param marks [Array<Array(Integer, Integer, Integer)>, nil] sizes of defs, reps and values of
447
+ # each nested buffer before the row, nil when there are none
448
+ # @return [void]
310
449
  def rollback_row(marks)
311
450
  @plan.each do |field, _, _, buf|
312
451
  next unless buf && buf.defs.bytesize > @buffered_rows
@@ -323,6 +462,14 @@ module Herringbone
323
462
  end
324
463
 
325
464
  # Record shredding: turns a nested value into (definition level, repetition level, value) entries
465
+ #
466
+ # @param field [Schema::Field] field the value belongs to
467
+ # @param value [Object, nil] Ruby value of the field
468
+ # @param parent_def [Integer] definition level recorded when +value+ is nil
469
+ # @param rep [Integer] repetition level of the first entry this value produces
470
+ # @return [void]
471
+ # @raise [EncodeError] when a required value is nil, a list or map value has the wrong type, a
472
+ # map key is nil, or a leaf value cannot be encoded
326
473
  def shred(field, value, parent_def, rep)
327
474
  if value.nil?
328
475
  raise EncodeError, "Field #{field.node.path.join(".")} is required but got nil" unless field.optional
@@ -379,6 +526,13 @@ module Herringbone
379
526
  end
380
527
  end
381
528
 
529
+ # Writes one column chunk of the current row group: an optional dictionary page, then data pages.
530
+ # Bloom filters and page indexes are queued, to be written after the row group and before the
531
+ # footer.
532
+ #
533
+ # @param col [Schema::Column] column being written
534
+ # @param buffer [ColumnBuffer] the column's buffered levels and values
535
+ # @return [Format::ColumnChunk] chunk with its ColumnMetaData, for the row group
382
536
  def write_column_chunk(col, buffer)
383
537
  type = col.type
384
538
  path = col.dotted_path
@@ -399,11 +553,7 @@ module Herringbone
399
553
  value_encoding = if dict_values
400
554
  E::RLE_DICTIONARY
401
555
  else
402
- enc = @encodings[path] || E::PLAIN
403
- unless VALID_ENCODINGS.fetch(enc).include?(type)
404
- raise ArgumentError, "Encoding #{E::NAMES[enc]} is not valid for #{T::NAMES[type]} column #{path}"
405
- end
406
- enc
556
+ @encodings[path] || E::PLAIN
407
557
  end
408
558
 
409
559
  chunk_start = @pos
@@ -430,6 +580,9 @@ module Herringbone
430
580
  else
431
581
  values.sum(&:bytesize) + 4 * values.size
432
582
  end
583
+ order = sort_key(col)
584
+ pages = []
585
+ first_row = 0
433
586
  page_ranges(buf, value_bytes).each do |from, to|
434
587
  n = to - from
435
588
  defs = buf.defs[from, n]
@@ -437,6 +590,12 @@ module Herringbone
437
590
  non_null = max_def.zero? ? n : defs.count(max_def)
438
591
  page_values = (indices || values)[value_index, non_null]
439
592
  value_index += non_null
593
+ page_offset = @pos
594
+ range = nil
595
+ if order && non_null.positive?
596
+ in_page = dict_values ? page_values.uniq.map { |i| dict_values[i] } : page_values
597
+ range = value_range(col, in_page, order)
598
+ end
440
599
  encoded = encode_values(page_values, value_encoding, type, col.type_length, dict_values&.size)
441
600
  rep_bytes = max_rep.positive? ? Encodings::RLE.encode_hybrid(reps, max_rep.bit_length) : "".b
442
601
  def_bytes = max_def.positive? ? Encodings::RLE.encode_hybrid(defs, max_def.bit_length) : "".b
@@ -446,6 +605,8 @@ module Herringbone
446
605
  num_rows = max_rep.zero? ? n : reps.count(0)
447
606
  write_data_page_v2(n, n - non_null, num_rows, rep_bytes, def_bytes, encoded, value_encoding)
448
607
  end
608
+ pages << PageInfo.new(page_offset, @pos - page_offset, first_row, n - non_null, non_null, range)
609
+ first_row += max_rep.zero? ? n : reps.count(0)
449
610
  end
450
611
 
451
612
  encodings = [E::RLE]
@@ -461,11 +622,178 @@ module Herringbone
461
622
  total_compressed_size: @pos - chunk_start,
462
623
  data_page_offset: data_offset,
463
624
  dictionary_page_offset: dictionary_offset,
464
- statistics: @statistics ? statistics_for(col, buf.defs, values || indices.uniq.map { |i| dict_values[i] }) : nil
625
+ statistics: statistics_for(col, buf.defs, values || indices.uniq.map { |i| dict_values[i] }, order)
626
+ )
627
+ chunk = Format::ColumnChunk.new(file_offset: chunk_start, meta_data: meta)
628
+ if (bloom = @bloom_filters[path])
629
+ @pending_bloom_filters << [meta, build_bloom_filter(col, bloom, dict_values, values)]
630
+ end
631
+ @page_indexes << [chunk, column_index_for(col, pages, order), offset_index_for(pages)]
632
+ chunk
633
+ end
634
+
635
+ # What the page indexes need to know about a written data page
636
+ #
637
+ # @!attribute offset
638
+ # @return [Integer] file offset of the page header
639
+ # @!attribute size
640
+ # @return [Integer] page size in the file, header included
641
+ # @!attribute first_row
642
+ # @return [Integer] index of the page's first row within the row group
643
+ # @!attribute nulls
644
+ # @return [Integer] null entries in the page
645
+ # @!attribute non_null
646
+ # @return [Integer] non-null values in the page
647
+ # @!attribute range
648
+ # @return [Array(Object, Object), nil] [min, max] physical values, nil when there are none
649
+ PageInfo = Struct.new(:offset, :size, :first_row, :nulls, :non_null, :range)
650
+
651
+ # @param pages [Array<PageInfo>] pages of a column chunk
652
+ # @return [Format::OffsetIndex]
653
+ def offset_index_for(pages)
654
+ Format::OffsetIndex.new(page_locations: pages.map do |p|
655
+ Format::PageLocation.new(offset: p.offset, compressed_page_size: p.size, first_row_index: p.first_row)
656
+ end)
657
+ end
658
+
659
+ # nil when the column has no defined sort order, or a page's values have no min/max (all NaN)
660
+ #
661
+ # @param col [Schema::Column] column the pages belong to
662
+ # @param pages [Array<PageInfo>] pages of the column chunk
663
+ # @param order [Proc, Method, nil] sort key from #sort_key
664
+ # @return [Format::ColumnIndex, nil]
665
+ def column_index_for(col, pages, order)
666
+ return nil unless order
667
+ return nil if pages.any? { |p| p.non_null.positive? && p.range.nil? }
668
+ Format::ColumnIndex.new(
669
+ null_pages: pages.map { |p| p.non_null.zero? },
670
+ min_values: pages.map { |p| p.range ? truncate_min(stat_bytes(col, p.range[0])) : "".b },
671
+ max_values: pages.map { |p| p.range ? truncate_max(stat_bytes(col, p.range[1])) : "".b },
672
+ boundary_order: boundary_order(pages.filter_map(&:range), order),
673
+ null_counts: pages.map(&:nulls)
465
674
  )
466
- Format::ColumnChunk.new(file_offset: chunk_start, meta_data: meta)
467
675
  end
468
676
 
677
+ # @param ranges [Array<Array(Object, Object)>] [min, max] of each non-null page, in page order
678
+ # @param order [Proc, Method] sort key from #sort_key
679
+ # @return [Integer] Format::BoundaryOrder: ASCENDING when both mins and maxes never decrease
680
+ # (or there are fewer than two pages), DESCENDING when they never increase, else UNORDERED
681
+ def boundary_order(ranges, order)
682
+ return Format::BoundaryOrder::ASCENDING if ranges.size < 2
683
+ cmp = ->(a, b) { order.equal?(IDENTITY) ? a <=> b : order.call(a) <=> order.call(b) }
684
+ pairs = ranges.each_cons(2)
685
+ if pairs.all? { |(a_min, a_max), (b_min, b_max)| cmp.call(a_min, b_min) <= 0 && cmp.call(a_max, b_max) <= 0 }
686
+ Format::BoundaryOrder::ASCENDING
687
+ elsif pairs.all? { |(a_min, a_max), (b_min, b_max)| cmp.call(a_min, b_min) >= 0 && cmp.call(a_max, b_max) >= 0 }
688
+ Format::BoundaryOrder::DESCENDING
689
+ else
690
+ Format::BoundaryOrder::UNORDERED
691
+ end
692
+ end
693
+
694
+ # Page indexes go after the last row group: all column indexes, then all offset indexes
695
+ #
696
+ # @return [void]
697
+ def write_page_indexes
698
+ @page_indexes.each do |chunk, column_index, _|
699
+ next unless column_index
700
+ bytes = column_index.encode
701
+ chunk.column_index_offset = @pos
702
+ chunk.column_index_length = bytes.bytesize
703
+ write_raw(bytes)
704
+ end
705
+ @page_indexes.each do |chunk, _, offset_index|
706
+ bytes = offset_index.encode
707
+ chunk.offset_index_offset = @pos
708
+ chunk.offset_index_length = bytes.bytesize
709
+ write_raw(bytes)
710
+ end
711
+ @page_indexes.clear
712
+ end
713
+
714
+ # Per-column settings accepted in the +bloom_filters:+ option
715
+ BLOOM_FILTER_OPTIONS = %i[ndv fpp max_bytes].freeze
716
+
717
+ # { dotted_path => { ndv:, fpp:, max_bytes: } } from the bloom_filters: option
718
+ #
719
+ # @param requested [Boolean, Array<String>, Hash{String => Boolean, Hash, nil}, nil] true for every
720
+ # column of a supported type, column paths, or column path => true / false / settings Hash.
721
+ # A path may also be given as an Array of names. Settings are +ndv+, +fpp+ and +max_bytes+.
722
+ # @return [Hash{String => Hash{Symbol => Numeric, nil}}] settings by dotted path; empty when disabled
723
+ # @raise [ArgumentError] for an unknown column, a column type without bloom filter support, or
724
+ # invalid settings
725
+ def bloom_filter_config(requested)
726
+ return {} if requested.nil? || requested == false
727
+ columns = @schema.columns.to_h { |c| [c.dotted_path, c] }
728
+ if requested == true
729
+ return columns.select { |_, c| BloomFilter::TYPES.include?(c.type) }.transform_values { bloom_filter_settings({}) }
730
+ end
731
+ requested = Array(requested).to_h { |path| [path, true] } unless requested.is_a?(Hash)
732
+ requested.each_with_object({}) do |(path, settings), config|
733
+ path = path.is_a?(Array) ? path.join(".") : path.to_s
734
+ col = columns[path] or raise ArgumentError, "bloom_filters: no such column #{path}"
735
+ next if settings.nil? || settings == false
736
+ unless BloomFilter::TYPES.include?(col.type)
737
+ raise ArgumentError, "bloom_filters: not supported for #{T::NAMES[col.type]} column #{path}"
738
+ end
739
+ settings = {} if settings == true
740
+ raise ArgumentError, "bloom_filters: expected true or a Hash for #{path}, got #{settings.inspect}" unless settings.is_a?(Hash)
741
+ config[path] = bloom_filter_settings(settings.transform_keys(&:to_sym), path)
742
+ end
743
+ end
744
+
745
+ # @param settings [Hash{Symbol => Object}] +ndv+, +fpp+ and +max_bytes+, all optional
746
+ # @param path [String, nil] column path, for error messages
747
+ # @return [Hash{Symbol => Numeric, nil}] +ndv+ (nil to count distinct values), +fpp+ and
748
+ # +max_bytes+ with defaults filled in
749
+ # @raise [ArgumentError] for unknown keys, a non-positive ndv, or an fpp outside (0, 1)
750
+ def bloom_filter_settings(settings, path = nil)
751
+ unknown = settings.keys - BLOOM_FILTER_OPTIONS
752
+ raise ArgumentError, "bloom_filters: unknown option #{unknown.join(", ")} for #{path}" unless unknown.empty?
753
+ ndv = settings[:ndv] && Integer(settings[:ndv])
754
+ raise ArgumentError, "bloom_filters: ndv must be positive for #{path}" if ndv && !ndv.positive?
755
+ fpp = Float(settings.fetch(:fpp, BloomFilter::DEFAULT_FPP))
756
+ raise ArgumentError, "bloom_filters: fpp must be between 0 and 1 for #{path}" unless fpp > 0 && fpp < 1
757
+ {ndv: ndv, fpp: fpp, max_bytes: Integer(settings.fetch(:max_bytes, BloomFilter::DEFAULT_MAX_BYTES))}
758
+ end
759
+
760
+ # A filter holding the chunk's values: +dict_values+ (already distinct) for dictionary-encoded
761
+ # chunks, +values+ otherwise. Each distinct value is hashed once, and the filter is sized from
762
+ # the configured ndv or from the number of distinct values.
763
+ #
764
+ # @param col [Schema::Column] column the filter is for
765
+ # @param settings [Hash{Symbol => Numeric, nil}] from #bloom_filter_settings
766
+ # @param dict_values [Array, nil] dictionary of the chunk, when dictionary-encoded
767
+ # @param values [Array, nil] the chunk's non-null physical values, used without a dictionary
768
+ # @return [BloomFilter]
769
+ def build_bloom_filter(col, settings, dict_values, values)
770
+ hashes = if dict_values
771
+ BloomFilter.hash_physical_all(dict_values, col.type)
772
+ else
773
+ BloomFilter.hash_physical_all(values, col.type, distinct: true)
774
+ end
775
+ size = BloomFilter.optimal_num_bytes(settings[:ndv] || hashes.size, settings[:fpp], max_bytes: settings[:max_bytes])
776
+ filter = BloomFilter.new(size, column: col)
777
+ filter.insert_hashes(hashes)
778
+ filter
779
+ end
780
+
781
+ # Bloom filters go right after the row group's column chunks, in column order
782
+ #
783
+ # @return [void]
784
+ def write_bloom_filters
785
+ @pending_bloom_filters.each do |meta, filter|
786
+ bytes = filter.encode
787
+ meta.bloom_filter_offset = @pos
788
+ meta.bloom_filter_length = bytes.bytesize
789
+ write_raw(bytes)
790
+ end
791
+ @pending_bloom_filters.clear
792
+ end
793
+
794
+ # @param col [Schema::Column] column to check
795
+ # @return [Boolean] whether the +dictionary:+ option asks for this column to be dictionary-encoded
796
+ # (never for BOOLEAN)
469
797
  def use_dictionary?(col)
470
798
  return false if col.type == T::BOOLEAN
471
799
  case @dictionary
@@ -475,13 +803,20 @@ module Herringbone
475
803
  end
476
804
  end
477
805
 
478
- # Returns [dictionary_values, indices] or nil when a dictionary is not worthwhile
806
+ # Returns [dictionary_values, indices] or nil when a dictionary is not worthwhile: more than about
807
+ # half the values are distinct, or the dictionary exceeds MAX_DICTIONARY_BYTES
808
+ #
809
+ # @param values [Array] non-null physical values of the chunk
810
+ # @param type [Integer] physical type
811
+ # @param type_length [Integer, nil] FIXED_LEN_BYTE_ARRAY width
812
+ # @return [Array(Array, Array<Integer>), nil]
479
813
  def build_dictionary(values, type, type_length)
480
814
  # Floats are keyed by bit pattern so that -0.0 and 0.0 (and NaNs) stay distinct
481
815
  if type == T::FLOAT || type == T::DOUBLE
482
816
  keys = values.pack("G*").unpack("Q>*")
483
817
  uniq = keys.uniq
484
818
  return nil if uniq.size > values.size / 2 + 1 && values.size > 16
819
+ return nil if uniq.size * 8 > MAX_DICTIONARY_BYTES
485
820
  index = uniq.each_with_index.to_h
486
821
  return [uniq.pack("Q>*").unpack("G*"), keys.map(&index)]
487
822
  end
@@ -497,14 +832,19 @@ module Herringbone
497
832
  [dict, values.map(&index)]
498
833
  end
499
834
 
500
- # Splits a column buffer into pages of roughly @page_size bytes. Repeated columns
835
+ # Splits a column buffer into pages of roughly @page_bytes bytes. Repeated columns
501
836
  # are only cut where a new row starts.
837
+ #
838
+ # @param buf [ColumnBuffer] column buffer with levels as Arrays
839
+ # @param value_bytes [Integer] estimated encoded size of all the chunk's values
840
+ # @return [Array<Array(Integer, Integer)>] [from, to) ranges of level entries, one per page
502
841
  def page_ranges(buf, value_bytes)
503
842
  n = buf.defs.size
504
843
  bytes = n + value_bytes
505
- pages = (bytes + @page_size - 1) / @page_size
506
- return [[0, n]] if pages <= 1
507
- per = (n + pages - 1) / pages
844
+ pages = (bytes + @page_bytes - 1) / @page_bytes
845
+ per = (pages <= 1) ? n : (n + pages - 1) / pages
846
+ per = @page_rows if per > @page_rows
847
+ return [[0, n]] if per >= n
508
848
  reps = buf.reps
509
849
  ranges = []
510
850
  start = 0
@@ -518,6 +858,8 @@ module Herringbone
518
858
  ranges
519
859
  end
520
860
 
861
+ # @param col [Schema::Column] column to check
862
+ # @return [Integer, nil] PLAIN-encoded bytes per value, nil for BYTE_ARRAY (variable width)
521
863
  def value_width(col)
522
864
  case col.type
523
865
  when T::BOOLEAN then 1
@@ -528,6 +870,15 @@ module Herringbone
528
870
  end
529
871
  end
530
872
 
873
+ # Encodes a page's values. RLE_DICTIONARY values are the dictionary indices, prefixed with the
874
+ # bit width byte; RLE is only used for BOOLEAN and carries the 4-byte length prefix.
875
+ #
876
+ # @param values [Array] physical values, or dictionary indices
877
+ # @param encoding [Integer] encoding id
878
+ # @param type [Integer] physical type
879
+ # @param type_length [Integer, nil] FIXED_LEN_BYTE_ARRAY width
880
+ # @param dict_size [Integer, nil] number of dictionary entries, for RLE_DICTIONARY
881
+ # @return [String] encoded binary page values
531
882
  def encode_values(values, encoding, type, type_length, dict_size)
532
883
  case encoding
533
884
  when E::PLAIN then Encodings::Plain.encode(values, type, type_length)
@@ -538,27 +889,41 @@ module Herringbone
538
889
  body = Encodings::RLE.encode_hybrid(values.map { |v| v ? 1 : 0 }, 1)
539
890
  [body.bytesize].pack("V") << body
540
891
  when E::DELTA_BINARY_PACKED
541
- Encodings::Delta.encode_binary_packed(values, type == T::INT32 ? 32 : 64)
892
+ Encodings::Delta.encode_binary_packed(values, (type == T::INT32) ? 32 : 64)
542
893
  when E::DELTA_LENGTH_BYTE_ARRAY then Encodings::Delta.encode_length_byte_array(values)
543
894
  when E::DELTA_BYTE_ARRAY then Encodings::Delta.encode_byte_array(values)
544
895
  when E::BYTE_STREAM_SPLIT
545
- width = { T::INT32 => 4, T::FLOAT => 4, T::INT64 => 8, T::DOUBLE => 8 }[type] || type_length
896
+ width = {T::INT32 => 4, T::FLOAT => 4, T::INT64 => 8, T::DOUBLE => 8}[type] || type_length
546
897
  Encodings::ByteStreamSplit.encode(Encodings::Plain.encode(values, type, type_length), width)
547
898
  end
548
899
  end
549
900
 
550
901
  # Writes a page, returning the uncompressed size including the header
902
+ #
903
+ # @param header [Format::PageHeader] header; sizes and CRC32 are filled in here
904
+ # @param body [String] uncompressed page body
905
+ # @param compressed [String, nil] bytes to write as the page body, when already prepared (v2 data
906
+ # pages, whose levels stay uncompressed); +body+ is compressed otherwise
907
+ # @return [Integer]
551
908
  def write_page(header, body, compressed = nil)
552
909
  compressed ||= Compression.compress(@codec, body)
553
910
  header.uncompressed_page_size ||= body.bytesize
554
911
  header.compressed_page_size = compressed.bytesize
555
- header.crc = Zlib.crc32(compressed).then { |c| c >= 0x8000_0000 ? c - 0x1_0000_0000 : c }
912
+ header.crc = Zlib.crc32(compressed).then { |c| (c >= 0x8000_0000) ? c - 0x1_0000_0000 : c }
556
913
  encoded = header.encode
557
914
  write_raw(encoded)
558
915
  write_raw(compressed)
559
916
  encoded.bytesize + header.uncompressed_page_size
560
917
  end
561
918
 
919
+ # A DATA_PAGE: length-prefixed repetition and definition levels, then the values, all compressed
920
+ #
921
+ # @param n [Integer] number of level entries (values including nulls)
922
+ # @param rep_bytes [String] RLE-encoded repetition levels, empty when the column has none
923
+ # @param def_bytes [String] RLE-encoded definition levels, empty when the column has none
924
+ # @param encoded [String] encoded values
925
+ # @param encoding [Integer] encoding id of the values
926
+ # @return [Integer] uncompressed size including the header
562
927
  def write_data_page_v1(n, rep_bytes, def_bytes, encoded, encoding)
563
928
  body = String.new(encoding: Encoding::BINARY)
564
929
  body << [rep_bytes.bytesize].pack("V") << rep_bytes unless rep_bytes.empty?
@@ -574,6 +939,16 @@ module Herringbone
574
939
  write_page(header, body)
575
940
  end
576
941
 
942
+ # A DATA_PAGE_V2: levels without length prefixes and uncompressed, then the compressed values
943
+ #
944
+ # @param n [Integer] number of level entries (values including nulls)
945
+ # @param nulls [Integer] null entries
946
+ # @param rows [Integer] rows in the page
947
+ # @param rep_bytes [String] RLE-encoded repetition levels, empty when the column has none
948
+ # @param def_bytes [String] RLE-encoded definition levels, empty when the column has none
949
+ # @param encoded [String] encoded values
950
+ # @param encoding [Integer] encoding id of the values
951
+ # @return [Integer] uncompressed size including the header
577
952
  def write_data_page_v2(n, nulls, rows, rep_bytes, def_bytes, encoded, encoding)
578
953
  compressed = Compression.compress(@codec, encoded)
579
954
  header = Format::PageHeader.new(
@@ -589,46 +964,116 @@ module Herringbone
589
964
  write_page(header, "".b, rep_bytes + def_bytes + compressed)
590
965
  end
591
966
 
592
- def statistics_for(col, defs, values)
967
+ # Statistics and column index bounds longer than this are truncated, see #truncate_min
968
+ STAT_TRUNCATE_BYTES = 64
969
+ # Sort key for types whose Ruby ordering already matches Parquet's; compared by identity to
970
+ # skip calling it
971
+ IDENTITY = ->(v) { v }
972
+
973
+ # Chunk statistics: null count, plus min/max (flagged exact unless truncated) when the column
974
+ # has a sort order and non-NaN values
975
+ #
976
+ # @param col [Schema::Column] column the chunk belongs to
977
+ # @param defs [Array<Integer>] definition levels of the chunk
978
+ # @param values [Array] distinct or all non-null physical values of the chunk
979
+ # @param order [Proc, Method, nil] sort key from #sort_key
980
+ # @return [Format::Statistics]
981
+ def statistics_for(col, defs, values, order)
593
982
  max_def = col.max_definition_level
594
983
  nulls = max_def.zero? ? 0 : defs.size - defs.count(max_def)
595
984
  stats = Format::Statistics.new(null_count: nulls)
596
- min, max = min_max(col, values)
597
- if min
598
- stats.min_value = min
599
- stats.max_value = max
600
- stats.is_min_value_exact = true
601
- stats.is_max_value_exact = true
985
+ range = order && value_range(col, values, order)
986
+ if range
987
+ min = stat_bytes(col, range[0])
988
+ max = stat_bytes(col, range[1])
989
+ stats.min_value = truncate_min(min)
990
+ stats.max_value = truncate_max(max)
991
+ stats.is_min_value_exact = stats.min_value.bytesize == min.bytesize
992
+ stats.is_max_value_exact = stats.max_value.bytesize == max.bytesize
602
993
  end
603
994
  stats
604
995
  end
605
996
 
606
- # Min/max for types whose Parquet sort order matches Ruby's comparison of the physical values
607
- def min_max(col, values)
608
- return nil if values.empty?
997
+ # A key giving the Parquet sort order of the column's physical values (IDENTITY when Ruby's own
998
+ # comparison already matches), or nil when the order is undefined (INT96)
999
+ #
1000
+ # @param col [Schema::Column] column to order
1001
+ # @return [Proc, Method, nil]
1002
+ def sort_key(col)
609
1003
  kind, _, signed = Types.logical_of(col.node)
610
1004
  case col.type
611
- when T::BOOLEAN
612
- [values.include?(false) ? "\x00".b : "\x01".b, values.include?(true) ? "\x01".b : "\x00".b]
1005
+ when T::BOOLEAN then ->(v) { v ? 1 : 0 }
613
1006
  when T::INT32, T::INT64
614
- return nil if kind == :integer && !signed
615
- fmt = col.type == T::INT32 ? "l<" : "q<"
616
- min, max = values.minmax
617
- [[min].pack(fmt), [max].pack(fmt)]
618
- when T::FLOAT, T::DOUBLE
619
- finite = values.reject(&:nan?)
620
- return nil if finite.empty?
621
- min, max = finite.minmax
1007
+ return IDENTITY unless kind == :integer && !signed
1008
+ mask = (col.type == T::INT32) ? 0xFFFF_FFFF : 0xFFFF_FFFF_FFFF_FFFF
1009
+ ->(v) { v & mask }
1010
+ when T::FLOAT, T::DOUBLE then IDENTITY
1011
+ when T::BYTE_ARRAY, T::FIXED_LEN_BYTE_ARRAY
1012
+ case kind
1013
+ when :decimal then Types.method(:be_to_int)
1014
+ when :float16 then ->(v) { Types.half_to_float(v.unpack1("v")) }
1015
+ else IDENTITY # unsigned lexicographic, which is how Ruby compares binary Strings
1016
+ end
1017
+ end
1018
+ end
1019
+
1020
+ # [min, max] of +values+ in column order, ignoring NaNs; nil if there is nothing to compare.
1021
+ # A zero float bound is normalized to -0.0 (min) / 0.0 (max), as the spec asks.
1022
+ #
1023
+ # @param col [Schema::Column] column the values belong to
1024
+ # @param values [Array] physical values
1025
+ # @param order [Proc, Method] sort key from #sort_key
1026
+ # @return [Array(Object, Object), nil]
1027
+ def value_range(col, values, order)
1028
+ floats = col.type == T::FLOAT || col.type == T::DOUBLE
1029
+ if floats
1030
+ values = values.reject(&:nan?)
1031
+ elsif Types.logical_of(col.node).first == :float16
1032
+ values = values.reject { |v| order.call(v).nan? }
1033
+ end
1034
+ return nil if values.empty?
1035
+ min, max = order.equal?(IDENTITY) ? values.minmax : values.minmax_by(&order)
1036
+ if floats
622
1037
  min = -0.0 if min.zero?
623
1038
  max = 0.0 if max.zero?
624
- fmt = col.type == T::FLOAT ? "e" : "E"
625
- [[min].pack(fmt), [max].pack(fmt)]
626
- when T::BYTE_ARRAY
627
- return nil if kind == :decimal
628
- min, max = values.minmax
629
- return nil if min.bytesize > 1024 || max.bytesize > 1024
630
- [min, max]
631
1039
  end
1040
+ [min, max]
1041
+ end
1042
+
1043
+ # @param col [Schema::Column] column the value belongs to
1044
+ # @param value [Object] physical value
1045
+ # @return [String] the value PLAIN-encoded, as statistics store it (byte arrays without length)
1046
+ def stat_bytes(col, value)
1047
+ case col.type
1048
+ when T::BOOLEAN then value ? "\x01".b : "\x00".b
1049
+ when T::INT32 then [value].pack("l<")
1050
+ when T::INT64 then [value].pack("q<")
1051
+ when T::FLOAT then [value].pack("e")
1052
+ when T::DOUBLE then [value].pack("E")
1053
+ else value.b
1054
+ end
1055
+ end
1056
+
1057
+ # Long byte-array bounds are truncated: a prefix is still a lower bound for the minimum, and
1058
+ # a prefix with its last byte incremented is an upper bound for the maximum
1059
+ #
1060
+ # @param bytes [String] encoded minimum
1061
+ # @return [String] at most STAT_TRUNCATE_BYTES bytes
1062
+ def truncate_min(bytes)
1063
+ (bytes.bytesize > STAT_TRUNCATE_BYTES) ? bytes.byteslice(0, STAT_TRUNCATE_BYTES) : bytes
1064
+ end
1065
+
1066
+ # @param bytes [String] encoded maximum
1067
+ # @return [String] at most STAT_TRUNCATE_BYTES bytes, with the last byte that is not 0xFF
1068
+ # incremented (trailing 0xFF bytes dropped); +bytes+ unchanged when it fits, or when the
1069
+ # whole prefix is 0xFF
1070
+ def truncate_max(bytes)
1071
+ return bytes if bytes.bytesize <= STAT_TRUNCATE_BYTES
1072
+ prefix = bytes.byteslice(0, STAT_TRUNCATE_BYTES).bytes
1073
+ prefix.pop while prefix.last == 0xFF
1074
+ return bytes if prefix.empty?
1075
+ prefix[-1] += 1
1076
+ prefix.pack("C*")
632
1077
  end
633
1078
  end
634
1079
  end