herringbone 0.2.0 → 0.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/README.md +56 -4
- data/lib/herringbone/active_record.rb +44 -12
- data/lib/herringbone/bloom_filter.rb +112 -12
- data/lib/herringbone/byte_values.rb +29 -0
- data/lib/herringbone/codecs/lz4.rb +66 -3
- data/lib/herringbone/codecs/snappy.rb +132 -3
- data/lib/herringbone/compression.rb +48 -0
- data/lib/herringbone/encodings/delta.rb +70 -13
- data/lib/herringbone/encodings/plain.rb +22 -0
- data/lib/herringbone/encodings/rle.rb +53 -1
- data/lib/herringbone/format.rb +150 -0
- data/lib/herringbone/inspector.rb +477 -81
- data/lib/herringbone/io_buffer_support.rb +3 -1
- data/lib/herringbone/reader/column_chunk_reader.rb +98 -6
- data/lib/herringbone/reader/column_cursor.rb +46 -2
- data/lib/herringbone/reader/numo.rb +365 -0
- data/lib/herringbone/reader/page_stream.rb +280 -4
- data/lib/herringbone/reader/scan.rb +103 -9
- data/lib/herringbone/reader.rb +308 -17
- data/lib/herringbone/schema.rb +306 -15
- data/lib/herringbone/thrift.rb +155 -4
- data/lib/herringbone/types.rb +176 -41
- data/lib/herringbone/version.rb +2 -1
- data/lib/herringbone/visualizer.rb +51 -11
- data/lib/herringbone/writer.rb +308 -31
- data/lib/herringbone/xxhash.rb +108 -2
- data/lib/herringbone.rb +28 -0
- metadata +3 -2
data/lib/herringbone/writer.rb
CHANGED
|
@@ -17,8 +17,9 @@ module Herringbone
|
|
|
17
17
|
# end
|
|
18
18
|
#
|
|
19
19
|
# The output is any IO that responds to #write (a File, StringIO, Tempfile, socket, pipe...); it
|
|
20
|
-
# is written sequentially and never seeked, rewound or closed by the writer
|
|
21
|
-
#
|
|
20
|
+
# is written sequentially and never seeked, rewound or closed by the writer; only #write is
|
|
21
|
+
# required, its return value is ignored, and #binmode and #flush are called when available.
|
|
22
|
+
# Herringbone does not open files by path.
|
|
22
23
|
#
|
|
23
24
|
# Options:
|
|
24
25
|
# compression: :snappy (default), :zstd, :gzip, :lz4 (LZ4_RAW), :lz4_hadoop, :brotli, :none
|
|
@@ -40,16 +41,21 @@ module Herringbone
|
|
|
40
41
|
#
|
|
41
42
|
# Statistics and page indexes (ColumnIndex/OffsetIndex) are always written.
|
|
42
43
|
class Writer
|
|
43
|
-
|
|
44
|
+
# Marker at the start and end of every Parquet file
|
|
45
|
+
MAGIC = "PAR1".b.freeze
|
|
46
|
+
# Shorthand for Format::Type (physical types)
|
|
44
47
|
T = Format::Type
|
|
48
|
+
# Shorthand for Format::Encoding
|
|
45
49
|
E = Format::Encoding
|
|
46
50
|
|
|
51
|
+
# Names accepted in the +encodings:+ option => Parquet encoding id
|
|
47
52
|
ENCODING_NAMES = {
|
|
48
53
|
plain: E::PLAIN, rle: E::RLE, delta_binary_packed: E::DELTA_BINARY_PACKED,
|
|
49
54
|
delta_length_byte_array: E::DELTA_LENGTH_BYTE_ARRAY, delta_byte_array: E::DELTA_BYTE_ARRAY,
|
|
50
55
|
byte_stream_split: E::BYTE_STREAM_SPLIT
|
|
51
56
|
}.freeze
|
|
52
57
|
|
|
58
|
+
# Encoding id => physical types it may be used for, per the Parquet encodings spec
|
|
53
59
|
VALID_ENCODINGS = {
|
|
54
60
|
E::PLAIN => T::NAMES.keys,
|
|
55
61
|
E::RLE => [T::BOOLEAN],
|
|
@@ -59,15 +65,41 @@ module Herringbone
|
|
|
59
65
|
E::BYTE_STREAM_SPLIT => [T::INT32, T::INT64, T::FLOAT, T::DOUBLE, T::FIXED_LEN_BYTE_ARRAY]
|
|
60
66
|
}.freeze
|
|
61
67
|
|
|
68
|
+
# Largest dictionary build_dictionary keeps; above it the chunk is written without a dictionary.
|
|
69
|
+
# Byte-array columns are dictionary-encoded by ByteValues, which applies its own
|
|
70
|
+
# ByteValues::MAX_DICTIONARY_BYTES as values arrive.
|
|
62
71
|
MAX_DICTIONARY_BYTES = 1024 * 1024
|
|
63
72
|
# The row group byte size is first estimated after this many rows, then after every row group
|
|
64
73
|
ESTIMATE_AFTER_ROWS = 1000
|
|
65
74
|
|
|
75
|
+
# @return [Schema] schema the rows are written with
|
|
66
76
|
attr_reader :schema
|
|
67
77
|
|
|
68
78
|
# Opens a writer on +io+. With a block, the file is finished (footer written) when the block
|
|
69
79
|
# returns and the block's value is returned; if the block raises, the writer is aborted and no
|
|
70
80
|
# footer is written. Without a block, call #close to finish. The IO is never closed.
|
|
81
|
+
#
|
|
82
|
+
# @param io [IO, #write] destination
|
|
83
|
+
# @param schema [Schema] schema of the rows
|
|
84
|
+
# @param options [Hash{Symbol => Object}] see the class description and #initialize
|
|
85
|
+
# @option options [Symbol] :compression (:snappy) codec, see Herringbone.codecs
|
|
86
|
+
# @option options [Integer] :row_group_bytes (16MB) approximate buffered size that triggers a row group
|
|
87
|
+
# @option options [Integer, nil] :row_group_rows (nil) also flush a row group after this many rows
|
|
88
|
+
# @option options [Integer] :page_bytes (1MB) approximate uncompressed data page size
|
|
89
|
+
# @option options [Integer] :page_rows (20_000) maximum rows per data page
|
|
90
|
+
# @option options [Integer] :data_page_version (1) 1 or 2
|
|
91
|
+
# @option options [Boolean, Array<String>] :dictionary (true) dictionary-encode all eligible columns,
|
|
92
|
+
# none, or only the listed dotted column paths
|
|
93
|
+
# @option options [Hash{String => Symbol}] :encodings ({}) dotted column path => value encoding
|
|
94
|
+
# for non-dictionary pages
|
|
95
|
+
# @option options [Hash{String => String}] :metadata ({}) key/value metadata for the footer
|
|
96
|
+
# @option options [Boolean, Array<String>, Hash{String => Boolean, Hash}] :bloom_filters (nil)
|
|
97
|
+
# columns to write split block bloom filters for
|
|
98
|
+
# @yield [writer] the open writer
|
|
99
|
+
# @yieldparam writer [Writer] writer to append rows to
|
|
100
|
+
# @yieldreturn [Object] returned by open
|
|
101
|
+
# @return [Writer, Object] the writer without a block, the block's value with one
|
|
102
|
+
# @raise [ArgumentError] for an invalid IO, schema or option
|
|
71
103
|
def self.open(io, schema, **options)
|
|
72
104
|
writer = new(io, schema, **options)
|
|
73
105
|
return writer unless block_given?
|
|
@@ -81,6 +113,27 @@ module Herringbone
|
|
|
81
113
|
result
|
|
82
114
|
end
|
|
83
115
|
|
|
116
|
+
# Validates the options and writes the leading magic bytes to +io+.
|
|
117
|
+
#
|
|
118
|
+
# @param io [IO, #write] destination; switched to binmode when it supports that
|
|
119
|
+
# @param schema [Schema] schema of the rows
|
|
120
|
+
# @param compression [Symbol, Integer] codec name (see Herringbone.codecs) or Format::Codec id
|
|
121
|
+
# @param row_group_bytes [Integer] approximate buffered size that triggers a row group
|
|
122
|
+
# @param row_group_rows [Integer, nil] also flush a row group after this many rows
|
|
123
|
+
# @param page_bytes [Integer] approximate uncompressed data page size
|
|
124
|
+
# @param page_rows [Integer] maximum level entries per data page (pages of repeated columns
|
|
125
|
+
# extend to the next row start)
|
|
126
|
+
# @param data_page_version [Integer] 1 or 2
|
|
127
|
+
# @param dictionary [Boolean, Array<String>] true for every column except BOOLEAN, FLOAT, DOUBLE
|
|
128
|
+
# and those listed in +encodings+; false for none; or the dotted paths of the columns to encode
|
|
129
|
+
# @param encodings [Hash{String, Symbol => Symbol, Integer}] dotted column path => value encoding
|
|
130
|
+
# (a name from ENCODING_NAMES or an encoding id) for non-dictionary pages
|
|
131
|
+
# @param metadata [Hash{#to_s => #to_s}] key/value metadata for the footer
|
|
132
|
+
# @param bloom_filters [Boolean, Array<String>, Hash{String => Boolean, Hash}, nil] see the class
|
|
133
|
+
# description
|
|
134
|
+
# @raise [ArgumentError] for an invalid IO, schema or option
|
|
135
|
+
# @raise [MissingCodecError] when the codec's optional gem cannot be loaded
|
|
136
|
+
# @raise [UnsupportedError] when the codec is not supported
|
|
84
137
|
def initialize(io, schema, compression: :snappy, row_group_bytes: 16 * 1024 * 1024, row_group_rows: nil,
|
|
85
138
|
page_bytes: 1024 * 1024, page_rows: 20_000, data_page_version: 1, dictionary: true, encodings: {},
|
|
86
139
|
metadata: {}, bloom_filters: nil)
|
|
@@ -100,8 +153,10 @@ module Herringbone
|
|
|
100
153
|
raise ArgumentError, "data_page_version must be 1 or 2" unless [1, 2].include?(@data_page_version)
|
|
101
154
|
@dictionary = dictionary
|
|
102
155
|
@encodings = encodings.to_h { |path, enc| [path.to_s, encoding_id(path, enc)] }
|
|
103
|
-
|
|
156
|
+
columns = schema.columns.to_h { |c| [c.dotted_path, c] }
|
|
157
|
+
unknown = @encodings.keys - columns.keys
|
|
104
158
|
raise ArgumentError, "encodings: no such column #{unknown.join(", ")}" unless unknown.empty?
|
|
159
|
+
@encodings.each { |path, enc| check_encoding!(columns[path], enc) }
|
|
105
160
|
@metadata = metadata
|
|
106
161
|
@bloom_filters = bloom_filter_config(bloom_filters)
|
|
107
162
|
@pending_bloom_filters = [] # [ColumnMetaData, BloomFilter] for the row group being written
|
|
@@ -116,7 +171,14 @@ module Herringbone
|
|
|
116
171
|
end
|
|
117
172
|
|
|
118
173
|
# Appends a row: a Hash keyed by top-level field names (Strings or Symbols), an Array of values
|
|
119
|
-
# in schema order, or an object responding to #attributes (ActiveRecord) or #to_h (Struct, Data)
|
|
174
|
+
# in schema order, or an object responding to #attributes (ActiveRecord) or #to_h (Struct, Data).
|
|
175
|
+
# A row that fails to encode leaves nothing behind in the buffers. Flushes a row group when the
|
|
176
|
+
# buffered rows reach the size limits.
|
|
177
|
+
#
|
|
178
|
+
# @param row [Hash, Array, #attributes, #to_h] row to append
|
|
179
|
+
# @return [self]
|
|
180
|
+
# @raise [Error] when the writer is closed
|
|
181
|
+
# @raise [EncodeError] when a required field is nil or missing, or a value cannot be encoded
|
|
120
182
|
def <<(row)
|
|
121
183
|
raise Error, "Writer is closed" if @closed
|
|
122
184
|
row = row_hash(row)
|
|
@@ -141,7 +203,7 @@ module Herringbone
|
|
|
141
203
|
rescue EncodeError => e
|
|
142
204
|
rollback_row(marks)
|
|
143
205
|
raise EncodeError, "Row #{@total_rows + @buffered_rows}: #{e.message}"
|
|
144
|
-
rescue
|
|
206
|
+
rescue
|
|
145
207
|
rollback_row(marks)
|
|
146
208
|
raise
|
|
147
209
|
end
|
|
@@ -151,9 +213,13 @@ module Herringbone
|
|
|
151
213
|
end
|
|
152
214
|
|
|
153
215
|
# Number of rows written so far, including buffered ones
|
|
216
|
+
#
|
|
217
|
+
# @return [Integer]
|
|
154
218
|
def rows_written = @total_rows + @buffered_rows
|
|
155
219
|
|
|
156
220
|
# Writes any buffered rows as a row group
|
|
221
|
+
#
|
|
222
|
+
# @return [void]
|
|
157
223
|
def flush_row_group
|
|
158
224
|
return if @buffered_rows.zero?
|
|
159
225
|
start = @pos
|
|
@@ -171,6 +237,10 @@ module Herringbone
|
|
|
171
237
|
reset_buffers
|
|
172
238
|
end
|
|
173
239
|
|
|
240
|
+
# Flushes buffered rows, then writes the page indexes and the footer and flushes the IO.
|
|
241
|
+
# Does nothing when already closed or aborted. The IO itself is not closed.
|
|
242
|
+
#
|
|
243
|
+
# @return [void]
|
|
174
244
|
def close
|
|
175
245
|
return if @closed
|
|
176
246
|
flush_row_group
|
|
@@ -194,23 +264,47 @@ module Herringbone
|
|
|
194
264
|
|
|
195
265
|
# Stops writing without finishing the file (no footer is written). Whatever was already written
|
|
196
266
|
# to the IO stays there; discarding it is up to the caller.
|
|
267
|
+
#
|
|
268
|
+
# @return [void]
|
|
197
269
|
def abort
|
|
198
270
|
@closed = true
|
|
199
271
|
end
|
|
200
272
|
|
|
201
273
|
private
|
|
202
274
|
|
|
203
|
-
# Pathname responds to #write too (writing a whole file by path), so it is rejected explicitly
|
|
275
|
+
# Pathname responds to #write too (writing a whole file by path), so it is rejected explicitly.
|
|
276
|
+
# A text-mode IO (a pipe, or a File opened with "w") transcodes what is written when
|
|
277
|
+
# Encoding.default_internal is set, as Rails does, and binary pages cannot be transcoded,
|
|
278
|
+
# so the IO is switched to binary mode.
|
|
279
|
+
#
|
|
280
|
+
# @param io [Object] candidate destination
|
|
281
|
+
# @return [IO, #write] +io+, in binary mode when it supports #binmode
|
|
282
|
+
# @raise [ArgumentError] when +io+ does not respond to #write, or is a String or Pathname
|
|
204
283
|
def check_io!(io)
|
|
205
|
-
|
|
284
|
+
if io.respond_to?(:write) && !io.is_a?(String) && !(defined?(Pathname) && io.is_a?(Pathname))
|
|
285
|
+
io.binmode if io.respond_to?(:binmode)
|
|
286
|
+
return io
|
|
287
|
+
end
|
|
206
288
|
raise ArgumentError, "Herringbone::Writer expects an IO that responds to #write " \
|
|
207
289
|
"(e.g. File.open(path, \"wb\") or StringIO.new), got #{io.class}"
|
|
208
290
|
end
|
|
209
291
|
|
|
210
292
|
# Levels are kept as binary Strings (one byte per entry). Values are an Array for numeric and
|
|
211
293
|
# boolean columns, and a compact ByteValues for BYTE_ARRAY / FIXED_LEN_BYTE_ARRAY columns.
|
|
294
|
+
#
|
|
295
|
+
# @!attribute defs
|
|
296
|
+
# @return [String, Array<Integer>] definition levels (unpacked to an Array while a chunk is written)
|
|
297
|
+
# @!attribute reps
|
|
298
|
+
# @return [String, Array<Integer>, nil] repetition levels, like +defs+; nil for non-repeated columns
|
|
299
|
+
# @!attribute values
|
|
300
|
+
# @return [Array, ByteValues] non-null physical values
|
|
212
301
|
ColumnBuffer = Struct.new(:defs, :reps, :values)
|
|
213
302
|
|
|
303
|
+
# Starts a new row group: fresh buffers per column, and the per-field write plan
|
|
304
|
+
# (field, name, Symbol name, buffer and encoder for flat fields, or nils for shredded ones).
|
|
305
|
+
#
|
|
306
|
+
# @return [void]
|
|
307
|
+
# @raise [UnsupportedError] when a column has more than 255 definition or repetition levels
|
|
214
308
|
def reset_buffers
|
|
215
309
|
@buffers = @schema.columns.map do |col|
|
|
216
310
|
if col.max_definition_level > 255 || col.max_repetition_level > 255
|
|
@@ -229,6 +323,8 @@ module Herringbone
|
|
|
229
323
|
@buffered_rows = 0
|
|
230
324
|
end
|
|
231
325
|
|
|
326
|
+
# @param col [Schema::Column] column to buffer
|
|
327
|
+
# @return [ByteValues, Array] empty value store for the column
|
|
232
328
|
def new_values_store(col)
|
|
233
329
|
case col.type
|
|
234
330
|
when T::BYTE_ARRAY then ByteValues.new(dictionary: use_dictionary?(col))
|
|
@@ -239,6 +335,8 @@ module Herringbone
|
|
|
239
335
|
|
|
240
336
|
# Flushes when the buffered values reach row_group_bytes (or row_group_rows rows). The bytes per
|
|
241
337
|
# row are estimated from the buffered values after the first rows, then refreshed per row group.
|
|
338
|
+
#
|
|
339
|
+
# @return [void]
|
|
242
340
|
def check_row_group_size
|
|
243
341
|
@bytes_per_row ||= estimate_bytes_per_row
|
|
244
342
|
limit = row_limit_for(@bytes_per_row)
|
|
@@ -251,11 +349,15 @@ module Herringbone
|
|
|
251
349
|
end
|
|
252
350
|
end
|
|
253
351
|
|
|
352
|
+
# @param bytes_per_row [Integer] estimated buffered bytes per row
|
|
353
|
+
# @return [Integer] rows to buffer before the next size check: what fits in row_group_bytes,
|
|
354
|
+
# at least ESTIMATE_AFTER_ROWS, at most row_group_rows
|
|
254
355
|
def row_limit_for(bytes_per_row)
|
|
255
356
|
by_bytes = [@row_group_bytes / [bytes_per_row, 1].max, ESTIMATE_AFTER_ROWS].max
|
|
256
357
|
@row_group_rows ? [@row_group_rows, by_bytes].min : by_bytes
|
|
257
358
|
end
|
|
258
359
|
|
|
360
|
+
# @return [Integer] memory held by the buffers divided by the buffered rows
|
|
259
361
|
def estimate_bytes_per_row
|
|
260
362
|
bytes = @schema.columns.sum do |col|
|
|
261
363
|
buf = @buffers[col.index]
|
|
@@ -263,29 +365,58 @@ module Herringbone
|
|
|
263
365
|
held = if values.is_a?(ByteValues)
|
|
264
366
|
values.memory_bytes
|
|
265
367
|
else
|
|
266
|
-
values.size * (col.type == T::INT96 ? 48 : 8)
|
|
368
|
+
values.size * ((col.type == T::INT96) ? 48 : 8)
|
|
267
369
|
end
|
|
268
370
|
held + buf.defs.bytesize + (buf.reps ? buf.reps.bytesize : 0)
|
|
269
371
|
end
|
|
270
372
|
bytes / [@buffered_rows, 1].max
|
|
271
373
|
end
|
|
272
374
|
|
|
375
|
+
# Writes to the IO and tracks the file offset, since the IO is never asked for its position
|
|
376
|
+
#
|
|
377
|
+
# @param bytes [String] binary data
|
|
378
|
+
# @return [void]
|
|
273
379
|
def write_raw(bytes)
|
|
274
380
|
@io.write(bytes)
|
|
275
381
|
@pos += bytes.bytesize
|
|
276
382
|
end
|
|
277
383
|
|
|
384
|
+
# @param path [String, Symbol] column path, for the error message
|
|
385
|
+
# @param enc [Symbol, String, Integer] encoding name from ENCODING_NAMES, or an encoding id
|
|
386
|
+
# @return [Integer] encoding id
|
|
387
|
+
# @raise [ArgumentError] for an unknown encoding name, or an id that is not in VALID_ENCODINGS
|
|
278
388
|
def encoding_id(path, enc)
|
|
279
|
-
|
|
389
|
+
if enc.is_a?(Integer)
|
|
390
|
+
return enc if VALID_ENCODINGS.key?(enc)
|
|
391
|
+
raise ArgumentError, "Encoding #{E::NAMES.fetch(enc, enc)} cannot be requested for #{path}"
|
|
392
|
+
end
|
|
280
393
|
ENCODING_NAMES.fetch(enc.to_s.downcase.to_sym) { raise ArgumentError, "Unknown encoding #{enc.inspect} for #{path}" }
|
|
281
394
|
end
|
|
282
395
|
|
|
396
|
+
# @param col [Schema::Column] column the encoding is requested for
|
|
397
|
+
# @param enc [Integer] encoding id, a key of VALID_ENCODINGS
|
|
398
|
+
# @return [void]
|
|
399
|
+
# @raise [ArgumentError] when the encoding is not valid for the column's physical type
|
|
400
|
+
def check_encoding!(col, enc)
|
|
401
|
+
return if VALID_ENCODINGS.fetch(enc).include?(col.type)
|
|
402
|
+
raise ArgumentError, "Encoding #{E::NAMES[enc]} is not valid for #{T::NAMES[col.type]} column #{col.dotted_path}"
|
|
403
|
+
end
|
|
404
|
+
|
|
405
|
+
# Reads a struct member, by String or Symbol key
|
|
406
|
+
#
|
|
407
|
+
# @param hash [Hash, #to_h, nil] struct value
|
|
408
|
+
# @param name [String] member name
|
|
409
|
+
# @return [Object, nil] member value, nil when +hash+ is nil or has no such key
|
|
410
|
+
# @raise [EncodeError] when +hash+ cannot be converted to a Hash
|
|
283
411
|
def lookup(hash, name)
|
|
284
412
|
return nil if hash.nil?
|
|
285
413
|
hash = as_hash(hash, name) unless hash.is_a?(Hash)
|
|
286
414
|
hash.fetch(name) { hash[name.to_sym] }
|
|
287
415
|
end
|
|
288
416
|
|
|
417
|
+
# @param row [Hash, Array, #attributes, #to_h] row given to #<<
|
|
418
|
+
# @return [Hash] the row keyed by top-level field names
|
|
419
|
+
# @raise [EncodeError] when an Array has the wrong size or the row cannot be converted to a Hash
|
|
289
420
|
def row_hash(row)
|
|
290
421
|
case row
|
|
291
422
|
when Hash then row
|
|
@@ -300,6 +431,10 @@ module Herringbone
|
|
|
300
431
|
end
|
|
301
432
|
end
|
|
302
433
|
|
|
434
|
+
# @param value [Hash, #to_h] value to convert
|
|
435
|
+
# @param what [String] description of the value, for the error message
|
|
436
|
+
# @return [Hash]
|
|
437
|
+
# @raise [EncodeError] when +value+ does not respond to #to_h
|
|
303
438
|
def as_hash(value, what)
|
|
304
439
|
return value if value.is_a?(Hash)
|
|
305
440
|
raise EncodeError, "Expected a Hash for #{what}, got #{value.class}" unless value.respond_to?(:to_h)
|
|
@@ -307,6 +442,10 @@ module Herringbone
|
|
|
307
442
|
end
|
|
308
443
|
|
|
309
444
|
# Removes the entries a failed row left behind, so the buffers stay aligned
|
|
445
|
+
#
|
|
446
|
+
# @param marks [Array<Array(Integer, Integer, Integer)>, nil] sizes of defs, reps and values of
|
|
447
|
+
# each nested buffer before the row, nil when there are none
|
|
448
|
+
# @return [void]
|
|
310
449
|
def rollback_row(marks)
|
|
311
450
|
@plan.each do |field, _, _, buf|
|
|
312
451
|
next unless buf && buf.defs.bytesize > @buffered_rows
|
|
@@ -323,6 +462,14 @@ module Herringbone
|
|
|
323
462
|
end
|
|
324
463
|
|
|
325
464
|
# Record shredding: turns a nested value into (definition level, repetition level, value) entries
|
|
465
|
+
#
|
|
466
|
+
# @param field [Schema::Field] field the value belongs to
|
|
467
|
+
# @param value [Object, nil] Ruby value of the field
|
|
468
|
+
# @param parent_def [Integer] definition level recorded when +value+ is nil
|
|
469
|
+
# @param rep [Integer] repetition level of the first entry this value produces
|
|
470
|
+
# @return [void]
|
|
471
|
+
# @raise [EncodeError] when a required value is nil, a list or map value has the wrong type, a
|
|
472
|
+
# map key is nil, or a leaf value cannot be encoded
|
|
326
473
|
def shred(field, value, parent_def, rep)
|
|
327
474
|
if value.nil?
|
|
328
475
|
raise EncodeError, "Field #{field.node.path.join(".")} is required but got nil" unless field.optional
|
|
@@ -379,6 +526,13 @@ module Herringbone
|
|
|
379
526
|
end
|
|
380
527
|
end
|
|
381
528
|
|
|
529
|
+
# Writes one column chunk of the current row group: an optional dictionary page, then data pages.
|
|
530
|
+
# Bloom filters and page indexes are queued, to be written after the row group and before the
|
|
531
|
+
# footer.
|
|
532
|
+
#
|
|
533
|
+
# @param col [Schema::Column] column being written
|
|
534
|
+
# @param buffer [ColumnBuffer] the column's buffered levels and values
|
|
535
|
+
# @return [Format::ColumnChunk] chunk with its ColumnMetaData, for the row group
|
|
382
536
|
def write_column_chunk(col, buffer)
|
|
383
537
|
type = col.type
|
|
384
538
|
path = col.dotted_path
|
|
@@ -399,11 +553,7 @@ module Herringbone
|
|
|
399
553
|
value_encoding = if dict_values
|
|
400
554
|
E::RLE_DICTIONARY
|
|
401
555
|
else
|
|
402
|
-
|
|
403
|
-
unless VALID_ENCODINGS.fetch(enc).include?(type)
|
|
404
|
-
raise ArgumentError, "Encoding #{E::NAMES[enc]} is not valid for #{T::NAMES[type]} column #{path}"
|
|
405
|
-
end
|
|
406
|
-
enc
|
|
556
|
+
@encodings[path] || E::PLAIN
|
|
407
557
|
end
|
|
408
558
|
|
|
409
559
|
chunk_start = @pos
|
|
@@ -482,8 +632,24 @@ module Herringbone
|
|
|
482
632
|
chunk
|
|
483
633
|
end
|
|
484
634
|
|
|
635
|
+
# What the page indexes need to know about a written data page
|
|
636
|
+
#
|
|
637
|
+
# @!attribute offset
|
|
638
|
+
# @return [Integer] file offset of the page header
|
|
639
|
+
# @!attribute size
|
|
640
|
+
# @return [Integer] page size in the file, header included
|
|
641
|
+
# @!attribute first_row
|
|
642
|
+
# @return [Integer] index of the page's first row within the row group
|
|
643
|
+
# @!attribute nulls
|
|
644
|
+
# @return [Integer] null entries in the page
|
|
645
|
+
# @!attribute non_null
|
|
646
|
+
# @return [Integer] non-null values in the page
|
|
647
|
+
# @!attribute range
|
|
648
|
+
# @return [Array(Object, Object), nil] [min, max] physical values, nil when there are none
|
|
485
649
|
PageInfo = Struct.new(:offset, :size, :first_row, :nulls, :non_null, :range)
|
|
486
650
|
|
|
651
|
+
# @param pages [Array<PageInfo>] pages of a column chunk
|
|
652
|
+
# @return [Format::OffsetIndex]
|
|
487
653
|
def offset_index_for(pages)
|
|
488
654
|
Format::OffsetIndex.new(page_locations: pages.map do |p|
|
|
489
655
|
Format::PageLocation.new(offset: p.offset, compressed_page_size: p.size, first_row_index: p.first_row)
|
|
@@ -491,6 +657,11 @@ module Herringbone
|
|
|
491
657
|
end
|
|
492
658
|
|
|
493
659
|
# nil when the column has no defined sort order, or a page's values have no min/max (all NaN)
|
|
660
|
+
#
|
|
661
|
+
# @param col [Schema::Column] column the pages belong to
|
|
662
|
+
# @param pages [Array<PageInfo>] pages of the column chunk
|
|
663
|
+
# @param order [Proc, Method, nil] sort key from #sort_key
|
|
664
|
+
# @return [Format::ColumnIndex, nil]
|
|
494
665
|
def column_index_for(col, pages, order)
|
|
495
666
|
return nil unless order
|
|
496
667
|
return nil if pages.any? { |p| p.non_null.positive? && p.range.nil? }
|
|
@@ -503,13 +674,17 @@ module Herringbone
|
|
|
503
674
|
)
|
|
504
675
|
end
|
|
505
676
|
|
|
677
|
+
# @param ranges [Array<Array(Object, Object)>] [min, max] of each non-null page, in page order
|
|
678
|
+
# @param order [Proc, Method] sort key from #sort_key
|
|
679
|
+
# @return [Integer] Format::BoundaryOrder: ASCENDING when both mins and maxes never decrease
|
|
680
|
+
# (or there are fewer than two pages), DESCENDING when they never increase, else UNORDERED
|
|
506
681
|
def boundary_order(ranges, order)
|
|
507
682
|
return Format::BoundaryOrder::ASCENDING if ranges.size < 2
|
|
508
683
|
cmp = ->(a, b) { order.equal?(IDENTITY) ? a <=> b : order.call(a) <=> order.call(b) }
|
|
509
684
|
pairs = ranges.each_cons(2)
|
|
510
|
-
if pairs.all? { |(a_min, a_max), (b_min, b_max)| cmp.(a_min, b_min) <= 0 && cmp.(a_max, b_max) <= 0 }
|
|
685
|
+
if pairs.all? { |(a_min, a_max), (b_min, b_max)| cmp.call(a_min, b_min) <= 0 && cmp.call(a_max, b_max) <= 0 }
|
|
511
686
|
Format::BoundaryOrder::ASCENDING
|
|
512
|
-
elsif pairs.all? { |(a_min, a_max), (b_min, b_max)| cmp.(a_min, b_min) >= 0 && cmp.(a_max, b_max) >= 0 }
|
|
687
|
+
elsif pairs.all? { |(a_min, a_max), (b_min, b_max)| cmp.call(a_min, b_min) >= 0 && cmp.call(a_max, b_max) >= 0 }
|
|
513
688
|
Format::BoundaryOrder::DESCENDING
|
|
514
689
|
else
|
|
515
690
|
Format::BoundaryOrder::UNORDERED
|
|
@@ -517,6 +692,8 @@ module Herringbone
|
|
|
517
692
|
end
|
|
518
693
|
|
|
519
694
|
# Page indexes go after the last row group: all column indexes, then all offset indexes
|
|
695
|
+
#
|
|
696
|
+
# @return [void]
|
|
520
697
|
def write_page_indexes
|
|
521
698
|
@page_indexes.each do |chunk, column_index, _|
|
|
522
699
|
next unless column_index
|
|
@@ -534,17 +711,25 @@ module Herringbone
|
|
|
534
711
|
@page_indexes.clear
|
|
535
712
|
end
|
|
536
713
|
|
|
714
|
+
# Per-column settings accepted in the +bloom_filters:+ option
|
|
537
715
|
BLOOM_FILTER_OPTIONS = %i[ndv fpp max_bytes].freeze
|
|
538
716
|
|
|
539
717
|
# { dotted_path => { ndv:, fpp:, max_bytes: } } from the bloom_filters: option
|
|
540
|
-
|
|
541
|
-
|
|
718
|
+
#
|
|
719
|
+
# @param requested [Boolean, Array<String>, Hash{String => Boolean, Hash, nil}, nil] true for every
|
|
720
|
+
# column of a supported type, column paths, or column path => true / false / settings Hash.
|
|
721
|
+
# A path may also be given as an Array of names. Settings are +ndv+, +fpp+ and +max_bytes+.
|
|
722
|
+
# @return [Hash{String => Hash{Symbol => Numeric, nil}}] settings by dotted path; empty when disabled
|
|
723
|
+
# @raise [ArgumentError] for an unknown column, a column type without bloom filter support, or
|
|
724
|
+
# invalid settings
|
|
725
|
+
def bloom_filter_config(requested)
|
|
726
|
+
return {} if requested.nil? || requested == false
|
|
542
727
|
columns = @schema.columns.to_h { |c| [c.dotted_path, c] }
|
|
543
|
-
if
|
|
728
|
+
if requested == true
|
|
544
729
|
return columns.select { |_, c| BloomFilter::TYPES.include?(c.type) }.transform_values { bloom_filter_settings({}) }
|
|
545
730
|
end
|
|
546
|
-
|
|
547
|
-
|
|
731
|
+
requested = Array(requested).to_h { |path| [path, true] } unless requested.is_a?(Hash)
|
|
732
|
+
requested.each_with_object({}) do |(path, settings), config|
|
|
548
733
|
path = path.is_a?(Array) ? path.join(".") : path.to_s
|
|
549
734
|
col = columns[path] or raise ArgumentError, "bloom_filters: no such column #{path}"
|
|
550
735
|
next if settings.nil? || settings == false
|
|
@@ -557,6 +742,11 @@ module Herringbone
|
|
|
557
742
|
end
|
|
558
743
|
end
|
|
559
744
|
|
|
745
|
+
# @param settings [Hash{Symbol => Object}] +ndv+, +fpp+ and +max_bytes+, all optional
|
|
746
|
+
# @param path [String, nil] column path, for error messages
|
|
747
|
+
# @return [Hash{Symbol => Numeric, nil}] +ndv+ (nil to count distinct values), +fpp+ and
|
|
748
|
+
# +max_bytes+ with defaults filled in
|
|
749
|
+
# @raise [ArgumentError] for unknown keys, a non-positive ndv, or an fpp outside (0, 1)
|
|
560
750
|
def bloom_filter_settings(settings, path = nil)
|
|
561
751
|
unknown = settings.keys - BLOOM_FILTER_OPTIONS
|
|
562
752
|
raise ArgumentError, "bloom_filters: unknown option #{unknown.join(", ")} for #{path}" unless unknown.empty?
|
|
@@ -564,12 +754,18 @@ module Herringbone
|
|
|
564
754
|
raise ArgumentError, "bloom_filters: ndv must be positive for #{path}" if ndv && !ndv.positive?
|
|
565
755
|
fpp = Float(settings.fetch(:fpp, BloomFilter::DEFAULT_FPP))
|
|
566
756
|
raise ArgumentError, "bloom_filters: fpp must be between 0 and 1 for #{path}" unless fpp > 0 && fpp < 1
|
|
567
|
-
{
|
|
757
|
+
{ndv: ndv, fpp: fpp, max_bytes: Integer(settings.fetch(:max_bytes, BloomFilter::DEFAULT_MAX_BYTES))}
|
|
568
758
|
end
|
|
569
759
|
|
|
570
760
|
# A filter holding the chunk's values: +dict_values+ (already distinct) for dictionary-encoded
|
|
571
761
|
# chunks, +values+ otherwise. Each distinct value is hashed once, and the filter is sized from
|
|
572
762
|
# the configured ndv or from the number of distinct values.
|
|
763
|
+
#
|
|
764
|
+
# @param col [Schema::Column] column the filter is for
|
|
765
|
+
# @param settings [Hash{Symbol => Numeric, nil}] from #bloom_filter_settings
|
|
766
|
+
# @param dict_values [Array, nil] dictionary of the chunk, when dictionary-encoded
|
|
767
|
+
# @param values [Array, nil] the chunk's non-null physical values, used without a dictionary
|
|
768
|
+
# @return [BloomFilter]
|
|
573
769
|
def build_bloom_filter(col, settings, dict_values, values)
|
|
574
770
|
hashes = if dict_values
|
|
575
771
|
BloomFilter.hash_physical_all(dict_values, col.type)
|
|
@@ -583,6 +779,8 @@ module Herringbone
|
|
|
583
779
|
end
|
|
584
780
|
|
|
585
781
|
# Bloom filters go right after the row group's column chunks, in column order
|
|
782
|
+
#
|
|
783
|
+
# @return [void]
|
|
586
784
|
def write_bloom_filters
|
|
587
785
|
@pending_bloom_filters.each do |meta, filter|
|
|
588
786
|
bytes = filter.encode
|
|
@@ -593,6 +791,9 @@ module Herringbone
|
|
|
593
791
|
@pending_bloom_filters.clear
|
|
594
792
|
end
|
|
595
793
|
|
|
794
|
+
# @param col [Schema::Column] column to check
|
|
795
|
+
# @return [Boolean] whether the +dictionary:+ option asks for this column to be dictionary-encoded
|
|
796
|
+
# (never for BOOLEAN)
|
|
596
797
|
def use_dictionary?(col)
|
|
597
798
|
return false if col.type == T::BOOLEAN
|
|
598
799
|
case @dictionary
|
|
@@ -602,13 +803,20 @@ module Herringbone
|
|
|
602
803
|
end
|
|
603
804
|
end
|
|
604
805
|
|
|
605
|
-
# Returns [dictionary_values, indices] or nil when a dictionary is not worthwhile
|
|
806
|
+
# Returns [dictionary_values, indices] or nil when a dictionary is not worthwhile: more than about
|
|
807
|
+
# half the values are distinct, or the dictionary exceeds MAX_DICTIONARY_BYTES
|
|
808
|
+
#
|
|
809
|
+
# @param values [Array] non-null physical values of the chunk
|
|
810
|
+
# @param type [Integer] physical type
|
|
811
|
+
# @param type_length [Integer, nil] FIXED_LEN_BYTE_ARRAY width
|
|
812
|
+
# @return [Array(Array, Array<Integer>), nil]
|
|
606
813
|
def build_dictionary(values, type, type_length)
|
|
607
814
|
# Floats are keyed by bit pattern so that -0.0 and 0.0 (and NaNs) stay distinct
|
|
608
815
|
if type == T::FLOAT || type == T::DOUBLE
|
|
609
816
|
keys = values.pack("G*").unpack("Q>*")
|
|
610
817
|
uniq = keys.uniq
|
|
611
818
|
return nil if uniq.size > values.size / 2 + 1 && values.size > 16
|
|
819
|
+
return nil if uniq.size * 8 > MAX_DICTIONARY_BYTES
|
|
612
820
|
index = uniq.each_with_index.to_h
|
|
613
821
|
return [uniq.pack("Q>*").unpack("G*"), keys.map(&index)]
|
|
614
822
|
end
|
|
@@ -626,11 +834,15 @@ module Herringbone
|
|
|
626
834
|
|
|
627
835
|
# Splits a column buffer into pages of roughly @page_bytes bytes. Repeated columns
|
|
628
836
|
# are only cut where a new row starts.
|
|
837
|
+
#
|
|
838
|
+
# @param buf [ColumnBuffer] column buffer with levels as Arrays
|
|
839
|
+
# @param value_bytes [Integer] estimated encoded size of all the chunk's values
|
|
840
|
+
# @return [Array<Array(Integer, Integer)>] [from, to) ranges of level entries, one per page
|
|
629
841
|
def page_ranges(buf, value_bytes)
|
|
630
842
|
n = buf.defs.size
|
|
631
843
|
bytes = n + value_bytes
|
|
632
844
|
pages = (bytes + @page_bytes - 1) / @page_bytes
|
|
633
|
-
per = pages <= 1 ? n : (n + pages - 1) / pages
|
|
845
|
+
per = (pages <= 1) ? n : (n + pages - 1) / pages
|
|
634
846
|
per = @page_rows if per > @page_rows
|
|
635
847
|
return [[0, n]] if per >= n
|
|
636
848
|
reps = buf.reps
|
|
@@ -646,6 +858,8 @@ module Herringbone
|
|
|
646
858
|
ranges
|
|
647
859
|
end
|
|
648
860
|
|
|
861
|
+
# @param col [Schema::Column] column to check
|
|
862
|
+
# @return [Integer, nil] PLAIN-encoded bytes per value, nil for BYTE_ARRAY (variable width)
|
|
649
863
|
def value_width(col)
|
|
650
864
|
case col.type
|
|
651
865
|
when T::BOOLEAN then 1
|
|
@@ -656,6 +870,15 @@ module Herringbone
|
|
|
656
870
|
end
|
|
657
871
|
end
|
|
658
872
|
|
|
873
|
+
# Encodes a page's values. RLE_DICTIONARY values are the dictionary indices, prefixed with the
|
|
874
|
+
# bit width byte; RLE is only used for BOOLEAN and carries the 4-byte length prefix.
|
|
875
|
+
#
|
|
876
|
+
# @param values [Array] physical values, or dictionary indices
|
|
877
|
+
# @param encoding [Integer] encoding id
|
|
878
|
+
# @param type [Integer] physical type
|
|
879
|
+
# @param type_length [Integer, nil] FIXED_LEN_BYTE_ARRAY width
|
|
880
|
+
# @param dict_size [Integer, nil] number of dictionary entries, for RLE_DICTIONARY
|
|
881
|
+
# @return [String] encoded binary page values
|
|
659
882
|
def encode_values(values, encoding, type, type_length, dict_size)
|
|
660
883
|
case encoding
|
|
661
884
|
when E::PLAIN then Encodings::Plain.encode(values, type, type_length)
|
|
@@ -666,27 +889,41 @@ module Herringbone
|
|
|
666
889
|
body = Encodings::RLE.encode_hybrid(values.map { |v| v ? 1 : 0 }, 1)
|
|
667
890
|
[body.bytesize].pack("V") << body
|
|
668
891
|
when E::DELTA_BINARY_PACKED
|
|
669
|
-
Encodings::Delta.encode_binary_packed(values, type == T::INT32 ? 32 : 64)
|
|
892
|
+
Encodings::Delta.encode_binary_packed(values, (type == T::INT32) ? 32 : 64)
|
|
670
893
|
when E::DELTA_LENGTH_BYTE_ARRAY then Encodings::Delta.encode_length_byte_array(values)
|
|
671
894
|
when E::DELTA_BYTE_ARRAY then Encodings::Delta.encode_byte_array(values)
|
|
672
895
|
when E::BYTE_STREAM_SPLIT
|
|
673
|
-
width = {
|
|
896
|
+
width = {T::INT32 => 4, T::FLOAT => 4, T::INT64 => 8, T::DOUBLE => 8}[type] || type_length
|
|
674
897
|
Encodings::ByteStreamSplit.encode(Encodings::Plain.encode(values, type, type_length), width)
|
|
675
898
|
end
|
|
676
899
|
end
|
|
677
900
|
|
|
678
901
|
# Writes a page, returning the uncompressed size including the header
|
|
902
|
+
#
|
|
903
|
+
# @param header [Format::PageHeader] header; sizes and CRC32 are filled in here
|
|
904
|
+
# @param body [String] uncompressed page body
|
|
905
|
+
# @param compressed [String, nil] bytes to write as the page body, when already prepared (v2 data
|
|
906
|
+
# pages, whose levels stay uncompressed); +body+ is compressed otherwise
|
|
907
|
+
# @return [Integer]
|
|
679
908
|
def write_page(header, body, compressed = nil)
|
|
680
909
|
compressed ||= Compression.compress(@codec, body)
|
|
681
910
|
header.uncompressed_page_size ||= body.bytesize
|
|
682
911
|
header.compressed_page_size = compressed.bytesize
|
|
683
|
-
header.crc = Zlib.crc32(compressed).then { |c| c >= 0x8000_0000 ? c - 0x1_0000_0000 : c }
|
|
912
|
+
header.crc = Zlib.crc32(compressed).then { |c| (c >= 0x8000_0000) ? c - 0x1_0000_0000 : c }
|
|
684
913
|
encoded = header.encode
|
|
685
914
|
write_raw(encoded)
|
|
686
915
|
write_raw(compressed)
|
|
687
916
|
encoded.bytesize + header.uncompressed_page_size
|
|
688
917
|
end
|
|
689
918
|
|
|
919
|
+
# A DATA_PAGE: length-prefixed repetition and definition levels, then the values, all compressed
|
|
920
|
+
#
|
|
921
|
+
# @param n [Integer] number of level entries (values including nulls)
|
|
922
|
+
# @param rep_bytes [String] RLE-encoded repetition levels, empty when the column has none
|
|
923
|
+
# @param def_bytes [String] RLE-encoded definition levels, empty when the column has none
|
|
924
|
+
# @param encoded [String] encoded values
|
|
925
|
+
# @param encoding [Integer] encoding id of the values
|
|
926
|
+
# @return [Integer] uncompressed size including the header
|
|
690
927
|
def write_data_page_v1(n, rep_bytes, def_bytes, encoded, encoding)
|
|
691
928
|
body = String.new(encoding: Encoding::BINARY)
|
|
692
929
|
body << [rep_bytes.bytesize].pack("V") << rep_bytes unless rep_bytes.empty?
|
|
@@ -702,6 +939,16 @@ module Herringbone
|
|
|
702
939
|
write_page(header, body)
|
|
703
940
|
end
|
|
704
941
|
|
|
942
|
+
# A DATA_PAGE_V2: levels without length prefixes and uncompressed, then the compressed values
|
|
943
|
+
#
|
|
944
|
+
# @param n [Integer] number of level entries (values including nulls)
|
|
945
|
+
# @param nulls [Integer] null entries
|
|
946
|
+
# @param rows [Integer] rows in the page
|
|
947
|
+
# @param rep_bytes [String] RLE-encoded repetition levels, empty when the column has none
|
|
948
|
+
# @param def_bytes [String] RLE-encoded definition levels, empty when the column has none
|
|
949
|
+
# @param encoded [String] encoded values
|
|
950
|
+
# @param encoding [Integer] encoding id of the values
|
|
951
|
+
# @return [Integer] uncompressed size including the header
|
|
705
952
|
def write_data_page_v2(n, nulls, rows, rep_bytes, def_bytes, encoded, encoding)
|
|
706
953
|
compressed = Compression.compress(@codec, encoded)
|
|
707
954
|
header = Format::PageHeader.new(
|
|
@@ -717,9 +964,20 @@ module Herringbone
|
|
|
717
964
|
write_page(header, "".b, rep_bytes + def_bytes + compressed)
|
|
718
965
|
end
|
|
719
966
|
|
|
967
|
+
# Statistics and column index bounds longer than this are truncated, see #truncate_min
|
|
720
968
|
STAT_TRUNCATE_BYTES = 64
|
|
969
|
+
# Sort key for types whose Ruby ordering already matches Parquet's; compared by identity to
|
|
970
|
+
# skip calling it
|
|
721
971
|
IDENTITY = ->(v) { v }
|
|
722
972
|
|
|
973
|
+
# Chunk statistics: null count, plus min/max (flagged exact unless truncated) when the column
|
|
974
|
+
# has a sort order and non-NaN values
|
|
975
|
+
#
|
|
976
|
+
# @param col [Schema::Column] column the chunk belongs to
|
|
977
|
+
# @param defs [Array<Integer>] definition levels of the chunk
|
|
978
|
+
# @param values [Array] distinct or all non-null physical values of the chunk
|
|
979
|
+
# @param order [Proc, Method, nil] sort key from #sort_key
|
|
980
|
+
# @return [Format::Statistics]
|
|
723
981
|
def statistics_for(col, defs, values, order)
|
|
724
982
|
max_def = col.max_definition_level
|
|
725
983
|
nulls = max_def.zero? ? 0 : defs.size - defs.count(max_def)
|
|
@@ -738,13 +996,16 @@ module Herringbone
|
|
|
738
996
|
|
|
739
997
|
# A key giving the Parquet sort order of the column's physical values (IDENTITY when Ruby's own
|
|
740
998
|
# comparison already matches), or nil when the order is undefined (INT96)
|
|
999
|
+
#
|
|
1000
|
+
# @param col [Schema::Column] column to order
|
|
1001
|
+
# @return [Proc, Method, nil]
|
|
741
1002
|
def sort_key(col)
|
|
742
1003
|
kind, _, signed = Types.logical_of(col.node)
|
|
743
1004
|
case col.type
|
|
744
1005
|
when T::BOOLEAN then ->(v) { v ? 1 : 0 }
|
|
745
1006
|
when T::INT32, T::INT64
|
|
746
1007
|
return IDENTITY unless kind == :integer && !signed
|
|
747
|
-
mask = col.type == T::INT32 ? 0xFFFF_FFFF : 0xFFFF_FFFF_FFFF_FFFF
|
|
1008
|
+
mask = (col.type == T::INT32) ? 0xFFFF_FFFF : 0xFFFF_FFFF_FFFF_FFFF
|
|
748
1009
|
->(v) { v & mask }
|
|
749
1010
|
when T::FLOAT, T::DOUBLE then IDENTITY
|
|
750
1011
|
when T::BYTE_ARRAY, T::FIXED_LEN_BYTE_ARRAY
|
|
@@ -756,7 +1017,13 @@ module Herringbone
|
|
|
756
1017
|
end
|
|
757
1018
|
end
|
|
758
1019
|
|
|
759
|
-
# [min, max] of +values+ in column order, ignoring NaNs; nil if there is nothing to compare
|
|
1020
|
+
# [min, max] of +values+ in column order, ignoring NaNs; nil if there is nothing to compare.
|
|
1021
|
+
# A zero float bound is normalized to -0.0 (min) / 0.0 (max), as the spec asks.
|
|
1022
|
+
#
|
|
1023
|
+
# @param col [Schema::Column] column the values belong to
|
|
1024
|
+
# @param values [Array] physical values
|
|
1025
|
+
# @param order [Proc, Method] sort key from #sort_key
|
|
1026
|
+
# @return [Array(Object, Object), nil]
|
|
760
1027
|
def value_range(col, values, order)
|
|
761
1028
|
floats = col.type == T::FLOAT || col.type == T::DOUBLE
|
|
762
1029
|
if floats
|
|
@@ -773,6 +1040,9 @@ module Herringbone
|
|
|
773
1040
|
[min, max]
|
|
774
1041
|
end
|
|
775
1042
|
|
|
1043
|
+
# @param col [Schema::Column] column the value belongs to
|
|
1044
|
+
# @param value [Object] physical value
|
|
1045
|
+
# @return [String] the value PLAIN-encoded, as statistics store it (byte arrays without length)
|
|
776
1046
|
def stat_bytes(col, value)
|
|
777
1047
|
case col.type
|
|
778
1048
|
when T::BOOLEAN then value ? "\x01".b : "\x00".b
|
|
@@ -786,10 +1056,17 @@ module Herringbone
|
|
|
786
1056
|
|
|
787
1057
|
# Long byte-array bounds are truncated: a prefix is still a lower bound for the minimum, and
|
|
788
1058
|
# a prefix with its last byte incremented is an upper bound for the maximum
|
|
1059
|
+
#
|
|
1060
|
+
# @param bytes [String] encoded minimum
|
|
1061
|
+
# @return [String] at most STAT_TRUNCATE_BYTES bytes
|
|
789
1062
|
def truncate_min(bytes)
|
|
790
|
-
bytes.bytesize > STAT_TRUNCATE_BYTES ? bytes.byteslice(0, STAT_TRUNCATE_BYTES) : bytes
|
|
1063
|
+
(bytes.bytesize > STAT_TRUNCATE_BYTES) ? bytes.byteslice(0, STAT_TRUNCATE_BYTES) : bytes
|
|
791
1064
|
end
|
|
792
1065
|
|
|
1066
|
+
# @param bytes [String] encoded maximum
|
|
1067
|
+
# @return [String] at most STAT_TRUNCATE_BYTES bytes, with the last byte that is not 0xFF
|
|
1068
|
+
# incremented (trailing 0xFF bytes dropped); +bytes+ unchanged when it fits, or when the
|
|
1069
|
+
# whole prefix is 0xFF
|
|
793
1070
|
def truncate_max(bytes)
|
|
794
1071
|
return bytes if bytes.bytesize <= STAT_TRUNCATE_BYTES
|
|
795
1072
|
prefix = bytes.byteslice(0, STAT_TRUNCATE_BYTES).bytes
|