herringbone 0.1.0 → 0.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/README.md +146 -74
- data/bin/herringbone +38 -26
- data/lib/herringbone/active_record.rb +10 -13
- data/lib/herringbone/bloom_filter.rb +270 -0
- data/lib/herringbone/compression.rb +73 -16
- data/lib/herringbone/encodings/delta.rb +7 -7
- data/lib/herringbone/encodings/plain.rb +7 -7
- data/lib/herringbone/encodings/rle.rb +2 -2
- data/lib/herringbone/format.rb +53 -0
- data/lib/herringbone/inspector.rb +1388 -0
- data/lib/herringbone/reader/column_chunk_reader.rb +296 -0
- data/lib/herringbone/reader/column_cursor.rb +181 -0
- data/lib/herringbone/reader/page_stream.rb +339 -0
- data/lib/herringbone/reader/scan.rb +256 -0
- data/lib/herringbone/reader.rb +294 -272
- data/lib/herringbone/schema.rb +21 -76
- data/lib/herringbone/types.rb +4 -4
- data/lib/herringbone/version.rb +1 -1
- data/lib/herringbone/visualizer.rb +1096 -0
- data/lib/herringbone/writer.rb +258 -90
- data/lib/herringbone/xxhash.rb +319 -0
- data/lib/herringbone.rb +36 -9
- metadata +12 -32
data/lib/herringbone/writer.rb
CHANGED
|
@@ -8,27 +8,37 @@ module Herringbone
|
|
|
8
8
|
# string :name
|
|
9
9
|
# list :tags, :string
|
|
10
10
|
# end
|
|
11
|
-
#
|
|
12
|
-
#
|
|
13
|
-
#
|
|
14
|
-
#
|
|
11
|
+
# File.open("out.parquet", "wb") do |file|
|
|
12
|
+
# Herringbone::Writer.open(file, schema) do |w|
|
|
13
|
+
# w << { "id" => 1, "name" => "one", "tags" => ["a", "b"] }
|
|
14
|
+
# w << [2, "two", []] # Arrays are taken in schema order
|
|
15
|
+
# w << order # objects responding to #attributes (ActiveRecord) or #to_h
|
|
16
|
+
# end
|
|
15
17
|
# end
|
|
16
18
|
#
|
|
17
|
-
#
|
|
18
|
-
#
|
|
19
|
-
#
|
|
19
|
+
# The output is any IO that responds to #write (a File, StringIO, Tempfile, socket, pipe...); it
|
|
20
|
+
# is written sequentially and never seeked, rewound or closed by the writer. Herringbone does
|
|
21
|
+
# not open files by path.
|
|
20
22
|
#
|
|
21
23
|
# Options:
|
|
22
|
-
# compression: :
|
|
24
|
+
# compression: :snappy (default), :zstd, :gzip, :lz4 (LZ4_RAW), :lz4_hadoop, :brotli, :none
|
|
25
|
+
# (:zstd and :brotli need the zstd-ruby / brotli gems)
|
|
23
26
|
# row_group_bytes: flush a row group once the buffered values take roughly this much memory
|
|
24
27
|
# (default 16MB). This bounds memory use while writing.
|
|
25
|
-
#
|
|
26
|
-
#
|
|
28
|
+
# row_group_rows: also flush after this many rows (default: no row limit)
|
|
29
|
+
# page_bytes: approximate uncompressed data page size (default 1MB)
|
|
30
|
+
# page_rows: at most this many rows per data page (default 20_000), which keeps the page
|
|
31
|
+
# index selective
|
|
27
32
|
# data_page_version: 1 (default) or 2
|
|
28
33
|
# dictionary: true/false, or an Array of column paths to dictionary-encode
|
|
29
34
|
# encodings: { "path.to.column" => :delta_binary_packed, ... } for non-dictionary pages
|
|
30
|
-
# statistics: write min/max/null_count statistics (default true)
|
|
31
35
|
# metadata: Hash of String => String key/value metadata for the footer
|
|
36
|
+
# bloom_filters: write split block bloom filters: true (every column that supports them),
|
|
37
|
+
# an Array of column paths, or { "path" => true | { ndv:, fpp:, max_bytes: } }.
|
|
38
|
+
# Without ndv: the distinct values of each row group are counted. fpp defaults
|
|
39
|
+
# to 0.01 and max_bytes to 1MB. Filters are written after each row group.
|
|
40
|
+
#
|
|
41
|
+
# Statistics and page indexes (ColumnIndex/OffsetIndex) are always written.
|
|
32
42
|
class Writer
|
|
33
43
|
MAGIC = "PAR1"
|
|
34
44
|
T = Format::Type
|
|
@@ -55,14 +65,15 @@ module Herringbone
|
|
|
55
65
|
|
|
56
66
|
attr_reader :schema
|
|
57
67
|
|
|
58
|
-
# Opens a writer on
|
|
59
|
-
#
|
|
60
|
-
|
|
61
|
-
|
|
68
|
+
# Opens a writer on +io+. With a block, the file is finished (footer written) when the block
|
|
69
|
+
# returns and the block's value is returned; if the block raises, the writer is aborted and no
|
|
70
|
+
# footer is written. Without a block, call #close to finish. The IO is never closed.
|
|
71
|
+
def self.open(io, schema, **options)
|
|
72
|
+
writer = new(io, schema, **options)
|
|
62
73
|
return writer unless block_given?
|
|
63
74
|
begin
|
|
64
75
|
result = yield writer
|
|
65
|
-
rescue Exception # rubocop:disable Lint/RescueException -- also
|
|
76
|
+
rescue Exception # rubocop:disable Lint/RescueException -- also abort on Interrupt
|
|
66
77
|
writer.abort
|
|
67
78
|
raise
|
|
68
79
|
end
|
|
@@ -70,29 +81,36 @@ module Herringbone
|
|
|
70
81
|
result
|
|
71
82
|
end
|
|
72
83
|
|
|
73
|
-
def initialize(
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
schema
|
|
84
|
+
def initialize(io, schema, compression: :snappy, row_group_bytes: 16 * 1024 * 1024, row_group_rows: nil,
|
|
85
|
+
page_bytes: 1024 * 1024, page_rows: 20_000, data_page_version: 1, dictionary: true, encodings: {},
|
|
86
|
+
metadata: {}, bloom_filters: nil)
|
|
87
|
+
raise ArgumentError, "Expected a Herringbone::Schema, got #{schema.class}" unless schema.is_a?(Schema)
|
|
88
|
+
@schema = schema
|
|
77
89
|
@codec = Compression.codec_id(compression)
|
|
90
|
+
# Fail before creating any file if the codec's library is missing
|
|
91
|
+
Compression.ensure_available!(@codec)
|
|
78
92
|
@row_group_bytes = Integer(row_group_bytes)
|
|
79
|
-
@
|
|
80
|
-
@row_limit = @
|
|
81
|
-
@
|
|
93
|
+
@row_group_rows = row_group_rows && Integer(row_group_rows)
|
|
94
|
+
@row_limit = @row_group_rows || ESTIMATE_AFTER_ROWS
|
|
95
|
+
@page_bytes = Integer(page_bytes)
|
|
96
|
+
@page_rows = Integer(page_rows)
|
|
97
|
+
raise ArgumentError, "page_rows must be positive" unless @page_rows.positive?
|
|
98
|
+
@page_indexes = [] # [ColumnChunk, ColumnIndex or nil, OffsetIndex] per written column chunk
|
|
82
99
|
@data_page_version = Integer(data_page_version)
|
|
83
100
|
raise ArgumentError, "data_page_version must be 1 or 2" unless [1, 2].include?(@data_page_version)
|
|
84
101
|
@dictionary = dictionary
|
|
85
102
|
@encodings = encodings.to_h { |path, enc| [path.to_s, encoding_id(path, enc)] }
|
|
86
103
|
unknown = @encodings.keys - schema.columns.map(&:dotted_path)
|
|
87
104
|
raise ArgumentError, "encodings: no such column #{unknown.join(", ")}" unless unknown.empty?
|
|
88
|
-
@statistics = statistics
|
|
89
105
|
@metadata = metadata
|
|
106
|
+
@bloom_filters = bloom_filter_config(bloom_filters)
|
|
107
|
+
@pending_bloom_filters = [] # [ColumnMetaData, BloomFilter] for the row group being written
|
|
90
108
|
@row_groups = []
|
|
91
109
|
@total_rows = 0
|
|
92
110
|
@pos = 0
|
|
93
111
|
@closed = false
|
|
94
112
|
@bytes_per_row = nil
|
|
95
|
-
|
|
113
|
+
@io = check_io!(io)
|
|
96
114
|
write_raw(MAGIC)
|
|
97
115
|
reset_buffers
|
|
98
116
|
end
|
|
@@ -131,12 +149,6 @@ module Herringbone
|
|
|
131
149
|
check_row_group_size if @buffered_rows >= @row_limit
|
|
132
150
|
self
|
|
133
151
|
end
|
|
134
|
-
alias_method :write, :<<
|
|
135
|
-
|
|
136
|
-
def write_rows(rows)
|
|
137
|
-
rows.each { |row| self << row }
|
|
138
|
-
self
|
|
139
|
-
end
|
|
140
152
|
|
|
141
153
|
# Number of rows written so far, including buffered ones
|
|
142
154
|
def rows_written = @total_rows + @buffered_rows
|
|
@@ -154,6 +166,7 @@ module Herringbone
|
|
|
154
166
|
total_compressed_size: @pos - start,
|
|
155
167
|
ordinal: @row_groups.size
|
|
156
168
|
)
|
|
169
|
+
write_bloom_filters
|
|
157
170
|
@total_rows += @buffered_rows
|
|
158
171
|
reset_buffers
|
|
159
172
|
end
|
|
@@ -161,13 +174,14 @@ module Herringbone
|
|
|
161
174
|
def close
|
|
162
175
|
return if @closed
|
|
163
176
|
flush_row_group
|
|
177
|
+
write_page_indexes
|
|
164
178
|
meta = Format::FileMetaData.new(
|
|
165
179
|
version: 2,
|
|
166
180
|
schema: @schema.to_elements,
|
|
167
181
|
num_rows: @total_rows,
|
|
168
182
|
row_groups: @row_groups,
|
|
169
183
|
key_value_metadata: @metadata.empty? ? nil : @metadata.map { |k, v| Format::KeyValue.new(key: k.to_s, value: v&.to_s) },
|
|
170
|
-
created_by: "herringbone
|
|
184
|
+
created_by: "herringbone-ruby #{VERSION}",
|
|
171
185
|
column_orders: @schema.columns.map { Format::ColumnOrder.new(type_order: Format::TypeDefinedOrder.new) }
|
|
172
186
|
)
|
|
173
187
|
footer = meta.encode
|
|
@@ -176,20 +190,23 @@ module Herringbone
|
|
|
176
190
|
write_raw(MAGIC)
|
|
177
191
|
@io.flush if @io.respond_to?(:flush)
|
|
178
192
|
@closed = true
|
|
179
|
-
finish_target
|
|
180
193
|
end
|
|
181
194
|
|
|
182
|
-
# Stops writing without
|
|
195
|
+
# Stops writing without finishing the file (no footer is written). Whatever was already written
|
|
196
|
+
# to the IO stays there; discarding it is up to the caller.
|
|
183
197
|
def abort
|
|
184
|
-
return if @closed
|
|
185
198
|
@closed = true
|
|
186
|
-
return unless @temp_path
|
|
187
|
-
@io.close unless @io.closed?
|
|
188
|
-
File.unlink(@temp_path) if File.exist?(@temp_path)
|
|
189
199
|
end
|
|
190
200
|
|
|
191
201
|
private
|
|
192
202
|
|
|
203
|
+
# Pathname responds to #write too (writing a whole file by path), so it is rejected explicitly
|
|
204
|
+
def check_io!(io)
|
|
205
|
+
return io if io.respond_to?(:write) && !io.is_a?(String) && !(defined?(Pathname) && io.is_a?(Pathname))
|
|
206
|
+
raise ArgumentError, "Herringbone::Writer expects an IO that responds to #write " \
|
|
207
|
+
"(e.g. File.open(path, \"wb\") or StringIO.new), got #{io.class}"
|
|
208
|
+
end
|
|
209
|
+
|
|
193
210
|
# Levels are kept as binary Strings (one byte per entry). Values are an Array for numeric and
|
|
194
211
|
# boolean columns, and a compact ByteValues for BYTE_ARRAY / FIXED_LEN_BYTE_ARRAY columns.
|
|
195
212
|
ColumnBuffer = Struct.new(:defs, :reps, :values)
|
|
@@ -220,24 +237,7 @@ module Herringbone
|
|
|
220
237
|
end
|
|
221
238
|
end
|
|
222
239
|
|
|
223
|
-
|
|
224
|
-
if target.respond_to?(:write)
|
|
225
|
-
@io = target
|
|
226
|
-
else
|
|
227
|
-
@path = target.to_s
|
|
228
|
-
dir = File.dirname(@path)
|
|
229
|
-
@temp_path = File.join(dir, ".#{File.basename(@path)}.#{Process.pid}.#{rand(1 << 32).to_s(36)}.tmp")
|
|
230
|
-
@io = File.open(@temp_path, "wb")
|
|
231
|
-
end
|
|
232
|
-
end
|
|
233
|
-
|
|
234
|
-
def finish_target
|
|
235
|
-
return unless @temp_path
|
|
236
|
-
@io.close
|
|
237
|
-
File.rename(@temp_path, @path)
|
|
238
|
-
end
|
|
239
|
-
|
|
240
|
-
# Flushes when the buffered values reach row_group_bytes (or row_group_size rows). The bytes per
|
|
240
|
+
# Flushes when the buffered values reach row_group_bytes (or row_group_rows rows). The bytes per
|
|
241
241
|
# row are estimated from the buffered values after the first rows, then refreshed per row group.
|
|
242
242
|
def check_row_group_size
|
|
243
243
|
@bytes_per_row ||= estimate_bytes_per_row
|
|
@@ -253,7 +253,7 @@ module Herringbone
|
|
|
253
253
|
|
|
254
254
|
def row_limit_for(bytes_per_row)
|
|
255
255
|
by_bytes = [@row_group_bytes / [bytes_per_row, 1].max, ESTIMATE_AFTER_ROWS].max
|
|
256
|
-
@
|
|
256
|
+
@row_group_rows ? [@row_group_rows, by_bytes].min : by_bytes
|
|
257
257
|
end
|
|
258
258
|
|
|
259
259
|
def estimate_bytes_per_row
|
|
@@ -430,6 +430,9 @@ module Herringbone
|
|
|
430
430
|
else
|
|
431
431
|
values.sum(&:bytesize) + 4 * values.size
|
|
432
432
|
end
|
|
433
|
+
order = sort_key(col)
|
|
434
|
+
pages = []
|
|
435
|
+
first_row = 0
|
|
433
436
|
page_ranges(buf, value_bytes).each do |from, to|
|
|
434
437
|
n = to - from
|
|
435
438
|
defs = buf.defs[from, n]
|
|
@@ -437,6 +440,12 @@ module Herringbone
|
|
|
437
440
|
non_null = max_def.zero? ? n : defs.count(max_def)
|
|
438
441
|
page_values = (indices || values)[value_index, non_null]
|
|
439
442
|
value_index += non_null
|
|
443
|
+
page_offset = @pos
|
|
444
|
+
range = nil
|
|
445
|
+
if order && non_null.positive?
|
|
446
|
+
in_page = dict_values ? page_values.uniq.map { |i| dict_values[i] } : page_values
|
|
447
|
+
range = value_range(col, in_page, order)
|
|
448
|
+
end
|
|
440
449
|
encoded = encode_values(page_values, value_encoding, type, col.type_length, dict_values&.size)
|
|
441
450
|
rep_bytes = max_rep.positive? ? Encodings::RLE.encode_hybrid(reps, max_rep.bit_length) : "".b
|
|
442
451
|
def_bytes = max_def.positive? ? Encodings::RLE.encode_hybrid(defs, max_def.bit_length) : "".b
|
|
@@ -446,6 +455,8 @@ module Herringbone
|
|
|
446
455
|
num_rows = max_rep.zero? ? n : reps.count(0)
|
|
447
456
|
write_data_page_v2(n, n - non_null, num_rows, rep_bytes, def_bytes, encoded, value_encoding)
|
|
448
457
|
end
|
|
458
|
+
pages << PageInfo.new(page_offset, @pos - page_offset, first_row, n - non_null, non_null, range)
|
|
459
|
+
first_row += max_rep.zero? ? n : reps.count(0)
|
|
449
460
|
end
|
|
450
461
|
|
|
451
462
|
encodings = [E::RLE]
|
|
@@ -461,9 +472,125 @@ module Herringbone
|
|
|
461
472
|
total_compressed_size: @pos - chunk_start,
|
|
462
473
|
data_page_offset: data_offset,
|
|
463
474
|
dictionary_page_offset: dictionary_offset,
|
|
464
|
-
statistics:
|
|
475
|
+
statistics: statistics_for(col, buf.defs, values || indices.uniq.map { |i| dict_values[i] }, order)
|
|
465
476
|
)
|
|
466
|
-
Format::ColumnChunk.new(file_offset: chunk_start, meta_data: meta)
|
|
477
|
+
chunk = Format::ColumnChunk.new(file_offset: chunk_start, meta_data: meta)
|
|
478
|
+
if (bloom = @bloom_filters[path])
|
|
479
|
+
@pending_bloom_filters << [meta, build_bloom_filter(col, bloom, dict_values, values)]
|
|
480
|
+
end
|
|
481
|
+
@page_indexes << [chunk, column_index_for(col, pages, order), offset_index_for(pages)]
|
|
482
|
+
chunk
|
|
483
|
+
end
|
|
484
|
+
|
|
485
|
+
PageInfo = Struct.new(:offset, :size, :first_row, :nulls, :non_null, :range)
|
|
486
|
+
|
|
487
|
+
def offset_index_for(pages)
|
|
488
|
+
Format::OffsetIndex.new(page_locations: pages.map do |p|
|
|
489
|
+
Format::PageLocation.new(offset: p.offset, compressed_page_size: p.size, first_row_index: p.first_row)
|
|
490
|
+
end)
|
|
491
|
+
end
|
|
492
|
+
|
|
493
|
+
# nil when the column has no defined sort order, or a page's values have no min/max (all NaN)
|
|
494
|
+
def column_index_for(col, pages, order)
|
|
495
|
+
return nil unless order
|
|
496
|
+
return nil if pages.any? { |p| p.non_null.positive? && p.range.nil? }
|
|
497
|
+
Format::ColumnIndex.new(
|
|
498
|
+
null_pages: pages.map { |p| p.non_null.zero? },
|
|
499
|
+
min_values: pages.map { |p| p.range ? truncate_min(stat_bytes(col, p.range[0])) : "".b },
|
|
500
|
+
max_values: pages.map { |p| p.range ? truncate_max(stat_bytes(col, p.range[1])) : "".b },
|
|
501
|
+
boundary_order: boundary_order(pages.filter_map(&:range), order),
|
|
502
|
+
null_counts: pages.map(&:nulls)
|
|
503
|
+
)
|
|
504
|
+
end
|
|
505
|
+
|
|
506
|
+
def boundary_order(ranges, order)
|
|
507
|
+
return Format::BoundaryOrder::ASCENDING if ranges.size < 2
|
|
508
|
+
cmp = ->(a, b) { order.equal?(IDENTITY) ? a <=> b : order.call(a) <=> order.call(b) }
|
|
509
|
+
pairs = ranges.each_cons(2)
|
|
510
|
+
if pairs.all? { |(a_min, a_max), (b_min, b_max)| cmp.(a_min, b_min) <= 0 && cmp.(a_max, b_max) <= 0 }
|
|
511
|
+
Format::BoundaryOrder::ASCENDING
|
|
512
|
+
elsif pairs.all? { |(a_min, a_max), (b_min, b_max)| cmp.(a_min, b_min) >= 0 && cmp.(a_max, b_max) >= 0 }
|
|
513
|
+
Format::BoundaryOrder::DESCENDING
|
|
514
|
+
else
|
|
515
|
+
Format::BoundaryOrder::UNORDERED
|
|
516
|
+
end
|
|
517
|
+
end
|
|
518
|
+
|
|
519
|
+
# Page indexes go after the last row group: all column indexes, then all offset indexes
|
|
520
|
+
def write_page_indexes
|
|
521
|
+
@page_indexes.each do |chunk, column_index, _|
|
|
522
|
+
next unless column_index
|
|
523
|
+
bytes = column_index.encode
|
|
524
|
+
chunk.column_index_offset = @pos
|
|
525
|
+
chunk.column_index_length = bytes.bytesize
|
|
526
|
+
write_raw(bytes)
|
|
527
|
+
end
|
|
528
|
+
@page_indexes.each do |chunk, _, offset_index|
|
|
529
|
+
bytes = offset_index.encode
|
|
530
|
+
chunk.offset_index_offset = @pos
|
|
531
|
+
chunk.offset_index_length = bytes.bytesize
|
|
532
|
+
write_raw(bytes)
|
|
533
|
+
end
|
|
534
|
+
@page_indexes.clear
|
|
535
|
+
end
|
|
536
|
+
|
|
537
|
+
BLOOM_FILTER_OPTIONS = %i[ndv fpp max_bytes].freeze
|
|
538
|
+
|
|
539
|
+
# { dotted_path => { ndv:, fpp:, max_bytes: } } from the bloom_filters: option
|
|
540
|
+
def bloom_filter_config(option)
|
|
541
|
+
return {} if option.nil? || option == false
|
|
542
|
+
columns = @schema.columns.to_h { |c| [c.dotted_path, c] }
|
|
543
|
+
if option == true
|
|
544
|
+
return columns.select { |_, c| BloomFilter::TYPES.include?(c.type) }.transform_values { bloom_filter_settings({}) }
|
|
545
|
+
end
|
|
546
|
+
option = Array(option).to_h { |path| [path, true] } unless option.is_a?(Hash)
|
|
547
|
+
option.each_with_object({}) do |(path, settings), config|
|
|
548
|
+
path = path.is_a?(Array) ? path.join(".") : path.to_s
|
|
549
|
+
col = columns[path] or raise ArgumentError, "bloom_filters: no such column #{path}"
|
|
550
|
+
next if settings.nil? || settings == false
|
|
551
|
+
unless BloomFilter::TYPES.include?(col.type)
|
|
552
|
+
raise ArgumentError, "bloom_filters: not supported for #{T::NAMES[col.type]} column #{path}"
|
|
553
|
+
end
|
|
554
|
+
settings = {} if settings == true
|
|
555
|
+
raise ArgumentError, "bloom_filters: expected true or a Hash for #{path}, got #{settings.inspect}" unless settings.is_a?(Hash)
|
|
556
|
+
config[path] = bloom_filter_settings(settings.transform_keys(&:to_sym), path)
|
|
557
|
+
end
|
|
558
|
+
end
|
|
559
|
+
|
|
560
|
+
def bloom_filter_settings(settings, path = nil)
|
|
561
|
+
unknown = settings.keys - BLOOM_FILTER_OPTIONS
|
|
562
|
+
raise ArgumentError, "bloom_filters: unknown option #{unknown.join(", ")} for #{path}" unless unknown.empty?
|
|
563
|
+
ndv = settings[:ndv] && Integer(settings[:ndv])
|
|
564
|
+
raise ArgumentError, "bloom_filters: ndv must be positive for #{path}" if ndv && !ndv.positive?
|
|
565
|
+
fpp = Float(settings.fetch(:fpp, BloomFilter::DEFAULT_FPP))
|
|
566
|
+
raise ArgumentError, "bloom_filters: fpp must be between 0 and 1 for #{path}" unless fpp > 0 && fpp < 1
|
|
567
|
+
{ ndv: ndv, fpp: fpp, max_bytes: Integer(settings.fetch(:max_bytes, BloomFilter::DEFAULT_MAX_BYTES)) }
|
|
568
|
+
end
|
|
569
|
+
|
|
570
|
+
# A filter holding the chunk's values: +dict_values+ (already distinct) for dictionary-encoded
|
|
571
|
+
# chunks, +values+ otherwise. Each distinct value is hashed once, and the filter is sized from
|
|
572
|
+
# the configured ndv or from the number of distinct values.
|
|
573
|
+
def build_bloom_filter(col, settings, dict_values, values)
|
|
574
|
+
hashes = if dict_values
|
|
575
|
+
BloomFilter.hash_physical_all(dict_values, col.type)
|
|
576
|
+
else
|
|
577
|
+
BloomFilter.hash_physical_all(values, col.type, distinct: true)
|
|
578
|
+
end
|
|
579
|
+
size = BloomFilter.optimal_num_bytes(settings[:ndv] || hashes.size, settings[:fpp], max_bytes: settings[:max_bytes])
|
|
580
|
+
filter = BloomFilter.new(size, column: col)
|
|
581
|
+
filter.insert_hashes(hashes)
|
|
582
|
+
filter
|
|
583
|
+
end
|
|
584
|
+
|
|
585
|
+
# Bloom filters go right after the row group's column chunks, in column order
|
|
586
|
+
def write_bloom_filters
|
|
587
|
+
@pending_bloom_filters.each do |meta, filter|
|
|
588
|
+
bytes = filter.encode
|
|
589
|
+
meta.bloom_filter_offset = @pos
|
|
590
|
+
meta.bloom_filter_length = bytes.bytesize
|
|
591
|
+
write_raw(bytes)
|
|
592
|
+
end
|
|
593
|
+
@pending_bloom_filters.clear
|
|
467
594
|
end
|
|
468
595
|
|
|
469
596
|
def use_dictionary?(col)
|
|
@@ -497,14 +624,15 @@ module Herringbone
|
|
|
497
624
|
[dict, values.map(&index)]
|
|
498
625
|
end
|
|
499
626
|
|
|
500
|
-
# Splits a column buffer into pages of roughly @
|
|
627
|
+
# Splits a column buffer into pages of roughly @page_bytes bytes. Repeated columns
|
|
501
628
|
# are only cut where a new row starts.
|
|
502
629
|
def page_ranges(buf, value_bytes)
|
|
503
630
|
n = buf.defs.size
|
|
504
631
|
bytes = n + value_bytes
|
|
505
|
-
pages = (bytes + @
|
|
506
|
-
|
|
507
|
-
per =
|
|
632
|
+
pages = (bytes + @page_bytes - 1) / @page_bytes
|
|
633
|
+
per = pages <= 1 ? n : (n + pages - 1) / pages
|
|
634
|
+
per = @page_rows if per > @page_rows
|
|
635
|
+
return [[0, n]] if per >= n
|
|
508
636
|
reps = buf.reps
|
|
509
637
|
ranges = []
|
|
510
638
|
start = 0
|
|
@@ -589,46 +717,86 @@ module Herringbone
|
|
|
589
717
|
write_page(header, "".b, rep_bytes + def_bytes + compressed)
|
|
590
718
|
end
|
|
591
719
|
|
|
592
|
-
|
|
720
|
+
STAT_TRUNCATE_BYTES = 64
|
|
721
|
+
IDENTITY = ->(v) { v }
|
|
722
|
+
|
|
723
|
+
def statistics_for(col, defs, values, order)
|
|
593
724
|
max_def = col.max_definition_level
|
|
594
725
|
nulls = max_def.zero? ? 0 : defs.size - defs.count(max_def)
|
|
595
726
|
stats = Format::Statistics.new(null_count: nulls)
|
|
596
|
-
|
|
597
|
-
if
|
|
598
|
-
|
|
599
|
-
|
|
600
|
-
stats.
|
|
601
|
-
stats.
|
|
727
|
+
range = order && value_range(col, values, order)
|
|
728
|
+
if range
|
|
729
|
+
min = stat_bytes(col, range[0])
|
|
730
|
+
max = stat_bytes(col, range[1])
|
|
731
|
+
stats.min_value = truncate_min(min)
|
|
732
|
+
stats.max_value = truncate_max(max)
|
|
733
|
+
stats.is_min_value_exact = stats.min_value.bytesize == min.bytesize
|
|
734
|
+
stats.is_max_value_exact = stats.max_value.bytesize == max.bytesize
|
|
602
735
|
end
|
|
603
736
|
stats
|
|
604
737
|
end
|
|
605
738
|
|
|
606
|
-
#
|
|
607
|
-
|
|
608
|
-
|
|
739
|
+
# A key giving the Parquet sort order of the column's physical values (IDENTITY when Ruby's own
|
|
740
|
+
# comparison already matches), or nil when the order is undefined (INT96)
|
|
741
|
+
def sort_key(col)
|
|
609
742
|
kind, _, signed = Types.logical_of(col.node)
|
|
610
743
|
case col.type
|
|
611
|
-
when T::BOOLEAN
|
|
612
|
-
[values.include?(false) ? "\x00".b : "\x01".b, values.include?(true) ? "\x01".b : "\x00".b]
|
|
744
|
+
when T::BOOLEAN then ->(v) { v ? 1 : 0 }
|
|
613
745
|
when T::INT32, T::INT64
|
|
614
|
-
return
|
|
615
|
-
|
|
616
|
-
|
|
617
|
-
|
|
618
|
-
when T::
|
|
619
|
-
|
|
620
|
-
|
|
621
|
-
|
|
746
|
+
return IDENTITY unless kind == :integer && !signed
|
|
747
|
+
mask = col.type == T::INT32 ? 0xFFFF_FFFF : 0xFFFF_FFFF_FFFF_FFFF
|
|
748
|
+
->(v) { v & mask }
|
|
749
|
+
when T::FLOAT, T::DOUBLE then IDENTITY
|
|
750
|
+
when T::BYTE_ARRAY, T::FIXED_LEN_BYTE_ARRAY
|
|
751
|
+
case kind
|
|
752
|
+
when :decimal then Types.method(:be_to_int)
|
|
753
|
+
when :float16 then ->(v) { Types.half_to_float(v.unpack1("v")) }
|
|
754
|
+
else IDENTITY # unsigned lexicographic, which is how Ruby compares binary Strings
|
|
755
|
+
end
|
|
756
|
+
end
|
|
757
|
+
end
|
|
758
|
+
|
|
759
|
+
# [min, max] of +values+ in column order, ignoring NaNs; nil if there is nothing to compare
|
|
760
|
+
def value_range(col, values, order)
|
|
761
|
+
floats = col.type == T::FLOAT || col.type == T::DOUBLE
|
|
762
|
+
if floats
|
|
763
|
+
values = values.reject(&:nan?)
|
|
764
|
+
elsif Types.logical_of(col.node).first == :float16
|
|
765
|
+
values = values.reject { |v| order.call(v).nan? }
|
|
766
|
+
end
|
|
767
|
+
return nil if values.empty?
|
|
768
|
+
min, max = order.equal?(IDENTITY) ? values.minmax : values.minmax_by(&order)
|
|
769
|
+
if floats
|
|
622
770
|
min = -0.0 if min.zero?
|
|
623
771
|
max = 0.0 if max.zero?
|
|
624
|
-
fmt = col.type == T::FLOAT ? "e" : "E"
|
|
625
|
-
[[min].pack(fmt), [max].pack(fmt)]
|
|
626
|
-
when T::BYTE_ARRAY
|
|
627
|
-
return nil if kind == :decimal
|
|
628
|
-
min, max = values.minmax
|
|
629
|
-
return nil if min.bytesize > 1024 || max.bytesize > 1024
|
|
630
|
-
[min, max]
|
|
631
772
|
end
|
|
773
|
+
[min, max]
|
|
774
|
+
end
|
|
775
|
+
|
|
776
|
+
def stat_bytes(col, value)
|
|
777
|
+
case col.type
|
|
778
|
+
when T::BOOLEAN then value ? "\x01".b : "\x00".b
|
|
779
|
+
when T::INT32 then [value].pack("l<")
|
|
780
|
+
when T::INT64 then [value].pack("q<")
|
|
781
|
+
when T::FLOAT then [value].pack("e")
|
|
782
|
+
when T::DOUBLE then [value].pack("E")
|
|
783
|
+
else value.b
|
|
784
|
+
end
|
|
785
|
+
end
|
|
786
|
+
|
|
787
|
+
# Long byte-array bounds are truncated: a prefix is still a lower bound for the minimum, and
|
|
788
|
+
# a prefix with its last byte incremented is an upper bound for the maximum
|
|
789
|
+
def truncate_min(bytes)
|
|
790
|
+
bytes.bytesize > STAT_TRUNCATE_BYTES ? bytes.byteslice(0, STAT_TRUNCATE_BYTES) : bytes
|
|
791
|
+
end
|
|
792
|
+
|
|
793
|
+
def truncate_max(bytes)
|
|
794
|
+
return bytes if bytes.bytesize <= STAT_TRUNCATE_BYTES
|
|
795
|
+
prefix = bytes.byteslice(0, STAT_TRUNCATE_BYTES).bytes
|
|
796
|
+
prefix.pop while prefix.last == 0xFF
|
|
797
|
+
return bytes if prefix.empty?
|
|
798
|
+
prefix[-1] += 1
|
|
799
|
+
prefix.pack("C*")
|
|
632
800
|
end
|
|
633
801
|
end
|
|
634
802
|
end
|