herringbone 0.1.0 → 0.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/README.md +198 -74
- data/bin/herringbone +38 -26
- data/lib/herringbone/active_record.rb +48 -19
- data/lib/herringbone/bloom_filter.rb +370 -0
- data/lib/herringbone/byte_values.rb +29 -0
- data/lib/herringbone/codecs/lz4.rb +66 -3
- data/lib/herringbone/codecs/snappy.rb +132 -3
- data/lib/herringbone/compression.rb +121 -16
- data/lib/herringbone/encodings/delta.rb +76 -19
- data/lib/herringbone/encodings/plain.rb +29 -7
- data/lib/herringbone/encodings/rle.rb +55 -3
- data/lib/herringbone/format.rb +203 -0
- data/lib/herringbone/inspector.rb +1784 -0
- data/lib/herringbone/io_buffer_support.rb +3 -1
- data/lib/herringbone/reader/column_chunk_reader.rb +388 -0
- data/lib/herringbone/reader/column_cursor.rb +225 -0
- data/lib/herringbone/reader/numo.rb +365 -0
- data/lib/herringbone/reader/page_stream.rb +615 -0
- data/lib/herringbone/reader/scan.rb +350 -0
- data/lib/herringbone/reader.rb +585 -272
- data/lib/herringbone/schema.rb +326 -90
- data/lib/herringbone/thrift.rb +155 -4
- data/lib/herringbone/types.rb +177 -42
- data/lib/herringbone/version.rb +2 -1
- data/lib/herringbone/visualizer.rb +1136 -0
- data/lib/herringbone/writer.rb +550 -105
- data/lib/herringbone/xxhash.rb +425 -0
- data/lib/herringbone.rb +64 -9
- metadata +14 -33
data/lib/herringbone/writer.rb
CHANGED
|
@@ -8,38 +8,54 @@ module Herringbone
|
|
|
8
8
|
# string :name
|
|
9
9
|
# list :tags, :string
|
|
10
10
|
# end
|
|
11
|
-
#
|
|
12
|
-
#
|
|
13
|
-
#
|
|
14
|
-
#
|
|
11
|
+
# File.open("out.parquet", "wb") do |file|
|
|
12
|
+
# Herringbone::Writer.open(file, schema) do |w|
|
|
13
|
+
# w << { "id" => 1, "name" => "one", "tags" => ["a", "b"] }
|
|
14
|
+
# w << [2, "two", []] # Arrays are taken in schema order
|
|
15
|
+
# w << order # objects responding to #attributes (ActiveRecord) or #to_h
|
|
16
|
+
# end
|
|
15
17
|
# end
|
|
16
18
|
#
|
|
17
|
-
#
|
|
18
|
-
#
|
|
19
|
-
#
|
|
19
|
+
# The output is any IO that responds to #write (a File, StringIO, Tempfile, socket, pipe...); it
|
|
20
|
+
# is written sequentially and never seeked, rewound or closed by the writer; only #write is
|
|
21
|
+
# required, its return value is ignored, and #binmode and #flush are called when available.
|
|
22
|
+
# Herringbone does not open files by path.
|
|
20
23
|
#
|
|
21
24
|
# Options:
|
|
22
|
-
# compression: :
|
|
25
|
+
# compression: :snappy (default), :zstd, :gzip, :lz4 (LZ4_RAW), :lz4_hadoop, :brotli, :none
|
|
26
|
+
# (:zstd and :brotli need the zstd-ruby / brotli gems)
|
|
23
27
|
# row_group_bytes: flush a row group once the buffered values take roughly this much memory
|
|
24
28
|
# (default 16MB). This bounds memory use while writing.
|
|
25
|
-
#
|
|
26
|
-
#
|
|
29
|
+
# row_group_rows: also flush after this many rows (default: no row limit)
|
|
30
|
+
# page_bytes: approximate uncompressed data page size (default 1MB)
|
|
31
|
+
# page_rows: at most this many rows per data page (default 20_000), which keeps the page
|
|
32
|
+
# index selective
|
|
27
33
|
# data_page_version: 1 (default) or 2
|
|
28
34
|
# dictionary: true/false, or an Array of column paths to dictionary-encode
|
|
29
35
|
# encodings: { "path.to.column" => :delta_binary_packed, ... } for non-dictionary pages
|
|
30
|
-
# statistics: write min/max/null_count statistics (default true)
|
|
31
36
|
# metadata: Hash of String => String key/value metadata for the footer
|
|
37
|
+
# bloom_filters: write split block bloom filters: true (every column that supports them),
|
|
38
|
+
# an Array of column paths, or { "path" => true | { ndv:, fpp:, max_bytes: } }.
|
|
39
|
+
# Without ndv: the distinct values of each row group are counted. fpp defaults
|
|
40
|
+
# to 0.01 and max_bytes to 1MB. Filters are written after each row group.
|
|
41
|
+
#
|
|
42
|
+
# Statistics and page indexes (ColumnIndex/OffsetIndex) are always written.
|
|
32
43
|
class Writer
|
|
33
|
-
|
|
44
|
+
# Marker at the start and end of every Parquet file
|
|
45
|
+
MAGIC = "PAR1".b.freeze
|
|
46
|
+
# Shorthand for Format::Type (physical types)
|
|
34
47
|
T = Format::Type
|
|
48
|
+
# Shorthand for Format::Encoding
|
|
35
49
|
E = Format::Encoding
|
|
36
50
|
|
|
51
|
+
# Names accepted in the +encodings:+ option => Parquet encoding id
|
|
37
52
|
ENCODING_NAMES = {
|
|
38
53
|
plain: E::PLAIN, rle: E::RLE, delta_binary_packed: E::DELTA_BINARY_PACKED,
|
|
39
54
|
delta_length_byte_array: E::DELTA_LENGTH_BYTE_ARRAY, delta_byte_array: E::DELTA_BYTE_ARRAY,
|
|
40
55
|
byte_stream_split: E::BYTE_STREAM_SPLIT
|
|
41
56
|
}.freeze
|
|
42
57
|
|
|
58
|
+
# Encoding id => physical types it may be used for, per the Parquet encodings spec
|
|
43
59
|
VALID_ENCODINGS = {
|
|
44
60
|
E::PLAIN => T::NAMES.keys,
|
|
45
61
|
E::RLE => [T::BOOLEAN],
|
|
@@ -49,20 +65,47 @@ module Herringbone
|
|
|
49
65
|
E::BYTE_STREAM_SPLIT => [T::INT32, T::INT64, T::FLOAT, T::DOUBLE, T::FIXED_LEN_BYTE_ARRAY]
|
|
50
66
|
}.freeze
|
|
51
67
|
|
|
68
|
+
# Largest dictionary build_dictionary keeps; above it the chunk is written without a dictionary.
|
|
69
|
+
# Byte-array columns are dictionary-encoded by ByteValues, which applies its own
|
|
70
|
+
# ByteValues::MAX_DICTIONARY_BYTES as values arrive.
|
|
52
71
|
MAX_DICTIONARY_BYTES = 1024 * 1024
|
|
53
72
|
# The row group byte size is first estimated after this many rows, then after every row group
|
|
54
73
|
ESTIMATE_AFTER_ROWS = 1000
|
|
55
74
|
|
|
75
|
+
# @return [Schema] schema the rows are written with
|
|
56
76
|
attr_reader :schema
|
|
57
77
|
|
|
58
|
-
# Opens a writer on
|
|
59
|
-
#
|
|
60
|
-
|
|
61
|
-
|
|
78
|
+
# Opens a writer on +io+. With a block, the file is finished (footer written) when the block
|
|
79
|
+
# returns and the block's value is returned; if the block raises, the writer is aborted and no
|
|
80
|
+
# footer is written. Without a block, call #close to finish. The IO is never closed.
|
|
81
|
+
#
|
|
82
|
+
# @param io [IO, #write] destination
|
|
83
|
+
# @param schema [Schema] schema of the rows
|
|
84
|
+
# @param options [Hash{Symbol => Object}] see the class description and #initialize
|
|
85
|
+
# @option options [Symbol] :compression (:snappy) codec, see Herringbone.codecs
|
|
86
|
+
# @option options [Integer] :row_group_bytes (16MB) approximate buffered size that triggers a row group
|
|
87
|
+
# @option options [Integer, nil] :row_group_rows (nil) also flush a row group after this many rows
|
|
88
|
+
# @option options [Integer] :page_bytes (1MB) approximate uncompressed data page size
|
|
89
|
+
# @option options [Integer] :page_rows (20_000) maximum rows per data page
|
|
90
|
+
# @option options [Integer] :data_page_version (1) 1 or 2
|
|
91
|
+
# @option options [Boolean, Array<String>] :dictionary (true) dictionary-encode all eligible columns,
|
|
92
|
+
# none, or only the listed dotted column paths
|
|
93
|
+
# @option options [Hash{String => Symbol}] :encodings ({}) dotted column path => value encoding
|
|
94
|
+
# for non-dictionary pages
|
|
95
|
+
# @option options [Hash{String => String}] :metadata ({}) key/value metadata for the footer
|
|
96
|
+
# @option options [Boolean, Array<String>, Hash{String => Boolean, Hash}] :bloom_filters (nil)
|
|
97
|
+
# columns to write split block bloom filters for
|
|
98
|
+
# @yield [writer] the open writer
|
|
99
|
+
# @yieldparam writer [Writer] writer to append rows to
|
|
100
|
+
# @yieldreturn [Object] returned by open
|
|
101
|
+
# @return [Writer, Object] the writer without a block, the block's value with one
|
|
102
|
+
# @raise [ArgumentError] for an invalid IO, schema or option
|
|
103
|
+
def self.open(io, schema, **options)
|
|
104
|
+
writer = new(io, schema, **options)
|
|
62
105
|
return writer unless block_given?
|
|
63
106
|
begin
|
|
64
107
|
result = yield writer
|
|
65
|
-
rescue Exception # rubocop:disable Lint/RescueException -- also
|
|
108
|
+
rescue Exception # rubocop:disable Lint/RescueException -- also abort on Interrupt
|
|
66
109
|
writer.abort
|
|
67
110
|
raise
|
|
68
111
|
end
|
|
@@ -70,35 +113,72 @@ module Herringbone
|
|
|
70
113
|
result
|
|
71
114
|
end
|
|
72
115
|
|
|
73
|
-
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
|
|
116
|
+
# Validates the options and writes the leading magic bytes to +io+.
|
|
117
|
+
#
|
|
118
|
+
# @param io [IO, #write] destination; switched to binmode when it supports that
|
|
119
|
+
# @param schema [Schema] schema of the rows
|
|
120
|
+
# @param compression [Symbol, Integer] codec name (see Herringbone.codecs) or Format::Codec id
|
|
121
|
+
# @param row_group_bytes [Integer] approximate buffered size that triggers a row group
|
|
122
|
+
# @param row_group_rows [Integer, nil] also flush a row group after this many rows
|
|
123
|
+
# @param page_bytes [Integer] approximate uncompressed data page size
|
|
124
|
+
# @param page_rows [Integer] maximum level entries per data page (pages of repeated columns
|
|
125
|
+
# extend to the next row start)
|
|
126
|
+
# @param data_page_version [Integer] 1 or 2
|
|
127
|
+
# @param dictionary [Boolean, Array<String>] true for every column except BOOLEAN, FLOAT, DOUBLE
|
|
128
|
+
# and those listed in +encodings+; false for none; or the dotted paths of the columns to encode
|
|
129
|
+
# @param encodings [Hash{String, Symbol => Symbol, Integer}] dotted column path => value encoding
|
|
130
|
+
# (a name from ENCODING_NAMES or an encoding id) for non-dictionary pages
|
|
131
|
+
# @param metadata [Hash{#to_s => #to_s}] key/value metadata for the footer
|
|
132
|
+
# @param bloom_filters [Boolean, Array<String>, Hash{String => Boolean, Hash}, nil] see the class
|
|
133
|
+
# description
|
|
134
|
+
# @raise [ArgumentError] for an invalid IO, schema or option
|
|
135
|
+
# @raise [MissingCodecError] when the codec's optional gem cannot be loaded
|
|
136
|
+
# @raise [UnsupportedError] when the codec is not supported
|
|
137
|
+
def initialize(io, schema, compression: :snappy, row_group_bytes: 16 * 1024 * 1024, row_group_rows: nil,
|
|
138
|
+
page_bytes: 1024 * 1024, page_rows: 20_000, data_page_version: 1, dictionary: true, encodings: {},
|
|
139
|
+
metadata: {}, bloom_filters: nil)
|
|
140
|
+
raise ArgumentError, "Expected a Herringbone::Schema, got #{schema.class}" unless schema.is_a?(Schema)
|
|
141
|
+
@schema = schema
|
|
77
142
|
@codec = Compression.codec_id(compression)
|
|
143
|
+
# Fail before creating any file if the codec's library is missing
|
|
144
|
+
Compression.ensure_available!(@codec)
|
|
78
145
|
@row_group_bytes = Integer(row_group_bytes)
|
|
79
|
-
@
|
|
80
|
-
@row_limit = @
|
|
81
|
-
@
|
|
146
|
+
@row_group_rows = row_group_rows && Integer(row_group_rows)
|
|
147
|
+
@row_limit = @row_group_rows || ESTIMATE_AFTER_ROWS
|
|
148
|
+
@page_bytes = Integer(page_bytes)
|
|
149
|
+
@page_rows = Integer(page_rows)
|
|
150
|
+
raise ArgumentError, "page_rows must be positive" unless @page_rows.positive?
|
|
151
|
+
@page_indexes = [] # [ColumnChunk, ColumnIndex or nil, OffsetIndex] per written column chunk
|
|
82
152
|
@data_page_version = Integer(data_page_version)
|
|
83
153
|
raise ArgumentError, "data_page_version must be 1 or 2" unless [1, 2].include?(@data_page_version)
|
|
84
154
|
@dictionary = dictionary
|
|
85
155
|
@encodings = encodings.to_h { |path, enc| [path.to_s, encoding_id(path, enc)] }
|
|
86
|
-
|
|
156
|
+
columns = schema.columns.to_h { |c| [c.dotted_path, c] }
|
|
157
|
+
unknown = @encodings.keys - columns.keys
|
|
87
158
|
raise ArgumentError, "encodings: no such column #{unknown.join(", ")}" unless unknown.empty?
|
|
88
|
-
@
|
|
159
|
+
@encodings.each { |path, enc| check_encoding!(columns[path], enc) }
|
|
89
160
|
@metadata = metadata
|
|
161
|
+
@bloom_filters = bloom_filter_config(bloom_filters)
|
|
162
|
+
@pending_bloom_filters = [] # [ColumnMetaData, BloomFilter] for the row group being written
|
|
90
163
|
@row_groups = []
|
|
91
164
|
@total_rows = 0
|
|
92
165
|
@pos = 0
|
|
93
166
|
@closed = false
|
|
94
167
|
@bytes_per_row = nil
|
|
95
|
-
|
|
168
|
+
@io = check_io!(io)
|
|
96
169
|
write_raw(MAGIC)
|
|
97
170
|
reset_buffers
|
|
98
171
|
end
|
|
99
172
|
|
|
100
173
|
# Appends a row: a Hash keyed by top-level field names (Strings or Symbols), an Array of values
|
|
101
|
-
# in schema order, or an object responding to #attributes (ActiveRecord) or #to_h (Struct, Data)
|
|
174
|
+
# in schema order, or an object responding to #attributes (ActiveRecord) or #to_h (Struct, Data).
|
|
175
|
+
# A row that fails to encode leaves nothing behind in the buffers. Flushes a row group when the
|
|
176
|
+
# buffered rows reach the size limits.
|
|
177
|
+
#
|
|
178
|
+
# @param row [Hash, Array, #attributes, #to_h] row to append
|
|
179
|
+
# @return [self]
|
|
180
|
+
# @raise [Error] when the writer is closed
|
|
181
|
+
# @raise [EncodeError] when a required field is nil or missing, or a value cannot be encoded
|
|
102
182
|
def <<(row)
|
|
103
183
|
raise Error, "Writer is closed" if @closed
|
|
104
184
|
row = row_hash(row)
|
|
@@ -123,7 +203,7 @@ module Herringbone
|
|
|
123
203
|
rescue EncodeError => e
|
|
124
204
|
rollback_row(marks)
|
|
125
205
|
raise EncodeError, "Row #{@total_rows + @buffered_rows}: #{e.message}"
|
|
126
|
-
rescue
|
|
206
|
+
rescue
|
|
127
207
|
rollback_row(marks)
|
|
128
208
|
raise
|
|
129
209
|
end
|
|
@@ -131,17 +211,15 @@ module Herringbone
|
|
|
131
211
|
check_row_group_size if @buffered_rows >= @row_limit
|
|
132
212
|
self
|
|
133
213
|
end
|
|
134
|
-
alias_method :write, :<<
|
|
135
|
-
|
|
136
|
-
def write_rows(rows)
|
|
137
|
-
rows.each { |row| self << row }
|
|
138
|
-
self
|
|
139
|
-
end
|
|
140
214
|
|
|
141
215
|
# Number of rows written so far, including buffered ones
|
|
216
|
+
#
|
|
217
|
+
# @return [Integer]
|
|
142
218
|
def rows_written = @total_rows + @buffered_rows
|
|
143
219
|
|
|
144
220
|
# Writes any buffered rows as a row group
|
|
221
|
+
#
|
|
222
|
+
# @return [void]
|
|
145
223
|
def flush_row_group
|
|
146
224
|
return if @buffered_rows.zero?
|
|
147
225
|
start = @pos
|
|
@@ -154,20 +232,26 @@ module Herringbone
|
|
|
154
232
|
total_compressed_size: @pos - start,
|
|
155
233
|
ordinal: @row_groups.size
|
|
156
234
|
)
|
|
235
|
+
write_bloom_filters
|
|
157
236
|
@total_rows += @buffered_rows
|
|
158
237
|
reset_buffers
|
|
159
238
|
end
|
|
160
239
|
|
|
240
|
+
# Flushes buffered rows, then writes the page indexes and the footer and flushes the IO.
|
|
241
|
+
# Does nothing when already closed or aborted. The IO itself is not closed.
|
|
242
|
+
#
|
|
243
|
+
# @return [void]
|
|
161
244
|
def close
|
|
162
245
|
return if @closed
|
|
163
246
|
flush_row_group
|
|
247
|
+
write_page_indexes
|
|
164
248
|
meta = Format::FileMetaData.new(
|
|
165
249
|
version: 2,
|
|
166
250
|
schema: @schema.to_elements,
|
|
167
251
|
num_rows: @total_rows,
|
|
168
252
|
row_groups: @row_groups,
|
|
169
253
|
key_value_metadata: @metadata.empty? ? nil : @metadata.map { |k, v| Format::KeyValue.new(key: k.to_s, value: v&.to_s) },
|
|
170
|
-
created_by: "herringbone
|
|
254
|
+
created_by: "herringbone-ruby #{VERSION}",
|
|
171
255
|
column_orders: @schema.columns.map { Format::ColumnOrder.new(type_order: Format::TypeDefinedOrder.new) }
|
|
172
256
|
)
|
|
173
257
|
footer = meta.encode
|
|
@@ -176,24 +260,51 @@ module Herringbone
|
|
|
176
260
|
write_raw(MAGIC)
|
|
177
261
|
@io.flush if @io.respond_to?(:flush)
|
|
178
262
|
@closed = true
|
|
179
|
-
finish_target
|
|
180
263
|
end
|
|
181
264
|
|
|
182
|
-
# Stops writing without
|
|
265
|
+
# Stops writing without finishing the file (no footer is written). Whatever was already written
|
|
266
|
+
# to the IO stays there; discarding it is up to the caller.
|
|
267
|
+
#
|
|
268
|
+
# @return [void]
|
|
183
269
|
def abort
|
|
184
|
-
return if @closed
|
|
185
270
|
@closed = true
|
|
186
|
-
return unless @temp_path
|
|
187
|
-
@io.close unless @io.closed?
|
|
188
|
-
File.unlink(@temp_path) if File.exist?(@temp_path)
|
|
189
271
|
end
|
|
190
272
|
|
|
191
273
|
private
|
|
192
274
|
|
|
275
|
+
# Pathname responds to #write too (writing a whole file by path), so it is rejected explicitly.
|
|
276
|
+
# A text-mode IO (a pipe, or a File opened with "w") transcodes what is written when
|
|
277
|
+
# Encoding.default_internal is set, as Rails does, and binary pages cannot be transcoded,
|
|
278
|
+
# so the IO is switched to binary mode.
|
|
279
|
+
#
|
|
280
|
+
# @param io [Object] candidate destination
|
|
281
|
+
# @return [IO, #write] +io+, in binary mode when it supports #binmode
|
|
282
|
+
# @raise [ArgumentError] when +io+ does not respond to #write, or is a String or Pathname
|
|
283
|
+
def check_io!(io)
|
|
284
|
+
if io.respond_to?(:write) && !io.is_a?(String) && !(defined?(Pathname) && io.is_a?(Pathname))
|
|
285
|
+
io.binmode if io.respond_to?(:binmode)
|
|
286
|
+
return io
|
|
287
|
+
end
|
|
288
|
+
raise ArgumentError, "Herringbone::Writer expects an IO that responds to #write " \
|
|
289
|
+
"(e.g. File.open(path, \"wb\") or StringIO.new), got #{io.class}"
|
|
290
|
+
end
|
|
291
|
+
|
|
193
292
|
# Levels are kept as binary Strings (one byte per entry). Values are an Array for numeric and
|
|
194
293
|
# boolean columns, and a compact ByteValues for BYTE_ARRAY / FIXED_LEN_BYTE_ARRAY columns.
|
|
294
|
+
#
|
|
295
|
+
# @!attribute defs
|
|
296
|
+
# @return [String, Array<Integer>] definition levels (unpacked to an Array while a chunk is written)
|
|
297
|
+
# @!attribute reps
|
|
298
|
+
# @return [String, Array<Integer>, nil] repetition levels, like +defs+; nil for non-repeated columns
|
|
299
|
+
# @!attribute values
|
|
300
|
+
# @return [Array, ByteValues] non-null physical values
|
|
195
301
|
ColumnBuffer = Struct.new(:defs, :reps, :values)
|
|
196
302
|
|
|
303
|
+
# Starts a new row group: fresh buffers per column, and the per-field write plan
|
|
304
|
+
# (field, name, Symbol name, buffer and encoder for flat fields, or nils for shredded ones).
|
|
305
|
+
#
|
|
306
|
+
# @return [void]
|
|
307
|
+
# @raise [UnsupportedError] when a column has more than 255 definition or repetition levels
|
|
197
308
|
def reset_buffers
|
|
198
309
|
@buffers = @schema.columns.map do |col|
|
|
199
310
|
if col.max_definition_level > 255 || col.max_repetition_level > 255
|
|
@@ -212,6 +323,8 @@ module Herringbone
|
|
|
212
323
|
@buffered_rows = 0
|
|
213
324
|
end
|
|
214
325
|
|
|
326
|
+
# @param col [Schema::Column] column to buffer
|
|
327
|
+
# @return [ByteValues, Array] empty value store for the column
|
|
215
328
|
def new_values_store(col)
|
|
216
329
|
case col.type
|
|
217
330
|
when T::BYTE_ARRAY then ByteValues.new(dictionary: use_dictionary?(col))
|
|
@@ -220,25 +333,10 @@ module Herringbone
|
|
|
220
333
|
end
|
|
221
334
|
end
|
|
222
335
|
|
|
223
|
-
|
|
224
|
-
if target.respond_to?(:write)
|
|
225
|
-
@io = target
|
|
226
|
-
else
|
|
227
|
-
@path = target.to_s
|
|
228
|
-
dir = File.dirname(@path)
|
|
229
|
-
@temp_path = File.join(dir, ".#{File.basename(@path)}.#{Process.pid}.#{rand(1 << 32).to_s(36)}.tmp")
|
|
230
|
-
@io = File.open(@temp_path, "wb")
|
|
231
|
-
end
|
|
232
|
-
end
|
|
233
|
-
|
|
234
|
-
def finish_target
|
|
235
|
-
return unless @temp_path
|
|
236
|
-
@io.close
|
|
237
|
-
File.rename(@temp_path, @path)
|
|
238
|
-
end
|
|
239
|
-
|
|
240
|
-
# Flushes when the buffered values reach row_group_bytes (or row_group_size rows). The bytes per
|
|
336
|
+
# Flushes when the buffered values reach row_group_bytes (or row_group_rows rows). The bytes per
|
|
241
337
|
# row are estimated from the buffered values after the first rows, then refreshed per row group.
|
|
338
|
+
#
|
|
339
|
+
# @return [void]
|
|
242
340
|
def check_row_group_size
|
|
243
341
|
@bytes_per_row ||= estimate_bytes_per_row
|
|
244
342
|
limit = row_limit_for(@bytes_per_row)
|
|
@@ -251,11 +349,15 @@ module Herringbone
|
|
|
251
349
|
end
|
|
252
350
|
end
|
|
253
351
|
|
|
352
|
+
# @param bytes_per_row [Integer] estimated buffered bytes per row
|
|
353
|
+
# @return [Integer] rows to buffer before the next size check: what fits in row_group_bytes,
|
|
354
|
+
# at least ESTIMATE_AFTER_ROWS, at most row_group_rows
|
|
254
355
|
def row_limit_for(bytes_per_row)
|
|
255
356
|
by_bytes = [@row_group_bytes / [bytes_per_row, 1].max, ESTIMATE_AFTER_ROWS].max
|
|
256
|
-
@
|
|
357
|
+
@row_group_rows ? [@row_group_rows, by_bytes].min : by_bytes
|
|
257
358
|
end
|
|
258
359
|
|
|
360
|
+
# @return [Integer] memory held by the buffers divided by the buffered rows
|
|
259
361
|
def estimate_bytes_per_row
|
|
260
362
|
bytes = @schema.columns.sum do |col|
|
|
261
363
|
buf = @buffers[col.index]
|
|
@@ -263,29 +365,58 @@ module Herringbone
|
|
|
263
365
|
held = if values.is_a?(ByteValues)
|
|
264
366
|
values.memory_bytes
|
|
265
367
|
else
|
|
266
|
-
values.size * (col.type == T::INT96 ? 48 : 8)
|
|
368
|
+
values.size * ((col.type == T::INT96) ? 48 : 8)
|
|
267
369
|
end
|
|
268
370
|
held + buf.defs.bytesize + (buf.reps ? buf.reps.bytesize : 0)
|
|
269
371
|
end
|
|
270
372
|
bytes / [@buffered_rows, 1].max
|
|
271
373
|
end
|
|
272
374
|
|
|
375
|
+
# Writes to the IO and tracks the file offset, since the IO is never asked for its position
|
|
376
|
+
#
|
|
377
|
+
# @param bytes [String] binary data
|
|
378
|
+
# @return [void]
|
|
273
379
|
def write_raw(bytes)
|
|
274
380
|
@io.write(bytes)
|
|
275
381
|
@pos += bytes.bytesize
|
|
276
382
|
end
|
|
277
383
|
|
|
384
|
+
# @param path [String, Symbol] column path, for the error message
|
|
385
|
+
# @param enc [Symbol, String, Integer] encoding name from ENCODING_NAMES, or an encoding id
|
|
386
|
+
# @return [Integer] encoding id
|
|
387
|
+
# @raise [ArgumentError] for an unknown encoding name, or an id that is not in VALID_ENCODINGS
|
|
278
388
|
def encoding_id(path, enc)
|
|
279
|
-
|
|
389
|
+
if enc.is_a?(Integer)
|
|
390
|
+
return enc if VALID_ENCODINGS.key?(enc)
|
|
391
|
+
raise ArgumentError, "Encoding #{E::NAMES.fetch(enc, enc)} cannot be requested for #{path}"
|
|
392
|
+
end
|
|
280
393
|
ENCODING_NAMES.fetch(enc.to_s.downcase.to_sym) { raise ArgumentError, "Unknown encoding #{enc.inspect} for #{path}" }
|
|
281
394
|
end
|
|
282
395
|
|
|
396
|
+
# @param col [Schema::Column] column the encoding is requested for
|
|
397
|
+
# @param enc [Integer] encoding id, a key of VALID_ENCODINGS
|
|
398
|
+
# @return [void]
|
|
399
|
+
# @raise [ArgumentError] when the encoding is not valid for the column's physical type
|
|
400
|
+
def check_encoding!(col, enc)
|
|
401
|
+
return if VALID_ENCODINGS.fetch(enc).include?(col.type)
|
|
402
|
+
raise ArgumentError, "Encoding #{E::NAMES[enc]} is not valid for #{T::NAMES[col.type]} column #{col.dotted_path}"
|
|
403
|
+
end
|
|
404
|
+
|
|
405
|
+
# Reads a struct member, by String or Symbol key
|
|
406
|
+
#
|
|
407
|
+
# @param hash [Hash, #to_h, nil] struct value
|
|
408
|
+
# @param name [String] member name
|
|
409
|
+
# @return [Object, nil] member value, nil when +hash+ is nil or has no such key
|
|
410
|
+
# @raise [EncodeError] when +hash+ cannot be converted to a Hash
|
|
283
411
|
def lookup(hash, name)
|
|
284
412
|
return nil if hash.nil?
|
|
285
413
|
hash = as_hash(hash, name) unless hash.is_a?(Hash)
|
|
286
414
|
hash.fetch(name) { hash[name.to_sym] }
|
|
287
415
|
end
|
|
288
416
|
|
|
417
|
+
# @param row [Hash, Array, #attributes, #to_h] row given to #<<
|
|
418
|
+
# @return [Hash] the row keyed by top-level field names
|
|
419
|
+
# @raise [EncodeError] when an Array has the wrong size or the row cannot be converted to a Hash
|
|
289
420
|
def row_hash(row)
|
|
290
421
|
case row
|
|
291
422
|
when Hash then row
|
|
@@ -300,6 +431,10 @@ module Herringbone
|
|
|
300
431
|
end
|
|
301
432
|
end
|
|
302
433
|
|
|
434
|
+
# @param value [Hash, #to_h] value to convert
|
|
435
|
+
# @param what [String] description of the value, for the error message
|
|
436
|
+
# @return [Hash]
|
|
437
|
+
# @raise [EncodeError] when +value+ does not respond to #to_h
|
|
303
438
|
def as_hash(value, what)
|
|
304
439
|
return value if value.is_a?(Hash)
|
|
305
440
|
raise EncodeError, "Expected a Hash for #{what}, got #{value.class}" unless value.respond_to?(:to_h)
|
|
@@ -307,6 +442,10 @@ module Herringbone
|
|
|
307
442
|
end
|
|
308
443
|
|
|
309
444
|
# Removes the entries a failed row left behind, so the buffers stay aligned
|
|
445
|
+
#
|
|
446
|
+
# @param marks [Array<Array(Integer, Integer, Integer)>, nil] sizes of defs, reps and values of
|
|
447
|
+
# each nested buffer before the row, nil when there are none
|
|
448
|
+
# @return [void]
|
|
310
449
|
def rollback_row(marks)
|
|
311
450
|
@plan.each do |field, _, _, buf|
|
|
312
451
|
next unless buf && buf.defs.bytesize > @buffered_rows
|
|
@@ -323,6 +462,14 @@ module Herringbone
|
|
|
323
462
|
end
|
|
324
463
|
|
|
325
464
|
# Record shredding: turns a nested value into (definition level, repetition level, value) entries
|
|
465
|
+
#
|
|
466
|
+
# @param field [Schema::Field] field the value belongs to
|
|
467
|
+
# @param value [Object, nil] Ruby value of the field
|
|
468
|
+
# @param parent_def [Integer] definition level recorded when +value+ is nil
|
|
469
|
+
# @param rep [Integer] repetition level of the first entry this value produces
|
|
470
|
+
# @return [void]
|
|
471
|
+
# @raise [EncodeError] when a required value is nil, a list or map value has the wrong type, a
|
|
472
|
+
# map key is nil, or a leaf value cannot be encoded
|
|
326
473
|
def shred(field, value, parent_def, rep)
|
|
327
474
|
if value.nil?
|
|
328
475
|
raise EncodeError, "Field #{field.node.path.join(".")} is required but got nil" unless field.optional
|
|
@@ -379,6 +526,13 @@ module Herringbone
|
|
|
379
526
|
end
|
|
380
527
|
end
|
|
381
528
|
|
|
529
|
+
# Writes one column chunk of the current row group: an optional dictionary page, then data pages.
|
|
530
|
+
# Bloom filters and page indexes are queued, to be written after the row group and before the
|
|
531
|
+
# footer.
|
|
532
|
+
#
|
|
533
|
+
# @param col [Schema::Column] column being written
|
|
534
|
+
# @param buffer [ColumnBuffer] the column's buffered levels and values
|
|
535
|
+
# @return [Format::ColumnChunk] chunk with its ColumnMetaData, for the row group
|
|
382
536
|
def write_column_chunk(col, buffer)
|
|
383
537
|
type = col.type
|
|
384
538
|
path = col.dotted_path
|
|
@@ -399,11 +553,7 @@ module Herringbone
|
|
|
399
553
|
value_encoding = if dict_values
|
|
400
554
|
E::RLE_DICTIONARY
|
|
401
555
|
else
|
|
402
|
-
|
|
403
|
-
unless VALID_ENCODINGS.fetch(enc).include?(type)
|
|
404
|
-
raise ArgumentError, "Encoding #{E::NAMES[enc]} is not valid for #{T::NAMES[type]} column #{path}"
|
|
405
|
-
end
|
|
406
|
-
enc
|
|
556
|
+
@encodings[path] || E::PLAIN
|
|
407
557
|
end
|
|
408
558
|
|
|
409
559
|
chunk_start = @pos
|
|
@@ -430,6 +580,9 @@ module Herringbone
|
|
|
430
580
|
else
|
|
431
581
|
values.sum(&:bytesize) + 4 * values.size
|
|
432
582
|
end
|
|
583
|
+
order = sort_key(col)
|
|
584
|
+
pages = []
|
|
585
|
+
first_row = 0
|
|
433
586
|
page_ranges(buf, value_bytes).each do |from, to|
|
|
434
587
|
n = to - from
|
|
435
588
|
defs = buf.defs[from, n]
|
|
@@ -437,6 +590,12 @@ module Herringbone
|
|
|
437
590
|
non_null = max_def.zero? ? n : defs.count(max_def)
|
|
438
591
|
page_values = (indices || values)[value_index, non_null]
|
|
439
592
|
value_index += non_null
|
|
593
|
+
page_offset = @pos
|
|
594
|
+
range = nil
|
|
595
|
+
if order && non_null.positive?
|
|
596
|
+
in_page = dict_values ? page_values.uniq.map { |i| dict_values[i] } : page_values
|
|
597
|
+
range = value_range(col, in_page, order)
|
|
598
|
+
end
|
|
440
599
|
encoded = encode_values(page_values, value_encoding, type, col.type_length, dict_values&.size)
|
|
441
600
|
rep_bytes = max_rep.positive? ? Encodings::RLE.encode_hybrid(reps, max_rep.bit_length) : "".b
|
|
442
601
|
def_bytes = max_def.positive? ? Encodings::RLE.encode_hybrid(defs, max_def.bit_length) : "".b
|
|
@@ -446,6 +605,8 @@ module Herringbone
|
|
|
446
605
|
num_rows = max_rep.zero? ? n : reps.count(0)
|
|
447
606
|
write_data_page_v2(n, n - non_null, num_rows, rep_bytes, def_bytes, encoded, value_encoding)
|
|
448
607
|
end
|
|
608
|
+
pages << PageInfo.new(page_offset, @pos - page_offset, first_row, n - non_null, non_null, range)
|
|
609
|
+
first_row += max_rep.zero? ? n : reps.count(0)
|
|
449
610
|
end
|
|
450
611
|
|
|
451
612
|
encodings = [E::RLE]
|
|
@@ -461,11 +622,178 @@ module Herringbone
|
|
|
461
622
|
total_compressed_size: @pos - chunk_start,
|
|
462
623
|
data_page_offset: data_offset,
|
|
463
624
|
dictionary_page_offset: dictionary_offset,
|
|
464
|
-
statistics:
|
|
625
|
+
statistics: statistics_for(col, buf.defs, values || indices.uniq.map { |i| dict_values[i] }, order)
|
|
626
|
+
)
|
|
627
|
+
chunk = Format::ColumnChunk.new(file_offset: chunk_start, meta_data: meta)
|
|
628
|
+
if (bloom = @bloom_filters[path])
|
|
629
|
+
@pending_bloom_filters << [meta, build_bloom_filter(col, bloom, dict_values, values)]
|
|
630
|
+
end
|
|
631
|
+
@page_indexes << [chunk, column_index_for(col, pages, order), offset_index_for(pages)]
|
|
632
|
+
chunk
|
|
633
|
+
end
|
|
634
|
+
|
|
635
|
+
# What the page indexes need to know about a written data page
|
|
636
|
+
#
|
|
637
|
+
# @!attribute offset
|
|
638
|
+
# @return [Integer] file offset of the page header
|
|
639
|
+
# @!attribute size
|
|
640
|
+
# @return [Integer] page size in the file, header included
|
|
641
|
+
# @!attribute first_row
|
|
642
|
+
# @return [Integer] index of the page's first row within the row group
|
|
643
|
+
# @!attribute nulls
|
|
644
|
+
# @return [Integer] null entries in the page
|
|
645
|
+
# @!attribute non_null
|
|
646
|
+
# @return [Integer] non-null values in the page
|
|
647
|
+
# @!attribute range
|
|
648
|
+
# @return [Array(Object, Object), nil] [min, max] physical values, nil when there are none
|
|
649
|
+
PageInfo = Struct.new(:offset, :size, :first_row, :nulls, :non_null, :range)
|
|
650
|
+
|
|
651
|
+
# @param pages [Array<PageInfo>] pages of a column chunk
|
|
652
|
+
# @return [Format::OffsetIndex]
|
|
653
|
+
def offset_index_for(pages)
|
|
654
|
+
Format::OffsetIndex.new(page_locations: pages.map do |p|
|
|
655
|
+
Format::PageLocation.new(offset: p.offset, compressed_page_size: p.size, first_row_index: p.first_row)
|
|
656
|
+
end)
|
|
657
|
+
end
|
|
658
|
+
|
|
659
|
+
# nil when the column has no defined sort order, or a page's values have no min/max (all NaN)
|
|
660
|
+
#
|
|
661
|
+
# @param col [Schema::Column] column the pages belong to
|
|
662
|
+
# @param pages [Array<PageInfo>] pages of the column chunk
|
|
663
|
+
# @param order [Proc, Method, nil] sort key from #sort_key
|
|
664
|
+
# @return [Format::ColumnIndex, nil]
|
|
665
|
+
def column_index_for(col, pages, order)
|
|
666
|
+
return nil unless order
|
|
667
|
+
return nil if pages.any? { |p| p.non_null.positive? && p.range.nil? }
|
|
668
|
+
Format::ColumnIndex.new(
|
|
669
|
+
null_pages: pages.map { |p| p.non_null.zero? },
|
|
670
|
+
min_values: pages.map { |p| p.range ? truncate_min(stat_bytes(col, p.range[0])) : "".b },
|
|
671
|
+
max_values: pages.map { |p| p.range ? truncate_max(stat_bytes(col, p.range[1])) : "".b },
|
|
672
|
+
boundary_order: boundary_order(pages.filter_map(&:range), order),
|
|
673
|
+
null_counts: pages.map(&:nulls)
|
|
465
674
|
)
|
|
466
|
-
Format::ColumnChunk.new(file_offset: chunk_start, meta_data: meta)
|
|
467
675
|
end
|
|
468
676
|
|
|
677
|
+
# @param ranges [Array<Array(Object, Object)>] [min, max] of each non-null page, in page order
|
|
678
|
+
# @param order [Proc, Method] sort key from #sort_key
|
|
679
|
+
# @return [Integer] Format::BoundaryOrder: ASCENDING when both mins and maxes never decrease
|
|
680
|
+
# (or there are fewer than two pages), DESCENDING when they never increase, else UNORDERED
|
|
681
|
+
def boundary_order(ranges, order)
|
|
682
|
+
return Format::BoundaryOrder::ASCENDING if ranges.size < 2
|
|
683
|
+
cmp = ->(a, b) { order.equal?(IDENTITY) ? a <=> b : order.call(a) <=> order.call(b) }
|
|
684
|
+
pairs = ranges.each_cons(2)
|
|
685
|
+
if pairs.all? { |(a_min, a_max), (b_min, b_max)| cmp.call(a_min, b_min) <= 0 && cmp.call(a_max, b_max) <= 0 }
|
|
686
|
+
Format::BoundaryOrder::ASCENDING
|
|
687
|
+
elsif pairs.all? { |(a_min, a_max), (b_min, b_max)| cmp.call(a_min, b_min) >= 0 && cmp.call(a_max, b_max) >= 0 }
|
|
688
|
+
Format::BoundaryOrder::DESCENDING
|
|
689
|
+
else
|
|
690
|
+
Format::BoundaryOrder::UNORDERED
|
|
691
|
+
end
|
|
692
|
+
end
|
|
693
|
+
|
|
694
|
+
# Page indexes go after the last row group: all column indexes, then all offset indexes
|
|
695
|
+
#
|
|
696
|
+
# @return [void]
|
|
697
|
+
def write_page_indexes
|
|
698
|
+
@page_indexes.each do |chunk, column_index, _|
|
|
699
|
+
next unless column_index
|
|
700
|
+
bytes = column_index.encode
|
|
701
|
+
chunk.column_index_offset = @pos
|
|
702
|
+
chunk.column_index_length = bytes.bytesize
|
|
703
|
+
write_raw(bytes)
|
|
704
|
+
end
|
|
705
|
+
@page_indexes.each do |chunk, _, offset_index|
|
|
706
|
+
bytes = offset_index.encode
|
|
707
|
+
chunk.offset_index_offset = @pos
|
|
708
|
+
chunk.offset_index_length = bytes.bytesize
|
|
709
|
+
write_raw(bytes)
|
|
710
|
+
end
|
|
711
|
+
@page_indexes.clear
|
|
712
|
+
end
|
|
713
|
+
|
|
714
|
+
# Per-column settings accepted in the +bloom_filters:+ option
|
|
715
|
+
BLOOM_FILTER_OPTIONS = %i[ndv fpp max_bytes].freeze
|
|
716
|
+
|
|
717
|
+
# { dotted_path => { ndv:, fpp:, max_bytes: } } from the bloom_filters: option
|
|
718
|
+
#
|
|
719
|
+
# @param requested [Boolean, Array<String>, Hash{String => Boolean, Hash, nil}, nil] true for every
|
|
720
|
+
# column of a supported type, column paths, or column path => true / false / settings Hash.
|
|
721
|
+
# A path may also be given as an Array of names. Settings are +ndv+, +fpp+ and +max_bytes+.
|
|
722
|
+
# @return [Hash{String => Hash{Symbol => Numeric, nil}}] settings by dotted path; empty when disabled
|
|
723
|
+
# @raise [ArgumentError] for an unknown column, a column type without bloom filter support, or
|
|
724
|
+
# invalid settings
|
|
725
|
+
def bloom_filter_config(requested)
|
|
726
|
+
return {} if requested.nil? || requested == false
|
|
727
|
+
columns = @schema.columns.to_h { |c| [c.dotted_path, c] }
|
|
728
|
+
if requested == true
|
|
729
|
+
return columns.select { |_, c| BloomFilter::TYPES.include?(c.type) }.transform_values { bloom_filter_settings({}) }
|
|
730
|
+
end
|
|
731
|
+
requested = Array(requested).to_h { |path| [path, true] } unless requested.is_a?(Hash)
|
|
732
|
+
requested.each_with_object({}) do |(path, settings), config|
|
|
733
|
+
path = path.is_a?(Array) ? path.join(".") : path.to_s
|
|
734
|
+
col = columns[path] or raise ArgumentError, "bloom_filters: no such column #{path}"
|
|
735
|
+
next if settings.nil? || settings == false
|
|
736
|
+
unless BloomFilter::TYPES.include?(col.type)
|
|
737
|
+
raise ArgumentError, "bloom_filters: not supported for #{T::NAMES[col.type]} column #{path}"
|
|
738
|
+
end
|
|
739
|
+
settings = {} if settings == true
|
|
740
|
+
raise ArgumentError, "bloom_filters: expected true or a Hash for #{path}, got #{settings.inspect}" unless settings.is_a?(Hash)
|
|
741
|
+
config[path] = bloom_filter_settings(settings.transform_keys(&:to_sym), path)
|
|
742
|
+
end
|
|
743
|
+
end
|
|
744
|
+
|
|
745
|
+
# @param settings [Hash{Symbol => Object}] +ndv+, +fpp+ and +max_bytes+, all optional
|
|
746
|
+
# @param path [String, nil] column path, for error messages
|
|
747
|
+
# @return [Hash{Symbol => Numeric, nil}] +ndv+ (nil to count distinct values), +fpp+ and
|
|
748
|
+
# +max_bytes+ with defaults filled in
|
|
749
|
+
# @raise [ArgumentError] for unknown keys, a non-positive ndv, or an fpp outside (0, 1)
|
|
750
|
+
def bloom_filter_settings(settings, path = nil)
|
|
751
|
+
unknown = settings.keys - BLOOM_FILTER_OPTIONS
|
|
752
|
+
raise ArgumentError, "bloom_filters: unknown option #{unknown.join(", ")} for #{path}" unless unknown.empty?
|
|
753
|
+
ndv = settings[:ndv] && Integer(settings[:ndv])
|
|
754
|
+
raise ArgumentError, "bloom_filters: ndv must be positive for #{path}" if ndv && !ndv.positive?
|
|
755
|
+
fpp = Float(settings.fetch(:fpp, BloomFilter::DEFAULT_FPP))
|
|
756
|
+
raise ArgumentError, "bloom_filters: fpp must be between 0 and 1 for #{path}" unless fpp > 0 && fpp < 1
|
|
757
|
+
{ndv: ndv, fpp: fpp, max_bytes: Integer(settings.fetch(:max_bytes, BloomFilter::DEFAULT_MAX_BYTES))}
|
|
758
|
+
end
|
|
759
|
+
|
|
760
|
+
# A filter holding the chunk's values: +dict_values+ (already distinct) for dictionary-encoded
|
|
761
|
+
# chunks, +values+ otherwise. Each distinct value is hashed once, and the filter is sized from
|
|
762
|
+
# the configured ndv or from the number of distinct values.
|
|
763
|
+
#
|
|
764
|
+
# @param col [Schema::Column] column the filter is for
|
|
765
|
+
# @param settings [Hash{Symbol => Numeric, nil}] from #bloom_filter_settings
|
|
766
|
+
# @param dict_values [Array, nil] dictionary of the chunk, when dictionary-encoded
|
|
767
|
+
# @param values [Array, nil] the chunk's non-null physical values, used without a dictionary
|
|
768
|
+
# @return [BloomFilter]
|
|
769
|
+
def build_bloom_filter(col, settings, dict_values, values)
|
|
770
|
+
hashes = if dict_values
|
|
771
|
+
BloomFilter.hash_physical_all(dict_values, col.type)
|
|
772
|
+
else
|
|
773
|
+
BloomFilter.hash_physical_all(values, col.type, distinct: true)
|
|
774
|
+
end
|
|
775
|
+
size = BloomFilter.optimal_num_bytes(settings[:ndv] || hashes.size, settings[:fpp], max_bytes: settings[:max_bytes])
|
|
776
|
+
filter = BloomFilter.new(size, column: col)
|
|
777
|
+
filter.insert_hashes(hashes)
|
|
778
|
+
filter
|
|
779
|
+
end
|
|
780
|
+
|
|
781
|
+
# Bloom filters go right after the row group's column chunks, in column order
|
|
782
|
+
#
|
|
783
|
+
# @return [void]
|
|
784
|
+
def write_bloom_filters
|
|
785
|
+
@pending_bloom_filters.each do |meta, filter|
|
|
786
|
+
bytes = filter.encode
|
|
787
|
+
meta.bloom_filter_offset = @pos
|
|
788
|
+
meta.bloom_filter_length = bytes.bytesize
|
|
789
|
+
write_raw(bytes)
|
|
790
|
+
end
|
|
791
|
+
@pending_bloom_filters.clear
|
|
792
|
+
end
|
|
793
|
+
|
|
794
|
+
# @param col [Schema::Column] column to check
|
|
795
|
+
# @return [Boolean] whether the +dictionary:+ option asks for this column to be dictionary-encoded
|
|
796
|
+
# (never for BOOLEAN)
|
|
469
797
|
def use_dictionary?(col)
|
|
470
798
|
return false if col.type == T::BOOLEAN
|
|
471
799
|
case @dictionary
|
|
@@ -475,13 +803,20 @@ module Herringbone
|
|
|
475
803
|
end
|
|
476
804
|
end
|
|
477
805
|
|
|
478
|
-
# Returns [dictionary_values, indices] or nil when a dictionary is not worthwhile
|
|
806
|
+
# Returns [dictionary_values, indices] or nil when a dictionary is not worthwhile: more than about
|
|
807
|
+
# half the values are distinct, or the dictionary exceeds MAX_DICTIONARY_BYTES
|
|
808
|
+
#
|
|
809
|
+
# @param values [Array] non-null physical values of the chunk
|
|
810
|
+
# @param type [Integer] physical type
|
|
811
|
+
# @param type_length [Integer, nil] FIXED_LEN_BYTE_ARRAY width
|
|
812
|
+
# @return [Array(Array, Array<Integer>), nil]
|
|
479
813
|
def build_dictionary(values, type, type_length)
|
|
480
814
|
# Floats are keyed by bit pattern so that -0.0 and 0.0 (and NaNs) stay distinct
|
|
481
815
|
if type == T::FLOAT || type == T::DOUBLE
|
|
482
816
|
keys = values.pack("G*").unpack("Q>*")
|
|
483
817
|
uniq = keys.uniq
|
|
484
818
|
return nil if uniq.size > values.size / 2 + 1 && values.size > 16
|
|
819
|
+
return nil if uniq.size * 8 > MAX_DICTIONARY_BYTES
|
|
485
820
|
index = uniq.each_with_index.to_h
|
|
486
821
|
return [uniq.pack("Q>*").unpack("G*"), keys.map(&index)]
|
|
487
822
|
end
|
|
@@ -497,14 +832,19 @@ module Herringbone
|
|
|
497
832
|
[dict, values.map(&index)]
|
|
498
833
|
end
|
|
499
834
|
|
|
500
|
-
# Splits a column buffer into pages of roughly @
|
|
835
|
+
# Splits a column buffer into pages of roughly @page_bytes bytes. Repeated columns
|
|
501
836
|
# are only cut where a new row starts.
|
|
837
|
+
#
|
|
838
|
+
# @param buf [ColumnBuffer] column buffer with levels as Arrays
|
|
839
|
+
# @param value_bytes [Integer] estimated encoded size of all the chunk's values
|
|
840
|
+
# @return [Array<Array(Integer, Integer)>] [from, to) ranges of level entries, one per page
|
|
502
841
|
def page_ranges(buf, value_bytes)
|
|
503
842
|
n = buf.defs.size
|
|
504
843
|
bytes = n + value_bytes
|
|
505
|
-
pages = (bytes + @
|
|
506
|
-
|
|
507
|
-
per =
|
|
844
|
+
pages = (bytes + @page_bytes - 1) / @page_bytes
|
|
845
|
+
per = (pages <= 1) ? n : (n + pages - 1) / pages
|
|
846
|
+
per = @page_rows if per > @page_rows
|
|
847
|
+
return [[0, n]] if per >= n
|
|
508
848
|
reps = buf.reps
|
|
509
849
|
ranges = []
|
|
510
850
|
start = 0
|
|
@@ -518,6 +858,8 @@ module Herringbone
|
|
|
518
858
|
ranges
|
|
519
859
|
end
|
|
520
860
|
|
|
861
|
+
# @param col [Schema::Column] column to check
|
|
862
|
+
# @return [Integer, nil] PLAIN-encoded bytes per value, nil for BYTE_ARRAY (variable width)
|
|
521
863
|
def value_width(col)
|
|
522
864
|
case col.type
|
|
523
865
|
when T::BOOLEAN then 1
|
|
@@ -528,6 +870,15 @@ module Herringbone
|
|
|
528
870
|
end
|
|
529
871
|
end
|
|
530
872
|
|
|
873
|
+
# Encodes a page's values. RLE_DICTIONARY values are the dictionary indices, prefixed with the
|
|
874
|
+
# bit width byte; RLE is only used for BOOLEAN and carries the 4-byte length prefix.
|
|
875
|
+
#
|
|
876
|
+
# @param values [Array] physical values, or dictionary indices
|
|
877
|
+
# @param encoding [Integer] encoding id
|
|
878
|
+
# @param type [Integer] physical type
|
|
879
|
+
# @param type_length [Integer, nil] FIXED_LEN_BYTE_ARRAY width
|
|
880
|
+
# @param dict_size [Integer, nil] number of dictionary entries, for RLE_DICTIONARY
|
|
881
|
+
# @return [String] encoded binary page values
|
|
531
882
|
def encode_values(values, encoding, type, type_length, dict_size)
|
|
532
883
|
case encoding
|
|
533
884
|
when E::PLAIN then Encodings::Plain.encode(values, type, type_length)
|
|
@@ -538,27 +889,41 @@ module Herringbone
|
|
|
538
889
|
body = Encodings::RLE.encode_hybrid(values.map { |v| v ? 1 : 0 }, 1)
|
|
539
890
|
[body.bytesize].pack("V") << body
|
|
540
891
|
when E::DELTA_BINARY_PACKED
|
|
541
|
-
Encodings::Delta.encode_binary_packed(values, type == T::INT32 ? 32 : 64)
|
|
892
|
+
Encodings::Delta.encode_binary_packed(values, (type == T::INT32) ? 32 : 64)
|
|
542
893
|
when E::DELTA_LENGTH_BYTE_ARRAY then Encodings::Delta.encode_length_byte_array(values)
|
|
543
894
|
when E::DELTA_BYTE_ARRAY then Encodings::Delta.encode_byte_array(values)
|
|
544
895
|
when E::BYTE_STREAM_SPLIT
|
|
545
|
-
width = {
|
|
896
|
+
width = {T::INT32 => 4, T::FLOAT => 4, T::INT64 => 8, T::DOUBLE => 8}[type] || type_length
|
|
546
897
|
Encodings::ByteStreamSplit.encode(Encodings::Plain.encode(values, type, type_length), width)
|
|
547
898
|
end
|
|
548
899
|
end
|
|
549
900
|
|
|
550
901
|
# Writes a page, returning the uncompressed size including the header
|
|
902
|
+
#
|
|
903
|
+
# @param header [Format::PageHeader] header; sizes and CRC32 are filled in here
|
|
904
|
+
# @param body [String] uncompressed page body
|
|
905
|
+
# @param compressed [String, nil] bytes to write as the page body, when already prepared (v2 data
|
|
906
|
+
# pages, whose levels stay uncompressed); +body+ is compressed otherwise
|
|
907
|
+
# @return [Integer]
|
|
551
908
|
def write_page(header, body, compressed = nil)
|
|
552
909
|
compressed ||= Compression.compress(@codec, body)
|
|
553
910
|
header.uncompressed_page_size ||= body.bytesize
|
|
554
911
|
header.compressed_page_size = compressed.bytesize
|
|
555
|
-
header.crc = Zlib.crc32(compressed).then { |c| c >= 0x8000_0000 ? c - 0x1_0000_0000 : c }
|
|
912
|
+
header.crc = Zlib.crc32(compressed).then { |c| (c >= 0x8000_0000) ? c - 0x1_0000_0000 : c }
|
|
556
913
|
encoded = header.encode
|
|
557
914
|
write_raw(encoded)
|
|
558
915
|
write_raw(compressed)
|
|
559
916
|
encoded.bytesize + header.uncompressed_page_size
|
|
560
917
|
end
|
|
561
918
|
|
|
919
|
+
# A DATA_PAGE: length-prefixed repetition and definition levels, then the values, all compressed
|
|
920
|
+
#
|
|
921
|
+
# @param n [Integer] number of level entries (values including nulls)
|
|
922
|
+
# @param rep_bytes [String] RLE-encoded repetition levels, empty when the column has none
|
|
923
|
+
# @param def_bytes [String] RLE-encoded definition levels, empty when the column has none
|
|
924
|
+
# @param encoded [String] encoded values
|
|
925
|
+
# @param encoding [Integer] encoding id of the values
|
|
926
|
+
# @return [Integer] uncompressed size including the header
|
|
562
927
|
def write_data_page_v1(n, rep_bytes, def_bytes, encoded, encoding)
|
|
563
928
|
body = String.new(encoding: Encoding::BINARY)
|
|
564
929
|
body << [rep_bytes.bytesize].pack("V") << rep_bytes unless rep_bytes.empty?
|
|
@@ -574,6 +939,16 @@ module Herringbone
|
|
|
574
939
|
write_page(header, body)
|
|
575
940
|
end
|
|
576
941
|
|
|
942
|
+
# A DATA_PAGE_V2: levels without length prefixes and uncompressed, then the compressed values
|
|
943
|
+
#
|
|
944
|
+
# @param n [Integer] number of level entries (values including nulls)
|
|
945
|
+
# @param nulls [Integer] null entries
|
|
946
|
+
# @param rows [Integer] rows in the page
|
|
947
|
+
# @param rep_bytes [String] RLE-encoded repetition levels, empty when the column has none
|
|
948
|
+
# @param def_bytes [String] RLE-encoded definition levels, empty when the column has none
|
|
949
|
+
# @param encoded [String] encoded values
|
|
950
|
+
# @param encoding [Integer] encoding id of the values
|
|
951
|
+
# @return [Integer] uncompressed size including the header
|
|
577
952
|
def write_data_page_v2(n, nulls, rows, rep_bytes, def_bytes, encoded, encoding)
|
|
578
953
|
compressed = Compression.compress(@codec, encoded)
|
|
579
954
|
header = Format::PageHeader.new(
|
|
@@ -589,46 +964,116 @@ module Herringbone
|
|
|
589
964
|
write_page(header, "".b, rep_bytes + def_bytes + compressed)
|
|
590
965
|
end
|
|
591
966
|
|
|
592
|
-
|
|
967
|
+
# Statistics and column index bounds longer than this are truncated, see #truncate_min
|
|
968
|
+
STAT_TRUNCATE_BYTES = 64
|
|
969
|
+
# Sort key for types whose Ruby ordering already matches Parquet's; compared by identity to
|
|
970
|
+
# skip calling it
|
|
971
|
+
IDENTITY = ->(v) { v }
|
|
972
|
+
|
|
973
|
+
# Chunk statistics: null count, plus min/max (flagged exact unless truncated) when the column
|
|
974
|
+
# has a sort order and non-NaN values
|
|
975
|
+
#
|
|
976
|
+
# @param col [Schema::Column] column the chunk belongs to
|
|
977
|
+
# @param defs [Array<Integer>] definition levels of the chunk
|
|
978
|
+
# @param values [Array] distinct or all non-null physical values of the chunk
|
|
979
|
+
# @param order [Proc, Method, nil] sort key from #sort_key
|
|
980
|
+
# @return [Format::Statistics]
|
|
981
|
+
def statistics_for(col, defs, values, order)
|
|
593
982
|
max_def = col.max_definition_level
|
|
594
983
|
nulls = max_def.zero? ? 0 : defs.size - defs.count(max_def)
|
|
595
984
|
stats = Format::Statistics.new(null_count: nulls)
|
|
596
|
-
|
|
597
|
-
if
|
|
598
|
-
|
|
599
|
-
|
|
600
|
-
stats.
|
|
601
|
-
stats.
|
|
985
|
+
range = order && value_range(col, values, order)
|
|
986
|
+
if range
|
|
987
|
+
min = stat_bytes(col, range[0])
|
|
988
|
+
max = stat_bytes(col, range[1])
|
|
989
|
+
stats.min_value = truncate_min(min)
|
|
990
|
+
stats.max_value = truncate_max(max)
|
|
991
|
+
stats.is_min_value_exact = stats.min_value.bytesize == min.bytesize
|
|
992
|
+
stats.is_max_value_exact = stats.max_value.bytesize == max.bytesize
|
|
602
993
|
end
|
|
603
994
|
stats
|
|
604
995
|
end
|
|
605
996
|
|
|
606
|
-
#
|
|
607
|
-
|
|
608
|
-
|
|
997
|
+
# A key giving the Parquet sort order of the column's physical values (IDENTITY when Ruby's own
|
|
998
|
+
# comparison already matches), or nil when the order is undefined (INT96)
|
|
999
|
+
#
|
|
1000
|
+
# @param col [Schema::Column] column to order
|
|
1001
|
+
# @return [Proc, Method, nil]
|
|
1002
|
+
def sort_key(col)
|
|
609
1003
|
kind, _, signed = Types.logical_of(col.node)
|
|
610
1004
|
case col.type
|
|
611
|
-
when T::BOOLEAN
|
|
612
|
-
[values.include?(false) ? "\x00".b : "\x01".b, values.include?(true) ? "\x01".b : "\x00".b]
|
|
1005
|
+
when T::BOOLEAN then ->(v) { v ? 1 : 0 }
|
|
613
1006
|
when T::INT32, T::INT64
|
|
614
|
-
return
|
|
615
|
-
|
|
616
|
-
|
|
617
|
-
|
|
618
|
-
when T::
|
|
619
|
-
|
|
620
|
-
|
|
621
|
-
|
|
1007
|
+
return IDENTITY unless kind == :integer && !signed
|
|
1008
|
+
mask = (col.type == T::INT32) ? 0xFFFF_FFFF : 0xFFFF_FFFF_FFFF_FFFF
|
|
1009
|
+
->(v) { v & mask }
|
|
1010
|
+
when T::FLOAT, T::DOUBLE then IDENTITY
|
|
1011
|
+
when T::BYTE_ARRAY, T::FIXED_LEN_BYTE_ARRAY
|
|
1012
|
+
case kind
|
|
1013
|
+
when :decimal then Types.method(:be_to_int)
|
|
1014
|
+
when :float16 then ->(v) { Types.half_to_float(v.unpack1("v")) }
|
|
1015
|
+
else IDENTITY # unsigned lexicographic, which is how Ruby compares binary Strings
|
|
1016
|
+
end
|
|
1017
|
+
end
|
|
1018
|
+
end
|
|
1019
|
+
|
|
1020
|
+
# [min, max] of +values+ in column order, ignoring NaNs; nil if there is nothing to compare.
|
|
1021
|
+
# A zero float bound is normalized to -0.0 (min) / 0.0 (max), as the spec asks.
|
|
1022
|
+
#
|
|
1023
|
+
# @param col [Schema::Column] column the values belong to
|
|
1024
|
+
# @param values [Array] physical values
|
|
1025
|
+
# @param order [Proc, Method] sort key from #sort_key
|
|
1026
|
+
# @return [Array(Object, Object), nil]
|
|
1027
|
+
def value_range(col, values, order)
|
|
1028
|
+
floats = col.type == T::FLOAT || col.type == T::DOUBLE
|
|
1029
|
+
if floats
|
|
1030
|
+
values = values.reject(&:nan?)
|
|
1031
|
+
elsif Types.logical_of(col.node).first == :float16
|
|
1032
|
+
values = values.reject { |v| order.call(v).nan? }
|
|
1033
|
+
end
|
|
1034
|
+
return nil if values.empty?
|
|
1035
|
+
min, max = order.equal?(IDENTITY) ? values.minmax : values.minmax_by(&order)
|
|
1036
|
+
if floats
|
|
622
1037
|
min = -0.0 if min.zero?
|
|
623
1038
|
max = 0.0 if max.zero?
|
|
624
|
-
fmt = col.type == T::FLOAT ? "e" : "E"
|
|
625
|
-
[[min].pack(fmt), [max].pack(fmt)]
|
|
626
|
-
when T::BYTE_ARRAY
|
|
627
|
-
return nil if kind == :decimal
|
|
628
|
-
min, max = values.minmax
|
|
629
|
-
return nil if min.bytesize > 1024 || max.bytesize > 1024
|
|
630
|
-
[min, max]
|
|
631
1039
|
end
|
|
1040
|
+
[min, max]
|
|
1041
|
+
end
|
|
1042
|
+
|
|
1043
|
+
# @param col [Schema::Column] column the value belongs to
|
|
1044
|
+
# @param value [Object] physical value
|
|
1045
|
+
# @return [String] the value PLAIN-encoded, as statistics store it (byte arrays without length)
|
|
1046
|
+
def stat_bytes(col, value)
|
|
1047
|
+
case col.type
|
|
1048
|
+
when T::BOOLEAN then value ? "\x01".b : "\x00".b
|
|
1049
|
+
when T::INT32 then [value].pack("l<")
|
|
1050
|
+
when T::INT64 then [value].pack("q<")
|
|
1051
|
+
when T::FLOAT then [value].pack("e")
|
|
1052
|
+
when T::DOUBLE then [value].pack("E")
|
|
1053
|
+
else value.b
|
|
1054
|
+
end
|
|
1055
|
+
end
|
|
1056
|
+
|
|
1057
|
+
# Long byte-array bounds are truncated: a prefix is still a lower bound for the minimum, and
|
|
1058
|
+
# a prefix with its last byte incremented is an upper bound for the maximum
|
|
1059
|
+
#
|
|
1060
|
+
# @param bytes [String] encoded minimum
|
|
1061
|
+
# @return [String] at most STAT_TRUNCATE_BYTES bytes
|
|
1062
|
+
def truncate_min(bytes)
|
|
1063
|
+
(bytes.bytesize > STAT_TRUNCATE_BYTES) ? bytes.byteslice(0, STAT_TRUNCATE_BYTES) : bytes
|
|
1064
|
+
end
|
|
1065
|
+
|
|
1066
|
+
# @param bytes [String] encoded maximum
|
|
1067
|
+
# @return [String] at most STAT_TRUNCATE_BYTES bytes, with the last byte that is not 0xFF
|
|
1068
|
+
# incremented (trailing 0xFF bytes dropped); +bytes+ unchanged when it fits, or when the
|
|
1069
|
+
# whole prefix is 0xFF
|
|
1070
|
+
def truncate_max(bytes)
|
|
1071
|
+
return bytes if bytes.bytesize <= STAT_TRUNCATE_BYTES
|
|
1072
|
+
prefix = bytes.byteslice(0, STAT_TRUNCATE_BYTES).bytes
|
|
1073
|
+
prefix.pop while prefix.last == 0xFF
|
|
1074
|
+
return bytes if prefix.empty?
|
|
1075
|
+
prefix[-1] += 1
|
|
1076
|
+
prefix.pack("C*")
|
|
632
1077
|
end
|
|
633
1078
|
end
|
|
634
1079
|
end
|