herringbone 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +7 -0
- data/LICENSE +21 -0
- data/README.md +209 -0
- data/bin/herringbone +54 -0
- data/lib/herringbone/active_record.rb +131 -0
- data/lib/herringbone/byte_values.rb +144 -0
- data/lib/herringbone/codecs/lz4.rb +305 -0
- data/lib/herringbone/codecs/snappy.rb +375 -0
- data/lib/herringbone/compression.rb +75 -0
- data/lib/herringbone/encodings/delta.rb +153 -0
- data/lib/herringbone/encodings/plain.rb +87 -0
- data/lib/herringbone/encodings/rle.rb +203 -0
- data/lib/herringbone/format.rb +309 -0
- data/lib/herringbone/io_buffer_support.rb +25 -0
- data/lib/herringbone/reader.rb +469 -0
- data/lib/herringbone/schema.rb +551 -0
- data/lib/herringbone/thrift.rb +334 -0
- data/lib/herringbone/types.rb +454 -0
- data/lib/herringbone/version.rb +5 -0
- data/lib/herringbone/writer.rb +634 -0
- data/lib/herringbone.rb +45 -0
- metadata +104 -0
|
@@ -0,0 +1,634 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Herringbone
|
|
4
|
+
# Writes Parquet files.
|
|
5
|
+
#
|
|
6
|
+
# schema = Herringbone::Schema.define do
|
|
7
|
+
# int64 :id, null: false
|
|
8
|
+
# string :name
|
|
9
|
+
# list :tags, :string
|
|
10
|
+
# end
|
|
11
|
+
# Herringbone::Writer.open("out.parquet", schema) do |w|
|
|
12
|
+
# w << { "id" => 1, "name" => "one", "tags" => ["a", "b"] }
|
|
13
|
+
# w << [2, "two", []] # Arrays are taken in schema order
|
|
14
|
+
# w << order # objects responding to #attributes (ActiveRecord) or #to_h
|
|
15
|
+
# end
|
|
16
|
+
#
|
|
17
|
+
# +schema+ is a Herringbone::Schema or a Hash spec (see Schema.define). The target is a path or an IO;
|
|
18
|
+
# paths are written to a temporary file next to the target and renamed into place on close, so a
|
|
19
|
+
# failed write never leaves a truncated file behind.
|
|
20
|
+
#
|
|
21
|
+
# Options:
|
|
22
|
+
# compression: :zstd (default), :snappy, :gzip, :lz4 (LZ4_RAW), :brotli, :none
|
|
23
|
+
# row_group_bytes: flush a row group once the buffered values take roughly this much memory
|
|
24
|
+
# (default 16MB). This bounds memory use while writing.
|
|
25
|
+
# row_group_size: also flush after this many rows (default: no row limit)
|
|
26
|
+
# page_size: approximate uncompressed data page size in bytes (default 1MB)
|
|
27
|
+
# data_page_version: 1 (default) or 2
|
|
28
|
+
# dictionary: true/false, or an Array of column paths to dictionary-encode
|
|
29
|
+
# encodings: { "path.to.column" => :delta_binary_packed, ... } for non-dictionary pages
|
|
30
|
+
# statistics: write min/max/null_count statistics (default true)
|
|
31
|
+
# metadata: Hash of String => String key/value metadata for the footer
|
|
32
|
+
class Writer
|
|
33
|
+
MAGIC = "PAR1"
|
|
34
|
+
T = Format::Type
|
|
35
|
+
E = Format::Encoding
|
|
36
|
+
|
|
37
|
+
ENCODING_NAMES = {
|
|
38
|
+
plain: E::PLAIN, rle: E::RLE, delta_binary_packed: E::DELTA_BINARY_PACKED,
|
|
39
|
+
delta_length_byte_array: E::DELTA_LENGTH_BYTE_ARRAY, delta_byte_array: E::DELTA_BYTE_ARRAY,
|
|
40
|
+
byte_stream_split: E::BYTE_STREAM_SPLIT
|
|
41
|
+
}.freeze
|
|
42
|
+
|
|
43
|
+
VALID_ENCODINGS = {
|
|
44
|
+
E::PLAIN => T::NAMES.keys,
|
|
45
|
+
E::RLE => [T::BOOLEAN],
|
|
46
|
+
E::DELTA_BINARY_PACKED => [T::INT32, T::INT64],
|
|
47
|
+
E::DELTA_LENGTH_BYTE_ARRAY => [T::BYTE_ARRAY],
|
|
48
|
+
E::DELTA_BYTE_ARRAY => [T::BYTE_ARRAY, T::FIXED_LEN_BYTE_ARRAY],
|
|
49
|
+
E::BYTE_STREAM_SPLIT => [T::INT32, T::INT64, T::FLOAT, T::DOUBLE, T::FIXED_LEN_BYTE_ARRAY]
|
|
50
|
+
}.freeze
|
|
51
|
+
|
|
52
|
+
MAX_DICTIONARY_BYTES = 1024 * 1024
|
|
53
|
+
# The row group byte size is first estimated after this many rows, then after every row group
|
|
54
|
+
ESTIMATE_AFTER_ROWS = 1000
|
|
55
|
+
|
|
56
|
+
attr_reader :schema
|
|
57
|
+
|
|
58
|
+
# Opens a writer on a path or an IO. With a block, the file is finished when the block returns
|
|
59
|
+
# (or discarded if it raises) and the block's value is returned; without one, call #close.
|
|
60
|
+
def self.open(target, schema, **options)
|
|
61
|
+
writer = new(target, schema, **options)
|
|
62
|
+
return writer unless block_given?
|
|
63
|
+
begin
|
|
64
|
+
result = yield writer
|
|
65
|
+
rescue Exception # rubocop:disable Lint/RescueException -- also discard on Interrupt
|
|
66
|
+
writer.abort
|
|
67
|
+
raise
|
|
68
|
+
end
|
|
69
|
+
writer.close
|
|
70
|
+
result
|
|
71
|
+
end
|
|
72
|
+
|
|
73
|
+
def initialize(target, schema, compression: :zstd, row_group_bytes: 16 * 1024 * 1024, row_group_size: nil,
|
|
74
|
+
page_size: 1024 * 1024, data_page_version: 1, dictionary: true, encodings: {}, statistics: true, metadata: {})
|
|
75
|
+
@schema = Schema.coerce(schema)
|
|
76
|
+
schema = @schema
|
|
77
|
+
@codec = Compression.codec_id(compression)
|
|
78
|
+
@row_group_bytes = Integer(row_group_bytes)
|
|
79
|
+
@row_group_size = row_group_size && Integer(row_group_size)
|
|
80
|
+
@row_limit = @row_group_size || ESTIMATE_AFTER_ROWS
|
|
81
|
+
@page_size = Integer(page_size)
|
|
82
|
+
@data_page_version = Integer(data_page_version)
|
|
83
|
+
raise ArgumentError, "data_page_version must be 1 or 2" unless [1, 2].include?(@data_page_version)
|
|
84
|
+
@dictionary = dictionary
|
|
85
|
+
@encodings = encodings.to_h { |path, enc| [path.to_s, encoding_id(path, enc)] }
|
|
86
|
+
unknown = @encodings.keys - schema.columns.map(&:dotted_path)
|
|
87
|
+
raise ArgumentError, "encodings: no such column #{unknown.join(", ")}" unless unknown.empty?
|
|
88
|
+
@statistics = statistics
|
|
89
|
+
@metadata = metadata
|
|
90
|
+
@row_groups = []
|
|
91
|
+
@total_rows = 0
|
|
92
|
+
@pos = 0
|
|
93
|
+
@closed = false
|
|
94
|
+
@bytes_per_row = nil
|
|
95
|
+
open_target(target)
|
|
96
|
+
write_raw(MAGIC)
|
|
97
|
+
reset_buffers
|
|
98
|
+
end
|
|
99
|
+
|
|
100
|
+
# Appends a row: a Hash keyed by top-level field names (Strings or Symbols), an Array of values
|
|
101
|
+
# in schema order, or an object responding to #attributes (ActiveRecord) or #to_h (Struct, Data)
|
|
102
|
+
def <<(row)
|
|
103
|
+
raise Error, "Writer is closed" if @closed
|
|
104
|
+
row = row_hash(row)
|
|
105
|
+
marks = @nested_buffers.empty? ? nil : @nested_buffers.map { |b| [b.defs.size, b.reps&.size, b.values.size] }
|
|
106
|
+
begin
|
|
107
|
+
@plan.each do |field, name, sym, buf, encoder|
|
|
108
|
+
value = row.fetch(name) { row[sym] }
|
|
109
|
+
if buf.nil?
|
|
110
|
+
shred(field, value, 0, 0)
|
|
111
|
+
elsif value.nil?
|
|
112
|
+
raise EncodeError, "Field #{name} is required but got nil" unless field.optional
|
|
113
|
+
buf.defs << 0
|
|
114
|
+
else
|
|
115
|
+
begin
|
|
116
|
+
buf.values << encoder.call(value)
|
|
117
|
+
rescue ArgumentError, TypeError, NoMethodError, RangeError => e
|
|
118
|
+
raise EncodeError, "Cannot write #{value.inspect} to #{name}: #{e.message}"
|
|
119
|
+
end
|
|
120
|
+
buf.defs << field.def_level
|
|
121
|
+
end
|
|
122
|
+
end
|
|
123
|
+
rescue EncodeError => e
|
|
124
|
+
rollback_row(marks)
|
|
125
|
+
raise EncodeError, "Row #{@total_rows + @buffered_rows}: #{e.message}"
|
|
126
|
+
rescue StandardError
|
|
127
|
+
rollback_row(marks)
|
|
128
|
+
raise
|
|
129
|
+
end
|
|
130
|
+
@buffered_rows += 1
|
|
131
|
+
check_row_group_size if @buffered_rows >= @row_limit
|
|
132
|
+
self
|
|
133
|
+
end
|
|
134
|
+
alias_method :write, :<<
|
|
135
|
+
|
|
136
|
+
def write_rows(rows)
|
|
137
|
+
rows.each { |row| self << row }
|
|
138
|
+
self
|
|
139
|
+
end
|
|
140
|
+
|
|
141
|
+
# Number of rows written so far, including buffered ones
|
|
142
|
+
def rows_written = @total_rows + @buffered_rows
|
|
143
|
+
|
|
144
|
+
# Writes any buffered rows as a row group
|
|
145
|
+
def flush_row_group
|
|
146
|
+
return if @buffered_rows.zero?
|
|
147
|
+
start = @pos
|
|
148
|
+
chunks = @schema.columns.map { |col| write_column_chunk(col, @buffers[col.index]) }
|
|
149
|
+
@row_groups << Format::RowGroup.new(
|
|
150
|
+
columns: chunks,
|
|
151
|
+
total_byte_size: chunks.sum { |c| c.meta_data.total_uncompressed_size },
|
|
152
|
+
num_rows: @buffered_rows,
|
|
153
|
+
file_offset: start,
|
|
154
|
+
total_compressed_size: @pos - start,
|
|
155
|
+
ordinal: @row_groups.size
|
|
156
|
+
)
|
|
157
|
+
@total_rows += @buffered_rows
|
|
158
|
+
reset_buffers
|
|
159
|
+
end
|
|
160
|
+
|
|
161
|
+
def close
|
|
162
|
+
return if @closed
|
|
163
|
+
flush_row_group
|
|
164
|
+
meta = Format::FileMetaData.new(
|
|
165
|
+
version: 2,
|
|
166
|
+
schema: @schema.to_elements,
|
|
167
|
+
num_rows: @total_rows,
|
|
168
|
+
row_groups: @row_groups,
|
|
169
|
+
key_value_metadata: @metadata.empty? ? nil : @metadata.map { |k, v| Format::KeyValue.new(key: k.to_s, value: v&.to_s) },
|
|
170
|
+
created_by: "herringbone version #{VERSION}",
|
|
171
|
+
column_orders: @schema.columns.map { Format::ColumnOrder.new(type_order: Format::TypeDefinedOrder.new) }
|
|
172
|
+
)
|
|
173
|
+
footer = meta.encode
|
|
174
|
+
write_raw(footer)
|
|
175
|
+
write_raw([footer.bytesize].pack("V"))
|
|
176
|
+
write_raw(MAGIC)
|
|
177
|
+
@io.flush if @io.respond_to?(:flush)
|
|
178
|
+
@closed = true
|
|
179
|
+
finish_target
|
|
180
|
+
end
|
|
181
|
+
|
|
182
|
+
# Stops writing without producing a file: a temporary file is deleted, an IO is left as is
|
|
183
|
+
def abort
|
|
184
|
+
return if @closed
|
|
185
|
+
@closed = true
|
|
186
|
+
return unless @temp_path
|
|
187
|
+
@io.close unless @io.closed?
|
|
188
|
+
File.unlink(@temp_path) if File.exist?(@temp_path)
|
|
189
|
+
end
|
|
190
|
+
|
|
191
|
+
private
|
|
192
|
+
|
|
193
|
+
# Levels are kept as binary Strings (one byte per entry). Values are an Array for numeric and
|
|
194
|
+
# boolean columns, and a compact ByteValues for BYTE_ARRAY / FIXED_LEN_BYTE_ARRAY columns.
|
|
195
|
+
ColumnBuffer = Struct.new(:defs, :reps, :values)
|
|
196
|
+
|
|
197
|
+
def reset_buffers
|
|
198
|
+
@buffers = @schema.columns.map do |col|
|
|
199
|
+
if col.max_definition_level > 255 || col.max_repetition_level > 255
|
|
200
|
+
raise UnsupportedError, "Column #{col.dotted_path} is nested too deeply"
|
|
201
|
+
end
|
|
202
|
+
ColumnBuffer.new(String.new(encoding: Encoding::BINARY),
|
|
203
|
+
col.max_repetition_level.positive? ? String.new(encoding: Encoding::BINARY) : nil,
|
|
204
|
+
new_values_store(col))
|
|
205
|
+
end
|
|
206
|
+
# Top-level, non-repeated leaves are written directly, everything else is shredded
|
|
207
|
+
@plan = @schema.fields.map do |field|
|
|
208
|
+
flat = field.leaf? && field.column.max_repetition_level.zero?
|
|
209
|
+
[field, field.name, field.name.to_sym, flat ? @buffers[field.column.index] : nil, flat ? field.column.encoder : nil]
|
|
210
|
+
end
|
|
211
|
+
@nested_buffers = @plan.reject { |p| p[3] }.flat_map { |p| p[0].leaves.map { |c| @buffers[c.index] } }
|
|
212
|
+
@buffered_rows = 0
|
|
213
|
+
end
|
|
214
|
+
|
|
215
|
+
def new_values_store(col)
|
|
216
|
+
case col.type
|
|
217
|
+
when T::BYTE_ARRAY then ByteValues.new(dictionary: use_dictionary?(col))
|
|
218
|
+
when T::FIXED_LEN_BYTE_ARRAY then ByteValues.new(width: col.type_length, dictionary: use_dictionary?(col))
|
|
219
|
+
else []
|
|
220
|
+
end
|
|
221
|
+
end
|
|
222
|
+
|
|
223
|
+
def open_target(target)
|
|
224
|
+
if target.respond_to?(:write)
|
|
225
|
+
@io = target
|
|
226
|
+
else
|
|
227
|
+
@path = target.to_s
|
|
228
|
+
dir = File.dirname(@path)
|
|
229
|
+
@temp_path = File.join(dir, ".#{File.basename(@path)}.#{Process.pid}.#{rand(1 << 32).to_s(36)}.tmp")
|
|
230
|
+
@io = File.open(@temp_path, "wb")
|
|
231
|
+
end
|
|
232
|
+
end
|
|
233
|
+
|
|
234
|
+
def finish_target
|
|
235
|
+
return unless @temp_path
|
|
236
|
+
@io.close
|
|
237
|
+
File.rename(@temp_path, @path)
|
|
238
|
+
end
|
|
239
|
+
|
|
240
|
+
# Flushes when the buffered values reach row_group_bytes (or row_group_size rows). The bytes per
|
|
241
|
+
# row are estimated from the buffered values after the first rows, then refreshed per row group.
|
|
242
|
+
def check_row_group_size
|
|
243
|
+
@bytes_per_row ||= estimate_bytes_per_row
|
|
244
|
+
limit = row_limit_for(@bytes_per_row)
|
|
245
|
+
if @buffered_rows >= limit
|
|
246
|
+
@bytes_per_row = estimate_bytes_per_row
|
|
247
|
+
flush_row_group
|
|
248
|
+
@row_limit = row_limit_for(@bytes_per_row)
|
|
249
|
+
else
|
|
250
|
+
@row_limit = limit
|
|
251
|
+
end
|
|
252
|
+
end
|
|
253
|
+
|
|
254
|
+
def row_limit_for(bytes_per_row)
|
|
255
|
+
by_bytes = [@row_group_bytes / [bytes_per_row, 1].max, ESTIMATE_AFTER_ROWS].max
|
|
256
|
+
@row_group_size ? [@row_group_size, by_bytes].min : by_bytes
|
|
257
|
+
end
|
|
258
|
+
|
|
259
|
+
def estimate_bytes_per_row
|
|
260
|
+
bytes = @schema.columns.sum do |col|
|
|
261
|
+
buf = @buffers[col.index]
|
|
262
|
+
values = buf.values
|
|
263
|
+
held = if values.is_a?(ByteValues)
|
|
264
|
+
values.memory_bytes
|
|
265
|
+
else
|
|
266
|
+
values.size * (col.type == T::INT96 ? 48 : 8)
|
|
267
|
+
end
|
|
268
|
+
held + buf.defs.bytesize + (buf.reps ? buf.reps.bytesize : 0)
|
|
269
|
+
end
|
|
270
|
+
bytes / [@buffered_rows, 1].max
|
|
271
|
+
end
|
|
272
|
+
|
|
273
|
+
def write_raw(bytes)
|
|
274
|
+
@io.write(bytes)
|
|
275
|
+
@pos += bytes.bytesize
|
|
276
|
+
end
|
|
277
|
+
|
|
278
|
+
def encoding_id(path, enc)
|
|
279
|
+
return enc if enc.is_a?(Integer)
|
|
280
|
+
ENCODING_NAMES.fetch(enc.to_s.downcase.to_sym) { raise ArgumentError, "Unknown encoding #{enc.inspect} for #{path}" }
|
|
281
|
+
end
|
|
282
|
+
|
|
283
|
+
def lookup(hash, name)
|
|
284
|
+
return nil if hash.nil?
|
|
285
|
+
hash = as_hash(hash, name) unless hash.is_a?(Hash)
|
|
286
|
+
hash.fetch(name) { hash[name.to_sym] }
|
|
287
|
+
end
|
|
288
|
+
|
|
289
|
+
def row_hash(row)
|
|
290
|
+
case row
|
|
291
|
+
when Hash then row
|
|
292
|
+
when Array
|
|
293
|
+
if row.size != @plan.size
|
|
294
|
+
raise EncodeError, "Row #{rows_written}: expected #{@plan.size} values in schema order, got #{row.size}"
|
|
295
|
+
end
|
|
296
|
+
@plan.each_with_index.to_h { |(_, name), i| [name, row[i]] }
|
|
297
|
+
else
|
|
298
|
+
return row.attributes if row.respond_to?(:attributes)
|
|
299
|
+
as_hash(row, "row #{rows_written}")
|
|
300
|
+
end
|
|
301
|
+
end
|
|
302
|
+
|
|
303
|
+
def as_hash(value, what)
|
|
304
|
+
return value if value.is_a?(Hash)
|
|
305
|
+
raise EncodeError, "Expected a Hash for #{what}, got #{value.class}" unless value.respond_to?(:to_h)
|
|
306
|
+
value.to_h
|
|
307
|
+
end
|
|
308
|
+
|
|
309
|
+
# Removes the entries a failed row left behind, so the buffers stay aligned
|
|
310
|
+
def rollback_row(marks)
|
|
311
|
+
@plan.each do |field, _, _, buf|
|
|
312
|
+
next unless buf && buf.defs.bytesize > @buffered_rows
|
|
313
|
+
d = buf.defs.getbyte(-1)
|
|
314
|
+
buf.defs.slice!(-1..)
|
|
315
|
+
buf.values.pop if d == field.def_level
|
|
316
|
+
end
|
|
317
|
+
marks&.each_with_index do |(defs, reps, values), i|
|
|
318
|
+
buf = @nested_buffers[i]
|
|
319
|
+
buf.defs.slice!(defs..)
|
|
320
|
+
buf.reps&.slice!(reps..)
|
|
321
|
+
buf.values.slice!(values..)
|
|
322
|
+
end
|
|
323
|
+
end
|
|
324
|
+
|
|
325
|
+
# Record shredding: turns a nested value into (definition level, repetition level, value) entries
|
|
326
|
+
def shred(field, value, parent_def, rep)
|
|
327
|
+
if value.nil?
|
|
328
|
+
raise EncodeError, "Field #{field.node.path.join(".")} is required but got nil" unless field.optional
|
|
329
|
+
field.leaves.each do |col|
|
|
330
|
+
buf = @buffers[col.index]
|
|
331
|
+
buf.defs << parent_def
|
|
332
|
+
buf.reps&.<< rep
|
|
333
|
+
end
|
|
334
|
+
return
|
|
335
|
+
end
|
|
336
|
+
|
|
337
|
+
d = field.def_level
|
|
338
|
+
case field.kind
|
|
339
|
+
when :leaf
|
|
340
|
+
buf = @buffers[field.column.index]
|
|
341
|
+
buf.defs << d
|
|
342
|
+
buf.reps&.<< rep
|
|
343
|
+
begin
|
|
344
|
+
buf.values << field.column.encoder.call(value)
|
|
345
|
+
rescue ArgumentError, TypeError, NoMethodError, RangeError => e
|
|
346
|
+
raise EncodeError, "Cannot write #{value.inspect} to #{field.column.dotted_path}: #{e.message}"
|
|
347
|
+
end
|
|
348
|
+
when :struct
|
|
349
|
+
field.children.each { |ch| shred(ch, lookup(value, ch.name), d, rep) }
|
|
350
|
+
when :list
|
|
351
|
+
raise EncodeError, "Expected an Array for #{field.node.path.join(".")}, got #{value.class}" unless value.respond_to?(:each_with_index)
|
|
352
|
+
if value.empty?
|
|
353
|
+
field.leaves.each do |col|
|
|
354
|
+
buf = @buffers[col.index]
|
|
355
|
+
buf.defs << d
|
|
356
|
+
buf.reps&.<< rep
|
|
357
|
+
end
|
|
358
|
+
else
|
|
359
|
+
value.each_with_index do |el, i|
|
|
360
|
+
shred(field.element, el, field.item_def, i.zero? ? rep : field.rep_level)
|
|
361
|
+
end
|
|
362
|
+
end
|
|
363
|
+
when :map
|
|
364
|
+
raise EncodeError, "Expected a Hash for #{field.node.path.join(".")}, got #{value.class}" unless value.respond_to?(:each_pair) || value.is_a?(Array)
|
|
365
|
+
if value.empty?
|
|
366
|
+
field.leaves.each do |col|
|
|
367
|
+
buf = @buffers[col.index]
|
|
368
|
+
buf.defs << d
|
|
369
|
+
buf.reps&.<< rep
|
|
370
|
+
end
|
|
371
|
+
else
|
|
372
|
+
value.each_with_index do |(k, v), i|
|
|
373
|
+
r = i.zero? ? rep : field.rep_level
|
|
374
|
+
raise EncodeError, "Map keys cannot be nil in #{field.node.path.join(".")}" if k.nil?
|
|
375
|
+
shred(field.key, k, field.item_def, r)
|
|
376
|
+
shred(field.value, v, field.item_def, r)
|
|
377
|
+
end
|
|
378
|
+
end
|
|
379
|
+
end
|
|
380
|
+
end
|
|
381
|
+
|
|
382
|
+
def write_column_chunk(col, buffer)
|
|
383
|
+
type = col.type
|
|
384
|
+
path = col.dotted_path
|
|
385
|
+
dict_values = nil
|
|
386
|
+
indices = nil
|
|
387
|
+
values = buffer.values
|
|
388
|
+
if values.is_a?(ByteValues)
|
|
389
|
+
kind, values, indices = values.materialize
|
|
390
|
+
if kind == :dictionary
|
|
391
|
+
dict_values = values
|
|
392
|
+
values = nil
|
|
393
|
+
end
|
|
394
|
+
elsif use_dictionary?(col) && !values.empty?
|
|
395
|
+
dict_values, indices = build_dictionary(values, type, col.type_length)
|
|
396
|
+
end
|
|
397
|
+
# Levels as Arrays of Integers, for this column only
|
|
398
|
+
buf = ColumnBuffer.new(buffer.defs.unpack("C*"), buffer.reps&.unpack("C*"), values)
|
|
399
|
+
value_encoding = if dict_values
|
|
400
|
+
E::RLE_DICTIONARY
|
|
401
|
+
else
|
|
402
|
+
enc = @encodings[path] || E::PLAIN
|
|
403
|
+
unless VALID_ENCODINGS.fetch(enc).include?(type)
|
|
404
|
+
raise ArgumentError, "Encoding #{E::NAMES[enc]} is not valid for #{T::NAMES[type]} column #{path}"
|
|
405
|
+
end
|
|
406
|
+
enc
|
|
407
|
+
end
|
|
408
|
+
|
|
409
|
+
chunk_start = @pos
|
|
410
|
+
uncompressed_total = 0
|
|
411
|
+
dictionary_offset = nil
|
|
412
|
+
if dict_values
|
|
413
|
+
dictionary_offset = @pos
|
|
414
|
+
plain = Encodings::Plain.encode(dict_values, type, col.type_length)
|
|
415
|
+
header = Format::PageHeader.new(
|
|
416
|
+
type: Format::PageType::DICTIONARY_PAGE,
|
|
417
|
+
dictionary_page_header: Format::DictionaryPageHeader.new(num_values: dict_values.size, encoding: E::PLAIN)
|
|
418
|
+
)
|
|
419
|
+
uncompressed_total += write_page(header, plain)
|
|
420
|
+
end
|
|
421
|
+
|
|
422
|
+
data_offset = @pos
|
|
423
|
+
max_def = col.max_definition_level
|
|
424
|
+
max_rep = col.max_repetition_level
|
|
425
|
+
value_index = 0
|
|
426
|
+
value_bytes = if indices
|
|
427
|
+
indices.size * 4
|
|
428
|
+
elsif (width = value_width(col))
|
|
429
|
+
values.size * width
|
|
430
|
+
else
|
|
431
|
+
values.sum(&:bytesize) + 4 * values.size
|
|
432
|
+
end
|
|
433
|
+
page_ranges(buf, value_bytes).each do |from, to|
|
|
434
|
+
n = to - from
|
|
435
|
+
defs = buf.defs[from, n]
|
|
436
|
+
reps = buf.reps&.slice(from, n)
|
|
437
|
+
non_null = max_def.zero? ? n : defs.count(max_def)
|
|
438
|
+
page_values = (indices || values)[value_index, non_null]
|
|
439
|
+
value_index += non_null
|
|
440
|
+
encoded = encode_values(page_values, value_encoding, type, col.type_length, dict_values&.size)
|
|
441
|
+
rep_bytes = max_rep.positive? ? Encodings::RLE.encode_hybrid(reps, max_rep.bit_length) : "".b
|
|
442
|
+
def_bytes = max_def.positive? ? Encodings::RLE.encode_hybrid(defs, max_def.bit_length) : "".b
|
|
443
|
+
uncompressed_total += if @data_page_version == 1
|
|
444
|
+
write_data_page_v1(n, rep_bytes, def_bytes, encoded, value_encoding)
|
|
445
|
+
else
|
|
446
|
+
num_rows = max_rep.zero? ? n : reps.count(0)
|
|
447
|
+
write_data_page_v2(n, n - non_null, num_rows, rep_bytes, def_bytes, encoded, value_encoding)
|
|
448
|
+
end
|
|
449
|
+
end
|
|
450
|
+
|
|
451
|
+
encodings = [E::RLE]
|
|
452
|
+
encodings << E::PLAIN if dict_values
|
|
453
|
+
encodings << value_encoding
|
|
454
|
+
meta = Format::ColumnMetaData.new(
|
|
455
|
+
type: type,
|
|
456
|
+
encodings: encodings.uniq,
|
|
457
|
+
path_in_schema: col.path,
|
|
458
|
+
codec: @codec,
|
|
459
|
+
num_values: buf.defs.size,
|
|
460
|
+
total_uncompressed_size: uncompressed_total,
|
|
461
|
+
total_compressed_size: @pos - chunk_start,
|
|
462
|
+
data_page_offset: data_offset,
|
|
463
|
+
dictionary_page_offset: dictionary_offset,
|
|
464
|
+
statistics: @statistics ? statistics_for(col, buf.defs, values || indices.uniq.map { |i| dict_values[i] }) : nil
|
|
465
|
+
)
|
|
466
|
+
Format::ColumnChunk.new(file_offset: chunk_start, meta_data: meta)
|
|
467
|
+
end
|
|
468
|
+
|
|
469
|
+
def use_dictionary?(col)
|
|
470
|
+
return false if col.type == T::BOOLEAN
|
|
471
|
+
case @dictionary
|
|
472
|
+
when true then ![T::BOOLEAN, T::FLOAT, T::DOUBLE].include?(col.type) && !@encodings.key?(col.dotted_path)
|
|
473
|
+
when false, nil then false
|
|
474
|
+
else Array(@dictionary).map(&:to_s).include?(col.dotted_path)
|
|
475
|
+
end
|
|
476
|
+
end
|
|
477
|
+
|
|
478
|
+
# Returns [dictionary_values, indices] or nil when a dictionary is not worthwhile
|
|
479
|
+
def build_dictionary(values, type, type_length)
|
|
480
|
+
# Floats are keyed by bit pattern so that -0.0 and 0.0 (and NaNs) stay distinct
|
|
481
|
+
if type == T::FLOAT || type == T::DOUBLE
|
|
482
|
+
keys = values.pack("G*").unpack("Q>*")
|
|
483
|
+
uniq = keys.uniq
|
|
484
|
+
return nil if uniq.size > values.size / 2 + 1 && values.size > 16
|
|
485
|
+
index = uniq.each_with_index.to_h
|
|
486
|
+
return [uniq.pack("Q>*").unpack("G*"), keys.map(&index)]
|
|
487
|
+
end
|
|
488
|
+
dict = values.uniq
|
|
489
|
+
return nil if dict.size > values.size / 2 + 1 && values.size > 16
|
|
490
|
+
size = case type
|
|
491
|
+
when T::BYTE_ARRAY then dict.sum { |v| v.bytesize + 4 }
|
|
492
|
+
when T::FIXED_LEN_BYTE_ARRAY then dict.size * type_length
|
|
493
|
+
else dict.size * 8
|
|
494
|
+
end
|
|
495
|
+
return nil if size > MAX_DICTIONARY_BYTES
|
|
496
|
+
index = dict.each_with_index.to_h
|
|
497
|
+
[dict, values.map(&index)]
|
|
498
|
+
end
|
|
499
|
+
|
|
500
|
+
# Splits a column buffer into pages of roughly @page_size bytes. Repeated columns
|
|
501
|
+
# are only cut where a new row starts.
|
|
502
|
+
def page_ranges(buf, value_bytes)
|
|
503
|
+
n = buf.defs.size
|
|
504
|
+
bytes = n + value_bytes
|
|
505
|
+
pages = (bytes + @page_size - 1) / @page_size
|
|
506
|
+
return [[0, n]] if pages <= 1
|
|
507
|
+
per = (n + pages - 1) / pages
|
|
508
|
+
reps = buf.reps
|
|
509
|
+
ranges = []
|
|
510
|
+
start = 0
|
|
511
|
+
while start < n
|
|
512
|
+
stop = start + per
|
|
513
|
+
stop = n if stop > n
|
|
514
|
+
stop += 1 while reps && stop < n && reps[stop] != 0
|
|
515
|
+
ranges << [start, stop]
|
|
516
|
+
start = stop
|
|
517
|
+
end
|
|
518
|
+
ranges
|
|
519
|
+
end
|
|
520
|
+
|
|
521
|
+
def value_width(col)
|
|
522
|
+
case col.type
|
|
523
|
+
when T::BOOLEAN then 1
|
|
524
|
+
when T::INT32, T::FLOAT then 4
|
|
525
|
+
when T::INT64, T::DOUBLE then 8
|
|
526
|
+
when T::INT96 then 12
|
|
527
|
+
when T::FIXED_LEN_BYTE_ARRAY then col.type_length
|
|
528
|
+
end
|
|
529
|
+
end
|
|
530
|
+
|
|
531
|
+
def encode_values(values, encoding, type, type_length, dict_size)
|
|
532
|
+
case encoding
|
|
533
|
+
when E::PLAIN then Encodings::Plain.encode(values, type, type_length)
|
|
534
|
+
when E::RLE_DICTIONARY
|
|
535
|
+
width = (dict_size - 1).bit_length
|
|
536
|
+
width.chr.b << Encodings::RLE.encode_hybrid(values, width)
|
|
537
|
+
when E::RLE
|
|
538
|
+
body = Encodings::RLE.encode_hybrid(values.map { |v| v ? 1 : 0 }, 1)
|
|
539
|
+
[body.bytesize].pack("V") << body
|
|
540
|
+
when E::DELTA_BINARY_PACKED
|
|
541
|
+
Encodings::Delta.encode_binary_packed(values, type == T::INT32 ? 32 : 64)
|
|
542
|
+
when E::DELTA_LENGTH_BYTE_ARRAY then Encodings::Delta.encode_length_byte_array(values)
|
|
543
|
+
when E::DELTA_BYTE_ARRAY then Encodings::Delta.encode_byte_array(values)
|
|
544
|
+
when E::BYTE_STREAM_SPLIT
|
|
545
|
+
width = { T::INT32 => 4, T::FLOAT => 4, T::INT64 => 8, T::DOUBLE => 8 }[type] || type_length
|
|
546
|
+
Encodings::ByteStreamSplit.encode(Encodings::Plain.encode(values, type, type_length), width)
|
|
547
|
+
end
|
|
548
|
+
end
|
|
549
|
+
|
|
550
|
+
# Writes a page, returning the uncompressed size including the header
|
|
551
|
+
def write_page(header, body, compressed = nil)
|
|
552
|
+
compressed ||= Compression.compress(@codec, body)
|
|
553
|
+
header.uncompressed_page_size ||= body.bytesize
|
|
554
|
+
header.compressed_page_size = compressed.bytesize
|
|
555
|
+
header.crc = Zlib.crc32(compressed).then { |c| c >= 0x8000_0000 ? c - 0x1_0000_0000 : c }
|
|
556
|
+
encoded = header.encode
|
|
557
|
+
write_raw(encoded)
|
|
558
|
+
write_raw(compressed)
|
|
559
|
+
encoded.bytesize + header.uncompressed_page_size
|
|
560
|
+
end
|
|
561
|
+
|
|
562
|
+
def write_data_page_v1(n, rep_bytes, def_bytes, encoded, encoding)
|
|
563
|
+
body = String.new(encoding: Encoding::BINARY)
|
|
564
|
+
body << [rep_bytes.bytesize].pack("V") << rep_bytes unless rep_bytes.empty?
|
|
565
|
+
body << [def_bytes.bytesize].pack("V") << def_bytes unless def_bytes.empty?
|
|
566
|
+
body << encoded
|
|
567
|
+
header = Format::PageHeader.new(
|
|
568
|
+
type: Format::PageType::DATA_PAGE,
|
|
569
|
+
data_page_header: Format::DataPageHeader.new(
|
|
570
|
+
num_values: n, encoding: encoding,
|
|
571
|
+
definition_level_encoding: E::RLE, repetition_level_encoding: E::RLE
|
|
572
|
+
)
|
|
573
|
+
)
|
|
574
|
+
write_page(header, body)
|
|
575
|
+
end
|
|
576
|
+
|
|
577
|
+
def write_data_page_v2(n, nulls, rows, rep_bytes, def_bytes, encoded, encoding)
|
|
578
|
+
compressed = Compression.compress(@codec, encoded)
|
|
579
|
+
header = Format::PageHeader.new(
|
|
580
|
+
type: Format::PageType::DATA_PAGE_V2,
|
|
581
|
+
uncompressed_page_size: rep_bytes.bytesize + def_bytes.bytesize + encoded.bytesize,
|
|
582
|
+
data_page_header_v2: Format::DataPageHeaderV2.new(
|
|
583
|
+
num_values: n, num_nulls: nulls, num_rows: rows, encoding: encoding,
|
|
584
|
+
definition_levels_byte_length: def_bytes.bytesize,
|
|
585
|
+
repetition_levels_byte_length: rep_bytes.bytesize,
|
|
586
|
+
is_compressed: @codec != Format::Codec::UNCOMPRESSED
|
|
587
|
+
)
|
|
588
|
+
)
|
|
589
|
+
write_page(header, "".b, rep_bytes + def_bytes + compressed)
|
|
590
|
+
end
|
|
591
|
+
|
|
592
|
+
def statistics_for(col, defs, values)
|
|
593
|
+
max_def = col.max_definition_level
|
|
594
|
+
nulls = max_def.zero? ? 0 : defs.size - defs.count(max_def)
|
|
595
|
+
stats = Format::Statistics.new(null_count: nulls)
|
|
596
|
+
min, max = min_max(col, values)
|
|
597
|
+
if min
|
|
598
|
+
stats.min_value = min
|
|
599
|
+
stats.max_value = max
|
|
600
|
+
stats.is_min_value_exact = true
|
|
601
|
+
stats.is_max_value_exact = true
|
|
602
|
+
end
|
|
603
|
+
stats
|
|
604
|
+
end
|
|
605
|
+
|
|
606
|
+
# Min/max for types whose Parquet sort order matches Ruby's comparison of the physical values
|
|
607
|
+
def min_max(col, values)
|
|
608
|
+
return nil if values.empty?
|
|
609
|
+
kind, _, signed = Types.logical_of(col.node)
|
|
610
|
+
case col.type
|
|
611
|
+
when T::BOOLEAN
|
|
612
|
+
[values.include?(false) ? "\x00".b : "\x01".b, values.include?(true) ? "\x01".b : "\x00".b]
|
|
613
|
+
when T::INT32, T::INT64
|
|
614
|
+
return nil if kind == :integer && !signed
|
|
615
|
+
fmt = col.type == T::INT32 ? "l<" : "q<"
|
|
616
|
+
min, max = values.minmax
|
|
617
|
+
[[min].pack(fmt), [max].pack(fmt)]
|
|
618
|
+
when T::FLOAT, T::DOUBLE
|
|
619
|
+
finite = values.reject(&:nan?)
|
|
620
|
+
return nil if finite.empty?
|
|
621
|
+
min, max = finite.minmax
|
|
622
|
+
min = -0.0 if min.zero?
|
|
623
|
+
max = 0.0 if max.zero?
|
|
624
|
+
fmt = col.type == T::FLOAT ? "e" : "E"
|
|
625
|
+
[[min].pack(fmt), [max].pack(fmt)]
|
|
626
|
+
when T::BYTE_ARRAY
|
|
627
|
+
return nil if kind == :decimal
|
|
628
|
+
min, max = values.minmax
|
|
629
|
+
return nil if min.bytesize > 1024 || max.bytesize > 1024
|
|
630
|
+
[min, max]
|
|
631
|
+
end
|
|
632
|
+
end
|
|
633
|
+
end
|
|
634
|
+
end
|
data/lib/herringbone.rb
ADDED
|
@@ -0,0 +1,45 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require "stringio"
|
|
4
|
+
require "pathname"
|
|
5
|
+
require_relative "herringbone/version"
|
|
6
|
+
|
|
7
|
+
# Pure-Ruby reader and writer for Apache Parquet files
|
|
8
|
+
module Herringbone
|
|
9
|
+
class Error < StandardError; end
|
|
10
|
+
class FormatError < Error; end
|
|
11
|
+
class DecodeError < FormatError; end
|
|
12
|
+
class EncodeError < Error; end
|
|
13
|
+
class UnsupportedError < Error; end
|
|
14
|
+
end
|
|
15
|
+
|
|
16
|
+
require_relative "herringbone/io_buffer_support"
|
|
17
|
+
require_relative "herringbone/codecs/snappy"
|
|
18
|
+
require_relative "herringbone/codecs/lz4"
|
|
19
|
+
require_relative "herringbone/thrift"
|
|
20
|
+
require_relative "herringbone/format"
|
|
21
|
+
require_relative "herringbone/encodings/rle"
|
|
22
|
+
require_relative "herringbone/encodings/plain"
|
|
23
|
+
require_relative "herringbone/encodings/delta"
|
|
24
|
+
require_relative "herringbone/compression"
|
|
25
|
+
require_relative "herringbone/types"
|
|
26
|
+
require_relative "herringbone/schema"
|
|
27
|
+
require_relative "herringbone/active_record"
|
|
28
|
+
require_relative "herringbone/reader"
|
|
29
|
+
require_relative "herringbone/byte_values"
|
|
30
|
+
require_relative "herringbone/writer"
|
|
31
|
+
|
|
32
|
+
module Herringbone
|
|
33
|
+
Delta = Encodings::Delta
|
|
34
|
+
|
|
35
|
+
module_function
|
|
36
|
+
|
|
37
|
+
def open(path, &block) = Reader.open(path, &block)
|
|
38
|
+
def read(path, columns: nil) = Reader.open(path) { |r| r.rows(columns: columns) }
|
|
39
|
+
|
|
40
|
+
# Writes an Enumerable of row Hashes. Without a schema, one is inferred from the first rows.
|
|
41
|
+
def write(path, rows, schema: nil, **options)
|
|
42
|
+
schema ||= Schema.infer(rows)
|
|
43
|
+
Writer.open(path, schema, **options) { |w| w.write_rows(rows) }
|
|
44
|
+
end
|
|
45
|
+
end
|