herringbone 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,634 @@
1
+ # frozen_string_literal: true
2
+
3
+ module Herringbone
4
+ # Writes Parquet files.
5
+ #
6
+ # schema = Herringbone::Schema.define do
7
+ # int64 :id, null: false
8
+ # string :name
9
+ # list :tags, :string
10
+ # end
11
+ # Herringbone::Writer.open("out.parquet", schema) do |w|
12
+ # w << { "id" => 1, "name" => "one", "tags" => ["a", "b"] }
13
+ # w << [2, "two", []] # Arrays are taken in schema order
14
+ # w << order # objects responding to #attributes (ActiveRecord) or #to_h
15
+ # end
16
+ #
17
+ # +schema+ is a Herringbone::Schema or a Hash spec (see Schema.define). The target is a path or an IO;
18
+ # paths are written to a temporary file next to the target and renamed into place on close, so a
19
+ # failed write never leaves a truncated file behind.
20
+ #
21
+ # Options:
22
+ # compression: :zstd (default), :snappy, :gzip, :lz4 (LZ4_RAW), :brotli, :none
23
+ # row_group_bytes: flush a row group once the buffered values take roughly this much memory
24
+ # (default 16MB). This bounds memory use while writing.
25
+ # row_group_size: also flush after this many rows (default: no row limit)
26
+ # page_size: approximate uncompressed data page size in bytes (default 1MB)
27
+ # data_page_version: 1 (default) or 2
28
+ # dictionary: true/false, or an Array of column paths to dictionary-encode
29
+ # encodings: { "path.to.column" => :delta_binary_packed, ... } for non-dictionary pages
30
+ # statistics: write min/max/null_count statistics (default true)
31
+ # metadata: Hash of String => String key/value metadata for the footer
32
+ class Writer
33
+ MAGIC = "PAR1"
34
+ T = Format::Type
35
+ E = Format::Encoding
36
+
37
+ ENCODING_NAMES = {
38
+ plain: E::PLAIN, rle: E::RLE, delta_binary_packed: E::DELTA_BINARY_PACKED,
39
+ delta_length_byte_array: E::DELTA_LENGTH_BYTE_ARRAY, delta_byte_array: E::DELTA_BYTE_ARRAY,
40
+ byte_stream_split: E::BYTE_STREAM_SPLIT
41
+ }.freeze
42
+
43
+ VALID_ENCODINGS = {
44
+ E::PLAIN => T::NAMES.keys,
45
+ E::RLE => [T::BOOLEAN],
46
+ E::DELTA_BINARY_PACKED => [T::INT32, T::INT64],
47
+ E::DELTA_LENGTH_BYTE_ARRAY => [T::BYTE_ARRAY],
48
+ E::DELTA_BYTE_ARRAY => [T::BYTE_ARRAY, T::FIXED_LEN_BYTE_ARRAY],
49
+ E::BYTE_STREAM_SPLIT => [T::INT32, T::INT64, T::FLOAT, T::DOUBLE, T::FIXED_LEN_BYTE_ARRAY]
50
+ }.freeze
51
+
52
+ MAX_DICTIONARY_BYTES = 1024 * 1024
53
+ # The row group byte size is first estimated after this many rows, then after every row group
54
+ ESTIMATE_AFTER_ROWS = 1000
55
+
56
+ attr_reader :schema
57
+
58
+ # Opens a writer on a path or an IO. With a block, the file is finished when the block returns
59
+ # (or discarded if it raises) and the block's value is returned; without one, call #close.
60
+ def self.open(target, schema, **options)
61
+ writer = new(target, schema, **options)
62
+ return writer unless block_given?
63
+ begin
64
+ result = yield writer
65
+ rescue Exception # rubocop:disable Lint/RescueException -- also discard on Interrupt
66
+ writer.abort
67
+ raise
68
+ end
69
+ writer.close
70
+ result
71
+ end
72
+
73
+ def initialize(target, schema, compression: :zstd, row_group_bytes: 16 * 1024 * 1024, row_group_size: nil,
74
+ page_size: 1024 * 1024, data_page_version: 1, dictionary: true, encodings: {}, statistics: true, metadata: {})
75
+ @schema = Schema.coerce(schema)
76
+ schema = @schema
77
+ @codec = Compression.codec_id(compression)
78
+ @row_group_bytes = Integer(row_group_bytes)
79
+ @row_group_size = row_group_size && Integer(row_group_size)
80
+ @row_limit = @row_group_size || ESTIMATE_AFTER_ROWS
81
+ @page_size = Integer(page_size)
82
+ @data_page_version = Integer(data_page_version)
83
+ raise ArgumentError, "data_page_version must be 1 or 2" unless [1, 2].include?(@data_page_version)
84
+ @dictionary = dictionary
85
+ @encodings = encodings.to_h { |path, enc| [path.to_s, encoding_id(path, enc)] }
86
+ unknown = @encodings.keys - schema.columns.map(&:dotted_path)
87
+ raise ArgumentError, "encodings: no such column #{unknown.join(", ")}" unless unknown.empty?
88
+ @statistics = statistics
89
+ @metadata = metadata
90
+ @row_groups = []
91
+ @total_rows = 0
92
+ @pos = 0
93
+ @closed = false
94
+ @bytes_per_row = nil
95
+ open_target(target)
96
+ write_raw(MAGIC)
97
+ reset_buffers
98
+ end
99
+
100
+ # Appends a row: a Hash keyed by top-level field names (Strings or Symbols), an Array of values
101
+ # in schema order, or an object responding to #attributes (ActiveRecord) or #to_h (Struct, Data)
102
+ def <<(row)
103
+ raise Error, "Writer is closed" if @closed
104
+ row = row_hash(row)
105
+ marks = @nested_buffers.empty? ? nil : @nested_buffers.map { |b| [b.defs.size, b.reps&.size, b.values.size] }
106
+ begin
107
+ @plan.each do |field, name, sym, buf, encoder|
108
+ value = row.fetch(name) { row[sym] }
109
+ if buf.nil?
110
+ shred(field, value, 0, 0)
111
+ elsif value.nil?
112
+ raise EncodeError, "Field #{name} is required but got nil" unless field.optional
113
+ buf.defs << 0
114
+ else
115
+ begin
116
+ buf.values << encoder.call(value)
117
+ rescue ArgumentError, TypeError, NoMethodError, RangeError => e
118
+ raise EncodeError, "Cannot write #{value.inspect} to #{name}: #{e.message}"
119
+ end
120
+ buf.defs << field.def_level
121
+ end
122
+ end
123
+ rescue EncodeError => e
124
+ rollback_row(marks)
125
+ raise EncodeError, "Row #{@total_rows + @buffered_rows}: #{e.message}"
126
+ rescue StandardError
127
+ rollback_row(marks)
128
+ raise
129
+ end
130
+ @buffered_rows += 1
131
+ check_row_group_size if @buffered_rows >= @row_limit
132
+ self
133
+ end
134
+ alias_method :write, :<<
135
+
136
+ def write_rows(rows)
137
+ rows.each { |row| self << row }
138
+ self
139
+ end
140
+
141
+ # Number of rows written so far, including buffered ones
142
+ def rows_written = @total_rows + @buffered_rows
143
+
144
+ # Writes any buffered rows as a row group
145
+ def flush_row_group
146
+ return if @buffered_rows.zero?
147
+ start = @pos
148
+ chunks = @schema.columns.map { |col| write_column_chunk(col, @buffers[col.index]) }
149
+ @row_groups << Format::RowGroup.new(
150
+ columns: chunks,
151
+ total_byte_size: chunks.sum { |c| c.meta_data.total_uncompressed_size },
152
+ num_rows: @buffered_rows,
153
+ file_offset: start,
154
+ total_compressed_size: @pos - start,
155
+ ordinal: @row_groups.size
156
+ )
157
+ @total_rows += @buffered_rows
158
+ reset_buffers
159
+ end
160
+
161
+ def close
162
+ return if @closed
163
+ flush_row_group
164
+ meta = Format::FileMetaData.new(
165
+ version: 2,
166
+ schema: @schema.to_elements,
167
+ num_rows: @total_rows,
168
+ row_groups: @row_groups,
169
+ key_value_metadata: @metadata.empty? ? nil : @metadata.map { |k, v| Format::KeyValue.new(key: k.to_s, value: v&.to_s) },
170
+ created_by: "herringbone version #{VERSION}",
171
+ column_orders: @schema.columns.map { Format::ColumnOrder.new(type_order: Format::TypeDefinedOrder.new) }
172
+ )
173
+ footer = meta.encode
174
+ write_raw(footer)
175
+ write_raw([footer.bytesize].pack("V"))
176
+ write_raw(MAGIC)
177
+ @io.flush if @io.respond_to?(:flush)
178
+ @closed = true
179
+ finish_target
180
+ end
181
+
182
+ # Stops writing without producing a file: a temporary file is deleted, an IO is left as is
183
+ def abort
184
+ return if @closed
185
+ @closed = true
186
+ return unless @temp_path
187
+ @io.close unless @io.closed?
188
+ File.unlink(@temp_path) if File.exist?(@temp_path)
189
+ end
190
+
191
+ private
192
+
193
+ # Levels are kept as binary Strings (one byte per entry). Values are an Array for numeric and
194
+ # boolean columns, and a compact ByteValues for BYTE_ARRAY / FIXED_LEN_BYTE_ARRAY columns.
195
+ ColumnBuffer = Struct.new(:defs, :reps, :values)
196
+
197
+ def reset_buffers
198
+ @buffers = @schema.columns.map do |col|
199
+ if col.max_definition_level > 255 || col.max_repetition_level > 255
200
+ raise UnsupportedError, "Column #{col.dotted_path} is nested too deeply"
201
+ end
202
+ ColumnBuffer.new(String.new(encoding: Encoding::BINARY),
203
+ col.max_repetition_level.positive? ? String.new(encoding: Encoding::BINARY) : nil,
204
+ new_values_store(col))
205
+ end
206
+ # Top-level, non-repeated leaves are written directly, everything else is shredded
207
+ @plan = @schema.fields.map do |field|
208
+ flat = field.leaf? && field.column.max_repetition_level.zero?
209
+ [field, field.name, field.name.to_sym, flat ? @buffers[field.column.index] : nil, flat ? field.column.encoder : nil]
210
+ end
211
+ @nested_buffers = @plan.reject { |p| p[3] }.flat_map { |p| p[0].leaves.map { |c| @buffers[c.index] } }
212
+ @buffered_rows = 0
213
+ end
214
+
215
+ def new_values_store(col)
216
+ case col.type
217
+ when T::BYTE_ARRAY then ByteValues.new(dictionary: use_dictionary?(col))
218
+ when T::FIXED_LEN_BYTE_ARRAY then ByteValues.new(width: col.type_length, dictionary: use_dictionary?(col))
219
+ else []
220
+ end
221
+ end
222
+
223
+ def open_target(target)
224
+ if target.respond_to?(:write)
225
+ @io = target
226
+ else
227
+ @path = target.to_s
228
+ dir = File.dirname(@path)
229
+ @temp_path = File.join(dir, ".#{File.basename(@path)}.#{Process.pid}.#{rand(1 << 32).to_s(36)}.tmp")
230
+ @io = File.open(@temp_path, "wb")
231
+ end
232
+ end
233
+
234
+ def finish_target
235
+ return unless @temp_path
236
+ @io.close
237
+ File.rename(@temp_path, @path)
238
+ end
239
+
240
+ # Flushes when the buffered values reach row_group_bytes (or row_group_size rows). The bytes per
241
+ # row are estimated from the buffered values after the first rows, then refreshed per row group.
242
+ def check_row_group_size
243
+ @bytes_per_row ||= estimate_bytes_per_row
244
+ limit = row_limit_for(@bytes_per_row)
245
+ if @buffered_rows >= limit
246
+ @bytes_per_row = estimate_bytes_per_row
247
+ flush_row_group
248
+ @row_limit = row_limit_for(@bytes_per_row)
249
+ else
250
+ @row_limit = limit
251
+ end
252
+ end
253
+
254
+ def row_limit_for(bytes_per_row)
255
+ by_bytes = [@row_group_bytes / [bytes_per_row, 1].max, ESTIMATE_AFTER_ROWS].max
256
+ @row_group_size ? [@row_group_size, by_bytes].min : by_bytes
257
+ end
258
+
259
+ def estimate_bytes_per_row
260
+ bytes = @schema.columns.sum do |col|
261
+ buf = @buffers[col.index]
262
+ values = buf.values
263
+ held = if values.is_a?(ByteValues)
264
+ values.memory_bytes
265
+ else
266
+ values.size * (col.type == T::INT96 ? 48 : 8)
267
+ end
268
+ held + buf.defs.bytesize + (buf.reps ? buf.reps.bytesize : 0)
269
+ end
270
+ bytes / [@buffered_rows, 1].max
271
+ end
272
+
273
+ def write_raw(bytes)
274
+ @io.write(bytes)
275
+ @pos += bytes.bytesize
276
+ end
277
+
278
+ def encoding_id(path, enc)
279
+ return enc if enc.is_a?(Integer)
280
+ ENCODING_NAMES.fetch(enc.to_s.downcase.to_sym) { raise ArgumentError, "Unknown encoding #{enc.inspect} for #{path}" }
281
+ end
282
+
283
+ def lookup(hash, name)
284
+ return nil if hash.nil?
285
+ hash = as_hash(hash, name) unless hash.is_a?(Hash)
286
+ hash.fetch(name) { hash[name.to_sym] }
287
+ end
288
+
289
+ def row_hash(row)
290
+ case row
291
+ when Hash then row
292
+ when Array
293
+ if row.size != @plan.size
294
+ raise EncodeError, "Row #{rows_written}: expected #{@plan.size} values in schema order, got #{row.size}"
295
+ end
296
+ @plan.each_with_index.to_h { |(_, name), i| [name, row[i]] }
297
+ else
298
+ return row.attributes if row.respond_to?(:attributes)
299
+ as_hash(row, "row #{rows_written}")
300
+ end
301
+ end
302
+
303
+ def as_hash(value, what)
304
+ return value if value.is_a?(Hash)
305
+ raise EncodeError, "Expected a Hash for #{what}, got #{value.class}" unless value.respond_to?(:to_h)
306
+ value.to_h
307
+ end
308
+
309
+ # Removes the entries a failed row left behind, so the buffers stay aligned
310
+ def rollback_row(marks)
311
+ @plan.each do |field, _, _, buf|
312
+ next unless buf && buf.defs.bytesize > @buffered_rows
313
+ d = buf.defs.getbyte(-1)
314
+ buf.defs.slice!(-1..)
315
+ buf.values.pop if d == field.def_level
316
+ end
317
+ marks&.each_with_index do |(defs, reps, values), i|
318
+ buf = @nested_buffers[i]
319
+ buf.defs.slice!(defs..)
320
+ buf.reps&.slice!(reps..)
321
+ buf.values.slice!(values..)
322
+ end
323
+ end
324
+
325
+ # Record shredding: turns a nested value into (definition level, repetition level, value) entries
326
+ def shred(field, value, parent_def, rep)
327
+ if value.nil?
328
+ raise EncodeError, "Field #{field.node.path.join(".")} is required but got nil" unless field.optional
329
+ field.leaves.each do |col|
330
+ buf = @buffers[col.index]
331
+ buf.defs << parent_def
332
+ buf.reps&.<< rep
333
+ end
334
+ return
335
+ end
336
+
337
+ d = field.def_level
338
+ case field.kind
339
+ when :leaf
340
+ buf = @buffers[field.column.index]
341
+ buf.defs << d
342
+ buf.reps&.<< rep
343
+ begin
344
+ buf.values << field.column.encoder.call(value)
345
+ rescue ArgumentError, TypeError, NoMethodError, RangeError => e
346
+ raise EncodeError, "Cannot write #{value.inspect} to #{field.column.dotted_path}: #{e.message}"
347
+ end
348
+ when :struct
349
+ field.children.each { |ch| shred(ch, lookup(value, ch.name), d, rep) }
350
+ when :list
351
+ raise EncodeError, "Expected an Array for #{field.node.path.join(".")}, got #{value.class}" unless value.respond_to?(:each_with_index)
352
+ if value.empty?
353
+ field.leaves.each do |col|
354
+ buf = @buffers[col.index]
355
+ buf.defs << d
356
+ buf.reps&.<< rep
357
+ end
358
+ else
359
+ value.each_with_index do |el, i|
360
+ shred(field.element, el, field.item_def, i.zero? ? rep : field.rep_level)
361
+ end
362
+ end
363
+ when :map
364
+ raise EncodeError, "Expected a Hash for #{field.node.path.join(".")}, got #{value.class}" unless value.respond_to?(:each_pair) || value.is_a?(Array)
365
+ if value.empty?
366
+ field.leaves.each do |col|
367
+ buf = @buffers[col.index]
368
+ buf.defs << d
369
+ buf.reps&.<< rep
370
+ end
371
+ else
372
+ value.each_with_index do |(k, v), i|
373
+ r = i.zero? ? rep : field.rep_level
374
+ raise EncodeError, "Map keys cannot be nil in #{field.node.path.join(".")}" if k.nil?
375
+ shred(field.key, k, field.item_def, r)
376
+ shred(field.value, v, field.item_def, r)
377
+ end
378
+ end
379
+ end
380
+ end
381
+
382
+ def write_column_chunk(col, buffer)
383
+ type = col.type
384
+ path = col.dotted_path
385
+ dict_values = nil
386
+ indices = nil
387
+ values = buffer.values
388
+ if values.is_a?(ByteValues)
389
+ kind, values, indices = values.materialize
390
+ if kind == :dictionary
391
+ dict_values = values
392
+ values = nil
393
+ end
394
+ elsif use_dictionary?(col) && !values.empty?
395
+ dict_values, indices = build_dictionary(values, type, col.type_length)
396
+ end
397
+ # Levels as Arrays of Integers, for this column only
398
+ buf = ColumnBuffer.new(buffer.defs.unpack("C*"), buffer.reps&.unpack("C*"), values)
399
+ value_encoding = if dict_values
400
+ E::RLE_DICTIONARY
401
+ else
402
+ enc = @encodings[path] || E::PLAIN
403
+ unless VALID_ENCODINGS.fetch(enc).include?(type)
404
+ raise ArgumentError, "Encoding #{E::NAMES[enc]} is not valid for #{T::NAMES[type]} column #{path}"
405
+ end
406
+ enc
407
+ end
408
+
409
+ chunk_start = @pos
410
+ uncompressed_total = 0
411
+ dictionary_offset = nil
412
+ if dict_values
413
+ dictionary_offset = @pos
414
+ plain = Encodings::Plain.encode(dict_values, type, col.type_length)
415
+ header = Format::PageHeader.new(
416
+ type: Format::PageType::DICTIONARY_PAGE,
417
+ dictionary_page_header: Format::DictionaryPageHeader.new(num_values: dict_values.size, encoding: E::PLAIN)
418
+ )
419
+ uncompressed_total += write_page(header, plain)
420
+ end
421
+
422
+ data_offset = @pos
423
+ max_def = col.max_definition_level
424
+ max_rep = col.max_repetition_level
425
+ value_index = 0
426
+ value_bytes = if indices
427
+ indices.size * 4
428
+ elsif (width = value_width(col))
429
+ values.size * width
430
+ else
431
+ values.sum(&:bytesize) + 4 * values.size
432
+ end
433
+ page_ranges(buf, value_bytes).each do |from, to|
434
+ n = to - from
435
+ defs = buf.defs[from, n]
436
+ reps = buf.reps&.slice(from, n)
437
+ non_null = max_def.zero? ? n : defs.count(max_def)
438
+ page_values = (indices || values)[value_index, non_null]
439
+ value_index += non_null
440
+ encoded = encode_values(page_values, value_encoding, type, col.type_length, dict_values&.size)
441
+ rep_bytes = max_rep.positive? ? Encodings::RLE.encode_hybrid(reps, max_rep.bit_length) : "".b
442
+ def_bytes = max_def.positive? ? Encodings::RLE.encode_hybrid(defs, max_def.bit_length) : "".b
443
+ uncompressed_total += if @data_page_version == 1
444
+ write_data_page_v1(n, rep_bytes, def_bytes, encoded, value_encoding)
445
+ else
446
+ num_rows = max_rep.zero? ? n : reps.count(0)
447
+ write_data_page_v2(n, n - non_null, num_rows, rep_bytes, def_bytes, encoded, value_encoding)
448
+ end
449
+ end
450
+
451
+ encodings = [E::RLE]
452
+ encodings << E::PLAIN if dict_values
453
+ encodings << value_encoding
454
+ meta = Format::ColumnMetaData.new(
455
+ type: type,
456
+ encodings: encodings.uniq,
457
+ path_in_schema: col.path,
458
+ codec: @codec,
459
+ num_values: buf.defs.size,
460
+ total_uncompressed_size: uncompressed_total,
461
+ total_compressed_size: @pos - chunk_start,
462
+ data_page_offset: data_offset,
463
+ dictionary_page_offset: dictionary_offset,
464
+ statistics: @statistics ? statistics_for(col, buf.defs, values || indices.uniq.map { |i| dict_values[i] }) : nil
465
+ )
466
+ Format::ColumnChunk.new(file_offset: chunk_start, meta_data: meta)
467
+ end
468
+
469
+ def use_dictionary?(col)
470
+ return false if col.type == T::BOOLEAN
471
+ case @dictionary
472
+ when true then ![T::BOOLEAN, T::FLOAT, T::DOUBLE].include?(col.type) && !@encodings.key?(col.dotted_path)
473
+ when false, nil then false
474
+ else Array(@dictionary).map(&:to_s).include?(col.dotted_path)
475
+ end
476
+ end
477
+
478
+ # Returns [dictionary_values, indices] or nil when a dictionary is not worthwhile
479
+ def build_dictionary(values, type, type_length)
480
+ # Floats are keyed by bit pattern so that -0.0 and 0.0 (and NaNs) stay distinct
481
+ if type == T::FLOAT || type == T::DOUBLE
482
+ keys = values.pack("G*").unpack("Q>*")
483
+ uniq = keys.uniq
484
+ return nil if uniq.size > values.size / 2 + 1 && values.size > 16
485
+ index = uniq.each_with_index.to_h
486
+ return [uniq.pack("Q>*").unpack("G*"), keys.map(&index)]
487
+ end
488
+ dict = values.uniq
489
+ return nil if dict.size > values.size / 2 + 1 && values.size > 16
490
+ size = case type
491
+ when T::BYTE_ARRAY then dict.sum { |v| v.bytesize + 4 }
492
+ when T::FIXED_LEN_BYTE_ARRAY then dict.size * type_length
493
+ else dict.size * 8
494
+ end
495
+ return nil if size > MAX_DICTIONARY_BYTES
496
+ index = dict.each_with_index.to_h
497
+ [dict, values.map(&index)]
498
+ end
499
+
500
+ # Splits a column buffer into pages of roughly @page_size bytes. Repeated columns
501
+ # are only cut where a new row starts.
502
+ def page_ranges(buf, value_bytes)
503
+ n = buf.defs.size
504
+ bytes = n + value_bytes
505
+ pages = (bytes + @page_size - 1) / @page_size
506
+ return [[0, n]] if pages <= 1
507
+ per = (n + pages - 1) / pages
508
+ reps = buf.reps
509
+ ranges = []
510
+ start = 0
511
+ while start < n
512
+ stop = start + per
513
+ stop = n if stop > n
514
+ stop += 1 while reps && stop < n && reps[stop] != 0
515
+ ranges << [start, stop]
516
+ start = stop
517
+ end
518
+ ranges
519
+ end
520
+
521
+ def value_width(col)
522
+ case col.type
523
+ when T::BOOLEAN then 1
524
+ when T::INT32, T::FLOAT then 4
525
+ when T::INT64, T::DOUBLE then 8
526
+ when T::INT96 then 12
527
+ when T::FIXED_LEN_BYTE_ARRAY then col.type_length
528
+ end
529
+ end
530
+
531
+ def encode_values(values, encoding, type, type_length, dict_size)
532
+ case encoding
533
+ when E::PLAIN then Encodings::Plain.encode(values, type, type_length)
534
+ when E::RLE_DICTIONARY
535
+ width = (dict_size - 1).bit_length
536
+ width.chr.b << Encodings::RLE.encode_hybrid(values, width)
537
+ when E::RLE
538
+ body = Encodings::RLE.encode_hybrid(values.map { |v| v ? 1 : 0 }, 1)
539
+ [body.bytesize].pack("V") << body
540
+ when E::DELTA_BINARY_PACKED
541
+ Encodings::Delta.encode_binary_packed(values, type == T::INT32 ? 32 : 64)
542
+ when E::DELTA_LENGTH_BYTE_ARRAY then Encodings::Delta.encode_length_byte_array(values)
543
+ when E::DELTA_BYTE_ARRAY then Encodings::Delta.encode_byte_array(values)
544
+ when E::BYTE_STREAM_SPLIT
545
+ width = { T::INT32 => 4, T::FLOAT => 4, T::INT64 => 8, T::DOUBLE => 8 }[type] || type_length
546
+ Encodings::ByteStreamSplit.encode(Encodings::Plain.encode(values, type, type_length), width)
547
+ end
548
+ end
549
+
550
+ # Writes a page, returning the uncompressed size including the header
551
+ def write_page(header, body, compressed = nil)
552
+ compressed ||= Compression.compress(@codec, body)
553
+ header.uncompressed_page_size ||= body.bytesize
554
+ header.compressed_page_size = compressed.bytesize
555
+ header.crc = Zlib.crc32(compressed).then { |c| c >= 0x8000_0000 ? c - 0x1_0000_0000 : c }
556
+ encoded = header.encode
557
+ write_raw(encoded)
558
+ write_raw(compressed)
559
+ encoded.bytesize + header.uncompressed_page_size
560
+ end
561
+
562
+ def write_data_page_v1(n, rep_bytes, def_bytes, encoded, encoding)
563
+ body = String.new(encoding: Encoding::BINARY)
564
+ body << [rep_bytes.bytesize].pack("V") << rep_bytes unless rep_bytes.empty?
565
+ body << [def_bytes.bytesize].pack("V") << def_bytes unless def_bytes.empty?
566
+ body << encoded
567
+ header = Format::PageHeader.new(
568
+ type: Format::PageType::DATA_PAGE,
569
+ data_page_header: Format::DataPageHeader.new(
570
+ num_values: n, encoding: encoding,
571
+ definition_level_encoding: E::RLE, repetition_level_encoding: E::RLE
572
+ )
573
+ )
574
+ write_page(header, body)
575
+ end
576
+
577
+ def write_data_page_v2(n, nulls, rows, rep_bytes, def_bytes, encoded, encoding)
578
+ compressed = Compression.compress(@codec, encoded)
579
+ header = Format::PageHeader.new(
580
+ type: Format::PageType::DATA_PAGE_V2,
581
+ uncompressed_page_size: rep_bytes.bytesize + def_bytes.bytesize + encoded.bytesize,
582
+ data_page_header_v2: Format::DataPageHeaderV2.new(
583
+ num_values: n, num_nulls: nulls, num_rows: rows, encoding: encoding,
584
+ definition_levels_byte_length: def_bytes.bytesize,
585
+ repetition_levels_byte_length: rep_bytes.bytesize,
586
+ is_compressed: @codec != Format::Codec::UNCOMPRESSED
587
+ )
588
+ )
589
+ write_page(header, "".b, rep_bytes + def_bytes + compressed)
590
+ end
591
+
592
+ def statistics_for(col, defs, values)
593
+ max_def = col.max_definition_level
594
+ nulls = max_def.zero? ? 0 : defs.size - defs.count(max_def)
595
+ stats = Format::Statistics.new(null_count: nulls)
596
+ min, max = min_max(col, values)
597
+ if min
598
+ stats.min_value = min
599
+ stats.max_value = max
600
+ stats.is_min_value_exact = true
601
+ stats.is_max_value_exact = true
602
+ end
603
+ stats
604
+ end
605
+
606
+ # Min/max for types whose Parquet sort order matches Ruby's comparison of the physical values
607
+ def min_max(col, values)
608
+ return nil if values.empty?
609
+ kind, _, signed = Types.logical_of(col.node)
610
+ case col.type
611
+ when T::BOOLEAN
612
+ [values.include?(false) ? "\x00".b : "\x01".b, values.include?(true) ? "\x01".b : "\x00".b]
613
+ when T::INT32, T::INT64
614
+ return nil if kind == :integer && !signed
615
+ fmt = col.type == T::INT32 ? "l<" : "q<"
616
+ min, max = values.minmax
617
+ [[min].pack(fmt), [max].pack(fmt)]
618
+ when T::FLOAT, T::DOUBLE
619
+ finite = values.reject(&:nan?)
620
+ return nil if finite.empty?
621
+ min, max = finite.minmax
622
+ min = -0.0 if min.zero?
623
+ max = 0.0 if max.zero?
624
+ fmt = col.type == T::FLOAT ? "e" : "E"
625
+ [[min].pack(fmt), [max].pack(fmt)]
626
+ when T::BYTE_ARRAY
627
+ return nil if kind == :decimal
628
+ min, max = values.minmax
629
+ return nil if min.bytesize > 1024 || max.bytesize > 1024
630
+ [min, max]
631
+ end
632
+ end
633
+ end
634
+ end
@@ -0,0 +1,45 @@
1
+ # frozen_string_literal: true
2
+
3
+ require "stringio"
4
+ require "pathname"
5
+ require_relative "herringbone/version"
6
+
7
+ # Pure-Ruby reader and writer for Apache Parquet files
8
+ module Herringbone
9
+ class Error < StandardError; end
10
+ class FormatError < Error; end
11
+ class DecodeError < FormatError; end
12
+ class EncodeError < Error; end
13
+ class UnsupportedError < Error; end
14
+ end
15
+
16
+ require_relative "herringbone/io_buffer_support"
17
+ require_relative "herringbone/codecs/snappy"
18
+ require_relative "herringbone/codecs/lz4"
19
+ require_relative "herringbone/thrift"
20
+ require_relative "herringbone/format"
21
+ require_relative "herringbone/encodings/rle"
22
+ require_relative "herringbone/encodings/plain"
23
+ require_relative "herringbone/encodings/delta"
24
+ require_relative "herringbone/compression"
25
+ require_relative "herringbone/types"
26
+ require_relative "herringbone/schema"
27
+ require_relative "herringbone/active_record"
28
+ require_relative "herringbone/reader"
29
+ require_relative "herringbone/byte_values"
30
+ require_relative "herringbone/writer"
31
+
32
+ module Herringbone
33
+ Delta = Encodings::Delta
34
+
35
+ module_function
36
+
37
+ def open(path, &block) = Reader.open(path, &block)
38
+ def read(path, columns: nil) = Reader.open(path) { |r| r.rows(columns: columns) }
39
+
40
+ # Writes an Enumerable of row Hashes. Without a schema, one is inferred from the first rows.
41
+ def write(path, rows, schema: nil, **options)
42
+ schema ||= Schema.infer(rows)
43
+ Writer.open(path, schema, **options) { |w| w.write_rows(rows) }
44
+ end
45
+ end