herringbone 0.4.0 → 0.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +14 -0
- data/README.md +150 -16
- data/lib/herringbone/active_record.rb +1 -1
- data/lib/herringbone/byte_values.rb +9 -0
- data/lib/herringbone/inferring_writer.rb +7 -0
- data/lib/herringbone/redaction/rewriter.rb +533 -0
- data/lib/herringbone/redaction.rb +287 -0
- data/lib/herringbone/schema.rb +121 -57
- data/lib/herringbone/simple_writer.rb +14 -4
- data/lib/herringbone/version.rb +1 -1
- data/lib/herringbone/writer.rb +141 -25
- data/lib/herringbone.rb +47 -5
- metadata +3 -1
data/lib/herringbone/writer.rb
CHANGED
|
@@ -3,10 +3,10 @@
|
|
|
3
3
|
module Herringbone
|
|
4
4
|
# Writes Parquet files.
|
|
5
5
|
#
|
|
6
|
-
# schema = Herringbone::Schema.define do
|
|
7
|
-
# int64 :id, null: false
|
|
8
|
-
# string :name
|
|
9
|
-
# list :tags, :string
|
|
6
|
+
# schema = Herringbone::Schema.define do |s|
|
|
7
|
+
# s.int64 :id, null: false
|
|
8
|
+
# s.string :name
|
|
9
|
+
# s.list :tags, :string
|
|
10
10
|
# end
|
|
11
11
|
# File.open("out.parquet", "wb") do |file|
|
|
12
12
|
# Herringbone::Writer.open(file, schema) do |w|
|
|
@@ -164,6 +164,7 @@ module Herringbone
|
|
|
164
164
|
@total_rows = 0
|
|
165
165
|
@pos = 0
|
|
166
166
|
@closed = false
|
|
167
|
+
@aborted = false
|
|
167
168
|
@bytes_per_row = nil
|
|
168
169
|
@io = check_io!(io)
|
|
169
170
|
write_raw(MAGIC)
|
|
@@ -222,21 +223,82 @@ module Herringbone
|
|
|
222
223
|
# @return [void]
|
|
223
224
|
def flush_row_group
|
|
224
225
|
return if @buffered_rows.zero?
|
|
226
|
+
write_row_group(@buffered_rows)
|
|
227
|
+
end
|
|
228
|
+
|
|
229
|
+
# A column chunk taken as it is from another file, for #write_row_group
|
|
230
|
+
#
|
|
231
|
+
# @!attribute chunk
|
|
232
|
+
# @return [Format::ColumnChunk] the chunk's footer entry in the source file
|
|
233
|
+
# @!attribute start
|
|
234
|
+
# @return [Integer] source file offset of the chunk's first page
|
|
235
|
+
# @!attribute bytes
|
|
236
|
+
# @return [String] the chunk's pages, headers included
|
|
237
|
+
# @!attribute column_index
|
|
238
|
+
# @return [String, nil] the chunk's encoded ColumnIndex
|
|
239
|
+
# @!attribute offset_index
|
|
240
|
+
# @return [Format::OffsetIndex, nil] the chunk's OffsetIndex, with source file offsets
|
|
241
|
+
# @!attribute bloom_filter
|
|
242
|
+
# @return [String, nil] the chunk's encoded bloom filter, header included
|
|
243
|
+
CopiedChunk = Struct.new(:chunk, :start, :bytes, :column_index, :offset_index, :bloom_filter)
|
|
244
|
+
|
|
245
|
+
# Internal (used by Redaction): writes a row group of +num_rows+ rows from the buffered values,
|
|
246
|
+
# except for the columns in +copies+, whose chunks are copied byte for byte from another file
|
|
247
|
+
# with their offsets rebased. Starts a new row group afterwards.
|
|
248
|
+
#
|
|
249
|
+
# @param num_rows [Integer] rows in the row group
|
|
250
|
+
# @param copies [Hash{Integer => CopiedChunk}] column index => chunk to copy instead of encoding
|
|
251
|
+
# @param codecs [Hash{Integer => Integer}] column index => codec id for an encoded chunk, instead
|
|
252
|
+
# of the +compression:+ option
|
|
253
|
+
# @param bloom_filters [Hash{Integer => Boolean}] column index => true to give an encoded chunk a
|
|
254
|
+
# bloom filter (default settings) when the +bloom_filters:+ option does not ask for one
|
|
255
|
+
# @param sorting_columns [Array<Format::SortingColumn>, nil] stored on the RowGroup as given
|
|
256
|
+
# @return [void]
|
|
257
|
+
# @raise [Error] when the writer is closed
|
|
258
|
+
def write_row_group(num_rows, copies: {}, codecs: {}, bloom_filters: {}, sorting_columns: nil)
|
|
259
|
+
raise Error, "Writer is closed" if @closed
|
|
225
260
|
start = @pos
|
|
226
|
-
chunks = @schema.columns.map
|
|
261
|
+
chunks = @schema.columns.map do |col|
|
|
262
|
+
if (copy = copies[col.index])
|
|
263
|
+
copy_column_chunk(copy)
|
|
264
|
+
else
|
|
265
|
+
bloom = @bloom_filters[col.dotted_path]
|
|
266
|
+
bloom ||= bloom_filter_settings({}) if bloom_filters[col.index] && BloomFilter::TYPES.include?(col.type)
|
|
267
|
+
write_column_chunk(col, @buffers[col.index], codec: codecs.fetch(col.index, @codec), bloom: bloom)
|
|
268
|
+
end
|
|
269
|
+
end
|
|
227
270
|
@row_groups << Format::RowGroup.new(
|
|
228
271
|
columns: chunks,
|
|
229
272
|
total_byte_size: chunks.sum { |c| c.meta_data.total_uncompressed_size },
|
|
230
|
-
num_rows:
|
|
273
|
+
num_rows: num_rows,
|
|
274
|
+
sorting_columns: sorting_columns,
|
|
231
275
|
file_offset: start,
|
|
232
276
|
total_compressed_size: @pos - start,
|
|
233
277
|
ordinal: @row_groups.size
|
|
234
278
|
)
|
|
235
279
|
write_bloom_filters
|
|
236
|
-
@total_rows +=
|
|
280
|
+
@total_rows += num_rows
|
|
237
281
|
reset_buffers
|
|
238
282
|
end
|
|
239
283
|
|
|
284
|
+
# Internal (used by Redaction): shreds one value per row of a top-level field into the buffers,
|
|
285
|
+
# for #write_row_group. Other fields are left as they are, so the caller decides which columns
|
|
286
|
+
# are encoded and which are copied.
|
|
287
|
+
#
|
|
288
|
+
# @param name [String] top-level field name
|
|
289
|
+
# @param values [Array] the field's value for each row
|
|
290
|
+
# @return [void]
|
|
291
|
+
# @raise [ArgumentError] when the schema has no such field
|
|
292
|
+
# @raise [EncodeError] when a value cannot be written to the field
|
|
293
|
+
def buffer_field(name, values)
|
|
294
|
+
field = @schema.field(name) or raise ArgumentError, "No such field #{name.inspect}"
|
|
295
|
+
values.each_with_index do |value, i|
|
|
296
|
+
shred(field, value, 0, 0)
|
|
297
|
+
rescue EncodeError => e
|
|
298
|
+
raise EncodeError.new("Row #{@total_rows + i}: #{e.message}", row: @total_rows + i, column: e.column, value: e.value)
|
|
299
|
+
end
|
|
300
|
+
end
|
|
301
|
+
|
|
240
302
|
# Flushes buffered rows, then writes the page indexes and the footer and flushes the IO.
|
|
241
303
|
# Does nothing when already closed or aborted. The IO itself is not closed.
|
|
242
304
|
#
|
|
@@ -267,9 +329,23 @@ module Herringbone
|
|
|
267
329
|
#
|
|
268
330
|
# @return [void]
|
|
269
331
|
def abort
|
|
332
|
+
@aborted = true unless @closed
|
|
270
333
|
@closed = true
|
|
271
334
|
end
|
|
272
335
|
|
|
336
|
+
# Short summary for the console, without the buffered values
|
|
337
|
+
#
|
|
338
|
+
# @return [String] state (open, closed or aborted), rows written including buffered ones,
|
|
339
|
+
# row groups flushed and the codec
|
|
340
|
+
def inspect
|
|
341
|
+
state = if @aborted then "aborted"
|
|
342
|
+
elsif @closed then "closed"
|
|
343
|
+
else "open"
|
|
344
|
+
end
|
|
345
|
+
"#<#{self.class.name} #{state} rows_written=#{rows_written} row_groups=#{@row_groups.size} " \
|
|
346
|
+
"compression=#{Compression::NAMES.fetch(@codec, @codec).inspect}>"
|
|
347
|
+
end
|
|
348
|
+
|
|
273
349
|
private
|
|
274
350
|
|
|
275
351
|
# Pathname responds to #write too (writing a whole file by path), so it is rejected explicitly.
|
|
@@ -534,8 +610,11 @@ module Herringbone
|
|
|
534
610
|
#
|
|
535
611
|
# @param col [Schema::Column] column being written
|
|
536
612
|
# @param buffer [ColumnBuffer] the column's buffered levels and values
|
|
613
|
+
# @param codec [Integer] codec id to compress the pages with
|
|
614
|
+
# @param bloom [Hash{Symbol => Numeric, nil}, nil] bloom filter settings (see
|
|
615
|
+
# #bloom_filter_settings), nil for no bloom filter
|
|
537
616
|
# @return [Format::ColumnChunk] chunk with its ColumnMetaData, for the row group
|
|
538
|
-
def write_column_chunk(col, buffer)
|
|
617
|
+
def write_column_chunk(col, buffer, codec: @codec, bloom: @bloom_filters[col.dotted_path])
|
|
539
618
|
type = col.type
|
|
540
619
|
path = col.dotted_path
|
|
541
620
|
dict_values = nil
|
|
@@ -568,7 +647,7 @@ module Herringbone
|
|
|
568
647
|
type: Format::PageType::DICTIONARY_PAGE,
|
|
569
648
|
dictionary_page_header: Format::DictionaryPageHeader.new(num_values: dict_values.size, encoding: E::PLAIN)
|
|
570
649
|
)
|
|
571
|
-
uncompressed_total += write_page(header, plain)
|
|
650
|
+
uncompressed_total += write_page(header, plain, codec: codec)
|
|
572
651
|
end
|
|
573
652
|
|
|
574
653
|
data_offset = @pos
|
|
@@ -602,10 +681,10 @@ module Herringbone
|
|
|
602
681
|
rep_bytes = max_rep.positive? ? Encodings::RLE.encode_hybrid(reps, max_rep.bit_length) : "".b
|
|
603
682
|
def_bytes = max_def.positive? ? Encodings::RLE.encode_hybrid(defs, max_def.bit_length) : "".b
|
|
604
683
|
uncompressed_total += if @data_page_version == 1
|
|
605
|
-
write_data_page_v1(n, rep_bytes, def_bytes, encoded, value_encoding)
|
|
684
|
+
write_data_page_v1(n, rep_bytes, def_bytes, encoded, value_encoding, codec)
|
|
606
685
|
else
|
|
607
686
|
num_rows = max_rep.zero? ? n : reps.count(0)
|
|
608
|
-
write_data_page_v2(n, n - non_null, num_rows, rep_bytes, def_bytes, encoded, value_encoding)
|
|
687
|
+
write_data_page_v2(n, n - non_null, num_rows, rep_bytes, def_bytes, encoded, value_encoding, codec)
|
|
609
688
|
end
|
|
610
689
|
pages << PageInfo.new(page_offset, @pos - page_offset, first_row, n - non_null, non_null, range)
|
|
611
690
|
first_row += max_rep.zero? ? n : reps.count(0)
|
|
@@ -618,7 +697,7 @@ module Herringbone
|
|
|
618
697
|
type: type,
|
|
619
698
|
encodings: encodings.uniq,
|
|
620
699
|
path_in_schema: col.path,
|
|
621
|
-
codec:
|
|
700
|
+
codec: codec,
|
|
622
701
|
num_values: buf.defs.size,
|
|
623
702
|
total_uncompressed_size: uncompressed_total,
|
|
624
703
|
total_compressed_size: @pos - chunk_start,
|
|
@@ -627,7 +706,7 @@ module Herringbone
|
|
|
627
706
|
statistics: statistics_for(col, buf.defs, values || indices.uniq.map { |i| dict_values[i] }, order)
|
|
628
707
|
)
|
|
629
708
|
chunk = Format::ColumnChunk.new(file_offset: chunk_start, meta_data: meta)
|
|
630
|
-
if
|
|
709
|
+
if bloom
|
|
631
710
|
@pending_bloom_filters << [meta, build_bloom_filter(col, bloom, dict_values, values)]
|
|
632
711
|
end
|
|
633
712
|
@page_indexes << [chunk, column_index_for(col, pages, order), offset_index_for(pages)]
|
|
@@ -658,6 +737,37 @@ module Herringbone
|
|
|
658
737
|
end)
|
|
659
738
|
end
|
|
660
739
|
|
|
740
|
+
# Writes a chunk copied from another file. The pages, the ColumnIndex and the bloom filter
|
|
741
|
+
# hold no file offsets and go out as they are; the ColumnMetaData and the OffsetIndex do, so
|
|
742
|
+
# those are rebased onto where the chunk lands in this file.
|
|
743
|
+
#
|
|
744
|
+
# @param copy [CopiedChunk] the chunk to copy
|
|
745
|
+
# @return [Format::ColumnChunk] chunk with its ColumnMetaData, for the row group
|
|
746
|
+
def copy_column_chunk(copy)
|
|
747
|
+
start = @pos
|
|
748
|
+
shift = start - copy.start
|
|
749
|
+
meta = Format::ColumnMetaData.decode(copy.chunk.meta_data.encode).first
|
|
750
|
+
meta.data_page_offset += shift
|
|
751
|
+
dict = meta.dictionary_page_offset
|
|
752
|
+
# Some writers store 0 when there is no dictionary page
|
|
753
|
+
meta.dictionary_page_offset = dict&.positive? ? dict + shift : nil
|
|
754
|
+
meta.index_page_offset += shift if meta.index_page_offset
|
|
755
|
+
meta.total_compressed_size = copy.bytes.bytesize
|
|
756
|
+
meta.bloom_filter_offset = meta.bloom_filter_length = nil
|
|
757
|
+
write_raw(copy.bytes)
|
|
758
|
+
chunk = Format::ColumnChunk.new(file_offset: start, meta_data: meta)
|
|
759
|
+
@pending_bloom_filters << [meta, copy.bloom_filter] if copy.bloom_filter
|
|
760
|
+
offset_index = copy.offset_index && Format::OffsetIndex.new(
|
|
761
|
+
page_locations: copy.offset_index.page_locations.map do |loc|
|
|
762
|
+
Format::PageLocation.new(offset: loc.offset + shift, compressed_page_size: loc.compressed_page_size,
|
|
763
|
+
first_row_index: loc.first_row_index)
|
|
764
|
+
end,
|
|
765
|
+
unencoded_byte_array_data_bytes: copy.offset_index.unencoded_byte_array_data_bytes
|
|
766
|
+
)
|
|
767
|
+
@page_indexes << [chunk, copy.column_index, offset_index]
|
|
768
|
+
chunk
|
|
769
|
+
end
|
|
770
|
+
|
|
661
771
|
# nil when the column has no defined sort order, or a page's values have no min/max (all NaN)
|
|
662
772
|
#
|
|
663
773
|
# @param col [Schema::Column] column the pages belong to
|
|
@@ -693,18 +803,20 @@ module Herringbone
|
|
|
693
803
|
end
|
|
694
804
|
end
|
|
695
805
|
|
|
696
|
-
# Page indexes go after the last row group: all column indexes, then all offset indexes
|
|
806
|
+
# Page indexes go after the last row group: all column indexes, then all offset indexes.
|
|
807
|
+
# A copied chunk brings its ColumnIndex already encoded, and may come without either index.
|
|
697
808
|
#
|
|
698
809
|
# @return [void]
|
|
699
810
|
def write_page_indexes
|
|
700
811
|
@page_indexes.each do |chunk, column_index, _|
|
|
701
812
|
next unless column_index
|
|
702
|
-
bytes = column_index.encode
|
|
813
|
+
bytes = column_index.is_a?(String) ? column_index : column_index.encode
|
|
703
814
|
chunk.column_index_offset = @pos
|
|
704
815
|
chunk.column_index_length = bytes.bytesize
|
|
705
816
|
write_raw(bytes)
|
|
706
817
|
end
|
|
707
818
|
@page_indexes.each do |chunk, _, offset_index|
|
|
819
|
+
next unless offset_index
|
|
708
820
|
bytes = offset_index.encode
|
|
709
821
|
chunk.offset_index_offset = @pos
|
|
710
822
|
chunk.offset_index_length = bytes.bytesize
|
|
@@ -780,12 +892,13 @@ module Herringbone
|
|
|
780
892
|
filter
|
|
781
893
|
end
|
|
782
894
|
|
|
783
|
-
# Bloom filters go right after the row group's column chunks, in column order
|
|
895
|
+
# Bloom filters go right after the row group's column chunks, in column order. A copied
|
|
896
|
+
# chunk brings its filter already encoded.
|
|
784
897
|
#
|
|
785
898
|
# @return [void]
|
|
786
899
|
def write_bloom_filters
|
|
787
900
|
@pending_bloom_filters.each do |meta, filter|
|
|
788
|
-
bytes = filter.encode
|
|
901
|
+
bytes = filter.is_a?(String) ? filter : filter.encode
|
|
789
902
|
meta.bloom_filter_offset = @pos
|
|
790
903
|
meta.bloom_filter_length = bytes.bytesize
|
|
791
904
|
write_raw(bytes)
|
|
@@ -906,9 +1019,10 @@ module Herringbone
|
|
|
906
1019
|
# @param body [String] uncompressed page body
|
|
907
1020
|
# @param compressed [String, nil] bytes to write as the page body, when already prepared (v2 data
|
|
908
1021
|
# pages, whose levels stay uncompressed); +body+ is compressed otherwise
|
|
1022
|
+
# @param codec [Integer] codec id to compress +body+ with
|
|
909
1023
|
# @return [Integer]
|
|
910
|
-
def write_page(header, body, compressed = nil)
|
|
911
|
-
compressed ||= Compression.compress(
|
|
1024
|
+
def write_page(header, body, compressed = nil, codec: @codec)
|
|
1025
|
+
compressed ||= Compression.compress(codec, body)
|
|
912
1026
|
header.uncompressed_page_size ||= body.bytesize
|
|
913
1027
|
header.compressed_page_size = compressed.bytesize
|
|
914
1028
|
header.crc = Zlib.crc32(compressed).then { |c| (c >= 0x8000_0000) ? c - 0x1_0000_0000 : c }
|
|
@@ -925,8 +1039,9 @@ module Herringbone
|
|
|
925
1039
|
# @param def_bytes [String] RLE-encoded definition levels, empty when the column has none
|
|
926
1040
|
# @param encoded [String] encoded values
|
|
927
1041
|
# @param encoding [Integer] encoding id of the values
|
|
1042
|
+
# @param codec [Integer] codec id to compress the page with
|
|
928
1043
|
# @return [Integer] uncompressed size including the header
|
|
929
|
-
def write_data_page_v1(n, rep_bytes, def_bytes, encoded, encoding)
|
|
1044
|
+
def write_data_page_v1(n, rep_bytes, def_bytes, encoded, encoding, codec)
|
|
930
1045
|
body = String.new(encoding: Encoding::BINARY)
|
|
931
1046
|
body << [rep_bytes.bytesize].pack("V") << rep_bytes unless rep_bytes.empty?
|
|
932
1047
|
body << [def_bytes.bytesize].pack("V") << def_bytes unless def_bytes.empty?
|
|
@@ -938,7 +1053,7 @@ module Herringbone
|
|
|
938
1053
|
definition_level_encoding: E::RLE, repetition_level_encoding: E::RLE
|
|
939
1054
|
)
|
|
940
1055
|
)
|
|
941
|
-
write_page(header, body)
|
|
1056
|
+
write_page(header, body, codec: codec)
|
|
942
1057
|
end
|
|
943
1058
|
|
|
944
1059
|
# A DATA_PAGE_V2: levels without length prefixes and uncompressed, then the compressed values
|
|
@@ -950,9 +1065,10 @@ module Herringbone
|
|
|
950
1065
|
# @param def_bytes [String] RLE-encoded definition levels, empty when the column has none
|
|
951
1066
|
# @param encoded [String] encoded values
|
|
952
1067
|
# @param encoding [Integer] encoding id of the values
|
|
1068
|
+
# @param codec [Integer] codec id to compress the values with
|
|
953
1069
|
# @return [Integer] uncompressed size including the header
|
|
954
|
-
def write_data_page_v2(n, nulls, rows, rep_bytes, def_bytes, encoded, encoding)
|
|
955
|
-
compressed = Compression.compress(
|
|
1070
|
+
def write_data_page_v2(n, nulls, rows, rep_bytes, def_bytes, encoded, encoding, codec)
|
|
1071
|
+
compressed = Compression.compress(codec, encoded)
|
|
956
1072
|
header = Format::PageHeader.new(
|
|
957
1073
|
type: Format::PageType::DATA_PAGE_V2,
|
|
958
1074
|
uncompressed_page_size: rep_bytes.bytesize + def_bytes.bytesize + encoded.bytesize,
|
|
@@ -960,10 +1076,10 @@ module Herringbone
|
|
|
960
1076
|
num_values: n, num_nulls: nulls, num_rows: rows, encoding: encoding,
|
|
961
1077
|
definition_levels_byte_length: def_bytes.bytesize,
|
|
962
1078
|
repetition_levels_byte_length: rep_bytes.bytesize,
|
|
963
|
-
is_compressed:
|
|
1079
|
+
is_compressed: codec != Format::Codec::UNCOMPRESSED
|
|
964
1080
|
)
|
|
965
1081
|
)
|
|
966
|
-
write_page(header, "".b, rep_bytes + def_bytes + compressed)
|
|
1082
|
+
write_page(header, "".b, rep_bytes + def_bytes + compressed, codec: codec)
|
|
967
1083
|
end
|
|
968
1084
|
|
|
969
1085
|
# Statistics and column index bounds longer than this are truncated, see #truncate_min
|
data/lib/herringbone.rb
CHANGED
|
@@ -64,6 +64,7 @@ require_relative "herringbone/xxhash"
|
|
|
64
64
|
require_relative "herringbone/bloom_filter"
|
|
65
65
|
require_relative "herringbone/inspector"
|
|
66
66
|
require_relative "herringbone/visualizer"
|
|
67
|
+
require_relative "herringbone/redaction"
|
|
67
68
|
|
|
68
69
|
module Herringbone
|
|
69
70
|
module_function
|
|
@@ -77,7 +78,7 @@ module Herringbone
|
|
|
77
78
|
# to Writer.
|
|
78
79
|
#
|
|
79
80
|
# File.open("orders.parquet", "wb") { |f| Herringbone.write(f, Order.where(created_at: 1.year.ago..)) }
|
|
80
|
-
# Herringbone.write(io, events.lazy.map(&:to_h)) { json :payload }
|
|
81
|
+
# Herringbone.write(io, events.lazy.map(&:to_h)) { |s| s.json :payload }
|
|
81
82
|
#
|
|
82
83
|
# @param io [IO, #write] destination; written sequentially, never closed
|
|
83
84
|
# @param records [Enumerable<Hash, Array, Object>, Class, #find_each] rows (Hashes, Arrays in schema
|
|
@@ -97,14 +98,17 @@ module Herringbone
|
|
|
97
98
|
# @option options [Hash{String => String}] :metadata ({}) key/value metadata for the footer
|
|
98
99
|
# @option options [Boolean, Array<String>, Hash{String => Boolean, Hash}] :bloom_filters (nil)
|
|
99
100
|
# columns to write split block bloom filters for, see Writer
|
|
100
|
-
# @yield optional
|
|
101
|
-
#
|
|
101
|
+
# @yield [s] optional, declares fields that replace inferred ones (see Schema.infer); ignored when
|
|
102
|
+
# the schema is not inferred
|
|
103
|
+
# @yieldparam s [Schema::Builder] the builder to declare fields on
|
|
104
|
+
# @yieldreturn [void]
|
|
102
105
|
# @return [Integer] number of rows written
|
|
103
106
|
# @raise [ArgumentError] when the schema has to be inferred and +records+ is empty or holds Array
|
|
104
|
-
# rows,
|
|
107
|
+
# rows, an option is invalid, or the block takes no parameter
|
|
105
108
|
# @raise [EncodeError] when a row does not fit the schema
|
|
106
109
|
# @raise [SchemaMismatch] when a row does not fit the inferred schema; the file is left unfinished
|
|
107
110
|
def write(io, records, schema: nil, **options, &overrides)
|
|
111
|
+
Schema::Builder.check_block!(overrides, "Herringbone.write(io, rows) { |s| s.json :payload }")
|
|
108
112
|
model = if records.respond_to?(:klass) then records.klass
|
|
109
113
|
elsif records.respond_to?(:columns) && records.respond_to?(:find_each) then records
|
|
110
114
|
end
|
|
@@ -112,7 +116,7 @@ module Herringbone
|
|
|
112
116
|
writer = if schema
|
|
113
117
|
Writer.new(io, schema, **options)
|
|
114
118
|
else
|
|
115
|
-
fix = "Herringbone.write(io, rows) {
|
|
119
|
+
fix = "Herringbone.write(io, rows) { |s| s.%s }"
|
|
116
120
|
InferringWriter.new(io, fix: fix, **options) { |sample| Schema.infer(sample, &overrides) }
|
|
117
121
|
end
|
|
118
122
|
begin
|
|
@@ -129,6 +133,44 @@ module Herringbone
|
|
|
129
133
|
writer.rows_written
|
|
130
134
|
end
|
|
131
135
|
|
|
136
|
+
# Rewrites +input+ into +output+ with rows removed or values replaced, building the Redaction from
|
|
137
|
+
# the block (or taking +redaction+). A shortcut for Redaction#apply.
|
|
138
|
+
#
|
|
139
|
+
# Herringbone.redact(input, output) do |r|
|
|
140
|
+
# r.where(user_id: 42).delete
|
|
141
|
+
# r.replace(:email) { |email| email && OpenSSL::HMAC.hexdigest("SHA256", KEY, email) }
|
|
142
|
+
# end
|
|
143
|
+
#
|
|
144
|
+
# @param input [IO, StringIO] the Parquet file, read with #seek and #read; not closed
|
|
145
|
+
# @param output [IO, #write] destination, written sequentially; not closed
|
|
146
|
+
# @param redaction [Redaction, nil] the redaction to apply, instead of a block
|
|
147
|
+
# @param writer_options [Hash{Symbol => Object}] Writer options for re-encoded column chunks, and
|
|
148
|
+
# +metadata:+ to replace the footer key/value metadata, see Redaction#apply
|
|
149
|
+
# @option writer_options [Symbol] :compression (codec of each source chunk) codec for re-encoded chunks
|
|
150
|
+
# @option writer_options [Boolean, Array<String>, Hash{String => Boolean, Hash}] :bloom_filters (nil)
|
|
151
|
+
# columns whose re-encoded chunks get a bloom filter, besides those whose source chunk had one
|
|
152
|
+
# @option writer_options [Hash{String => String}] :metadata (the input's) footer key/value metadata
|
|
153
|
+
# @option writer_options [Integer] :page_bytes (1MB) approximate uncompressed data page size
|
|
154
|
+
# @option writer_options [Integer] :page_rows (20_000) maximum rows per data page
|
|
155
|
+
# @option writer_options [Integer] :data_page_version (1) 1 or 2
|
|
156
|
+
# @option writer_options [Boolean, Array<String>] :dictionary (true) see Writer
|
|
157
|
+
# @option writer_options [Hash{String => Symbol}] :encodings ({}) see Writer
|
|
158
|
+
# @yield [r] declares the statements on a new Redaction (see Redaction.new)
|
|
159
|
+
# @yieldparam r [Redaction] the redaction being built
|
|
160
|
+
# @yieldreturn [void]
|
|
161
|
+
# @return [Redaction::Report] what was done
|
|
162
|
+
# @raise [ArgumentError] when given both or neither of +redaction+ and a block, a block that takes
|
|
163
|
+
# no parameter, or when the redaction does not fit the file's schema
|
|
164
|
+
# @raise [EncodeError] when a replacement value cannot be written to its column
|
|
165
|
+
def redact(input, output, redaction = nil, **writer_options, &block)
|
|
166
|
+
raise ArgumentError, "Herringbone.redact takes a Redaction or a block, not both" if redaction && block
|
|
167
|
+
raise ArgumentError, "Herringbone.redact needs a Redaction or a block" unless redaction || block
|
|
168
|
+
if block&.arity&.zero?
|
|
169
|
+
raise ArgumentError, "The block receives the redaction: Herringbone.redact(input, output) { |r| r.where(user_id: 42).delete }"
|
|
170
|
+
end
|
|
171
|
+
(redaction || Redaction.new(&block)).apply(input, output, **writer_options)
|
|
172
|
+
end
|
|
173
|
+
|
|
132
174
|
# Compression codecs this process can read and write, e.g. [:none, :snappy, :gzip, :lz4, :lz4_hadoop, :zstd].
|
|
133
175
|
# :zstd and :brotli are listed when the zstd-ruby / brotli gems are loaded.
|
|
134
176
|
#
|
metadata
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
--- !ruby/object:Gem::Specification
|
|
2
2
|
name: herringbone
|
|
3
3
|
version: !ruby/object:Gem::Version
|
|
4
|
-
version: 0.
|
|
4
|
+
version: 0.5.0
|
|
5
5
|
platform: ruby
|
|
6
6
|
authors:
|
|
7
7
|
- Julik Tarkhanov
|
|
@@ -58,6 +58,8 @@ files:
|
|
|
58
58
|
- lib/herringbone/reader/numo.rb
|
|
59
59
|
- lib/herringbone/reader/page_stream.rb
|
|
60
60
|
- lib/herringbone/reader/scan.rb
|
|
61
|
+
- lib/herringbone/redaction.rb
|
|
62
|
+
- lib/herringbone/redaction/rewriter.rb
|
|
61
63
|
- lib/herringbone/schema.rb
|
|
62
64
|
- lib/herringbone/simple_writer.rb
|
|
63
65
|
- lib/herringbone/thrift.rb
|