herringbone 0.4.0 → 0.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -3,10 +3,10 @@
3
3
  module Herringbone
4
4
  # Writes Parquet files.
5
5
  #
6
- # schema = Herringbone::Schema.define do
7
- # int64 :id, null: false
8
- # string :name
9
- # list :tags, :string
6
+ # schema = Herringbone::Schema.define do |s|
7
+ # s.int64 :id, null: false
8
+ # s.string :name
9
+ # s.list :tags, :string
10
10
  # end
11
11
  # File.open("out.parquet", "wb") do |file|
12
12
  # Herringbone::Writer.open(file, schema) do |w|
@@ -164,6 +164,7 @@ module Herringbone
164
164
  @total_rows = 0
165
165
  @pos = 0
166
166
  @closed = false
167
+ @aborted = false
167
168
  @bytes_per_row = nil
168
169
  @io = check_io!(io)
169
170
  write_raw(MAGIC)
@@ -222,21 +223,82 @@ module Herringbone
222
223
  # @return [void]
223
224
  def flush_row_group
224
225
  return if @buffered_rows.zero?
226
+ write_row_group(@buffered_rows)
227
+ end
228
+
229
+ # A column chunk taken as it is from another file, for #write_row_group
230
+ #
231
+ # @!attribute chunk
232
+ # @return [Format::ColumnChunk] the chunk's footer entry in the source file
233
+ # @!attribute start
234
+ # @return [Integer] source file offset of the chunk's first page
235
+ # @!attribute bytes
236
+ # @return [String] the chunk's pages, headers included
237
+ # @!attribute column_index
238
+ # @return [String, nil] the chunk's encoded ColumnIndex
239
+ # @!attribute offset_index
240
+ # @return [Format::OffsetIndex, nil] the chunk's OffsetIndex, with source file offsets
241
+ # @!attribute bloom_filter
242
+ # @return [String, nil] the chunk's encoded bloom filter, header included
243
+ CopiedChunk = Struct.new(:chunk, :start, :bytes, :column_index, :offset_index, :bloom_filter)
244
+
245
+ # Internal (used by Redaction): writes a row group of +num_rows+ rows from the buffered values,
246
+ # except for the columns in +copies+, whose chunks are copied byte for byte from another file
247
+ # with their offsets rebased. Starts a new row group afterwards.
248
+ #
249
+ # @param num_rows [Integer] rows in the row group
250
+ # @param copies [Hash{Integer => CopiedChunk}] column index => chunk to copy instead of encoding
251
+ # @param codecs [Hash{Integer => Integer}] column index => codec id for an encoded chunk, instead
252
+ # of the +compression:+ option
253
+ # @param bloom_filters [Hash{Integer => Boolean}] column index => true to give an encoded chunk a
254
+ # bloom filter (default settings) when the +bloom_filters:+ option does not ask for one
255
+ # @param sorting_columns [Array<Format::SortingColumn>, nil] stored on the RowGroup as given
256
+ # @return [void]
257
+ # @raise [Error] when the writer is closed
258
+ def write_row_group(num_rows, copies: {}, codecs: {}, bloom_filters: {}, sorting_columns: nil)
259
+ raise Error, "Writer is closed" if @closed
225
260
  start = @pos
226
- chunks = @schema.columns.map { |col| write_column_chunk(col, @buffers[col.index]) }
261
+ chunks = @schema.columns.map do |col|
262
+ if (copy = copies[col.index])
263
+ copy_column_chunk(copy)
264
+ else
265
+ bloom = @bloom_filters[col.dotted_path]
266
+ bloom ||= bloom_filter_settings({}) if bloom_filters[col.index] && BloomFilter::TYPES.include?(col.type)
267
+ write_column_chunk(col, @buffers[col.index], codec: codecs.fetch(col.index, @codec), bloom: bloom)
268
+ end
269
+ end
227
270
  @row_groups << Format::RowGroup.new(
228
271
  columns: chunks,
229
272
  total_byte_size: chunks.sum { |c| c.meta_data.total_uncompressed_size },
230
- num_rows: @buffered_rows,
273
+ num_rows: num_rows,
274
+ sorting_columns: sorting_columns,
231
275
  file_offset: start,
232
276
  total_compressed_size: @pos - start,
233
277
  ordinal: @row_groups.size
234
278
  )
235
279
  write_bloom_filters
236
- @total_rows += @buffered_rows
280
+ @total_rows += num_rows
237
281
  reset_buffers
238
282
  end
239
283
 
284
+ # Internal (used by Redaction): shreds one value per row of a top-level field into the buffers,
285
+ # for #write_row_group. Other fields are left as they are, so the caller decides which columns
286
+ # are encoded and which are copied.
287
+ #
288
+ # @param name [String] top-level field name
289
+ # @param values [Array] the field's value for each row
290
+ # @return [void]
291
+ # @raise [ArgumentError] when the schema has no such field
292
+ # @raise [EncodeError] when a value cannot be written to the field
293
+ def buffer_field(name, values)
294
+ field = @schema.field(name) or raise ArgumentError, "No such field #{name.inspect}"
295
+ values.each_with_index do |value, i|
296
+ shred(field, value, 0, 0)
297
+ rescue EncodeError => e
298
+ raise EncodeError.new("Row #{@total_rows + i}: #{e.message}", row: @total_rows + i, column: e.column, value: e.value)
299
+ end
300
+ end
301
+
240
302
  # Flushes buffered rows, then writes the page indexes and the footer and flushes the IO.
241
303
  # Does nothing when already closed or aborted. The IO itself is not closed.
242
304
  #
@@ -267,9 +329,23 @@ module Herringbone
267
329
  #
268
330
  # @return [void]
269
331
  def abort
332
+ @aborted = true unless @closed
270
333
  @closed = true
271
334
  end
272
335
 
336
+ # Short summary for the console, without the buffered values
337
+ #
338
+ # @return [String] state (open, closed or aborted), rows written including buffered ones,
339
+ # row groups flushed and the codec
340
+ def inspect
341
+ state = if @aborted then "aborted"
342
+ elsif @closed then "closed"
343
+ else "open"
344
+ end
345
+ "#<#{self.class.name} #{state} rows_written=#{rows_written} row_groups=#{@row_groups.size} " \
346
+ "compression=#{Compression::NAMES.fetch(@codec, @codec).inspect}>"
347
+ end
348
+
273
349
  private
274
350
 
275
351
  # Pathname responds to #write too (writing a whole file by path), so it is rejected explicitly.
@@ -534,8 +610,11 @@ module Herringbone
534
610
  #
535
611
  # @param col [Schema::Column] column being written
536
612
  # @param buffer [ColumnBuffer] the column's buffered levels and values
613
+ # @param codec [Integer] codec id to compress the pages with
614
+ # @param bloom [Hash{Symbol => Numeric, nil}, nil] bloom filter settings (see
615
+ # #bloom_filter_settings), nil for no bloom filter
537
616
  # @return [Format::ColumnChunk] chunk with its ColumnMetaData, for the row group
538
- def write_column_chunk(col, buffer)
617
+ def write_column_chunk(col, buffer, codec: @codec, bloom: @bloom_filters[col.dotted_path])
539
618
  type = col.type
540
619
  path = col.dotted_path
541
620
  dict_values = nil
@@ -568,7 +647,7 @@ module Herringbone
568
647
  type: Format::PageType::DICTIONARY_PAGE,
569
648
  dictionary_page_header: Format::DictionaryPageHeader.new(num_values: dict_values.size, encoding: E::PLAIN)
570
649
  )
571
- uncompressed_total += write_page(header, plain)
650
+ uncompressed_total += write_page(header, plain, codec: codec)
572
651
  end
573
652
 
574
653
  data_offset = @pos
@@ -602,10 +681,10 @@ module Herringbone
602
681
  rep_bytes = max_rep.positive? ? Encodings::RLE.encode_hybrid(reps, max_rep.bit_length) : "".b
603
682
  def_bytes = max_def.positive? ? Encodings::RLE.encode_hybrid(defs, max_def.bit_length) : "".b
604
683
  uncompressed_total += if @data_page_version == 1
605
- write_data_page_v1(n, rep_bytes, def_bytes, encoded, value_encoding)
684
+ write_data_page_v1(n, rep_bytes, def_bytes, encoded, value_encoding, codec)
606
685
  else
607
686
  num_rows = max_rep.zero? ? n : reps.count(0)
608
- write_data_page_v2(n, n - non_null, num_rows, rep_bytes, def_bytes, encoded, value_encoding)
687
+ write_data_page_v2(n, n - non_null, num_rows, rep_bytes, def_bytes, encoded, value_encoding, codec)
609
688
  end
610
689
  pages << PageInfo.new(page_offset, @pos - page_offset, first_row, n - non_null, non_null, range)
611
690
  first_row += max_rep.zero? ? n : reps.count(0)
@@ -618,7 +697,7 @@ module Herringbone
618
697
  type: type,
619
698
  encodings: encodings.uniq,
620
699
  path_in_schema: col.path,
621
- codec: @codec,
700
+ codec: codec,
622
701
  num_values: buf.defs.size,
623
702
  total_uncompressed_size: uncompressed_total,
624
703
  total_compressed_size: @pos - chunk_start,
@@ -627,7 +706,7 @@ module Herringbone
627
706
  statistics: statistics_for(col, buf.defs, values || indices.uniq.map { |i| dict_values[i] }, order)
628
707
  )
629
708
  chunk = Format::ColumnChunk.new(file_offset: chunk_start, meta_data: meta)
630
- if (bloom = @bloom_filters[path])
709
+ if bloom
631
710
  @pending_bloom_filters << [meta, build_bloom_filter(col, bloom, dict_values, values)]
632
711
  end
633
712
  @page_indexes << [chunk, column_index_for(col, pages, order), offset_index_for(pages)]
@@ -658,6 +737,37 @@ module Herringbone
658
737
  end)
659
738
  end
660
739
 
740
+ # Writes a chunk copied from another file. The pages, the ColumnIndex and the bloom filter
741
+ # hold no file offsets and go out as they are; the ColumnMetaData and the OffsetIndex do, so
742
+ # those are rebased onto where the chunk lands in this file.
743
+ #
744
+ # @param copy [CopiedChunk] the chunk to copy
745
+ # @return [Format::ColumnChunk] chunk with its ColumnMetaData, for the row group
746
+ def copy_column_chunk(copy)
747
+ start = @pos
748
+ shift = start - copy.start
749
+ meta = Format::ColumnMetaData.decode(copy.chunk.meta_data.encode).first
750
+ meta.data_page_offset += shift
751
+ dict = meta.dictionary_page_offset
752
+ # Some writers store 0 when there is no dictionary page
753
+ meta.dictionary_page_offset = dict&.positive? ? dict + shift : nil
754
+ meta.index_page_offset += shift if meta.index_page_offset
755
+ meta.total_compressed_size = copy.bytes.bytesize
756
+ meta.bloom_filter_offset = meta.bloom_filter_length = nil
757
+ write_raw(copy.bytes)
758
+ chunk = Format::ColumnChunk.new(file_offset: start, meta_data: meta)
759
+ @pending_bloom_filters << [meta, copy.bloom_filter] if copy.bloom_filter
760
+ offset_index = copy.offset_index && Format::OffsetIndex.new(
761
+ page_locations: copy.offset_index.page_locations.map do |loc|
762
+ Format::PageLocation.new(offset: loc.offset + shift, compressed_page_size: loc.compressed_page_size,
763
+ first_row_index: loc.first_row_index)
764
+ end,
765
+ unencoded_byte_array_data_bytes: copy.offset_index.unencoded_byte_array_data_bytes
766
+ )
767
+ @page_indexes << [chunk, copy.column_index, offset_index]
768
+ chunk
769
+ end
770
+
661
771
  # nil when the column has no defined sort order, or a page's values have no min/max (all NaN)
662
772
  #
663
773
  # @param col [Schema::Column] column the pages belong to
@@ -693,18 +803,20 @@ module Herringbone
693
803
  end
694
804
  end
695
805
 
696
- # Page indexes go after the last row group: all column indexes, then all offset indexes
806
+ # Page indexes go after the last row group: all column indexes, then all offset indexes.
807
+ # A copied chunk brings its ColumnIndex already encoded, and may come without either index.
697
808
  #
698
809
  # @return [void]
699
810
  def write_page_indexes
700
811
  @page_indexes.each do |chunk, column_index, _|
701
812
  next unless column_index
702
- bytes = column_index.encode
813
+ bytes = column_index.is_a?(String) ? column_index : column_index.encode
703
814
  chunk.column_index_offset = @pos
704
815
  chunk.column_index_length = bytes.bytesize
705
816
  write_raw(bytes)
706
817
  end
707
818
  @page_indexes.each do |chunk, _, offset_index|
819
+ next unless offset_index
708
820
  bytes = offset_index.encode
709
821
  chunk.offset_index_offset = @pos
710
822
  chunk.offset_index_length = bytes.bytesize
@@ -780,12 +892,13 @@ module Herringbone
780
892
  filter
781
893
  end
782
894
 
783
- # Bloom filters go right after the row group's column chunks, in column order
895
+ # Bloom filters go right after the row group's column chunks, in column order. A copied
896
+ # chunk brings its filter already encoded.
784
897
  #
785
898
  # @return [void]
786
899
  def write_bloom_filters
787
900
  @pending_bloom_filters.each do |meta, filter|
788
- bytes = filter.encode
901
+ bytes = filter.is_a?(String) ? filter : filter.encode
789
902
  meta.bloom_filter_offset = @pos
790
903
  meta.bloom_filter_length = bytes.bytesize
791
904
  write_raw(bytes)
@@ -906,9 +1019,10 @@ module Herringbone
906
1019
  # @param body [String] uncompressed page body
907
1020
  # @param compressed [String, nil] bytes to write as the page body, when already prepared (v2 data
908
1021
  # pages, whose levels stay uncompressed); +body+ is compressed otherwise
1022
+ # @param codec [Integer] codec id to compress +body+ with
909
1023
  # @return [Integer]
910
- def write_page(header, body, compressed = nil)
911
- compressed ||= Compression.compress(@codec, body)
1024
+ def write_page(header, body, compressed = nil, codec: @codec)
1025
+ compressed ||= Compression.compress(codec, body)
912
1026
  header.uncompressed_page_size ||= body.bytesize
913
1027
  header.compressed_page_size = compressed.bytesize
914
1028
  header.crc = Zlib.crc32(compressed).then { |c| (c >= 0x8000_0000) ? c - 0x1_0000_0000 : c }
@@ -925,8 +1039,9 @@ module Herringbone
925
1039
  # @param def_bytes [String] RLE-encoded definition levels, empty when the column has none
926
1040
  # @param encoded [String] encoded values
927
1041
  # @param encoding [Integer] encoding id of the values
1042
+ # @param codec [Integer] codec id to compress the page with
928
1043
  # @return [Integer] uncompressed size including the header
929
- def write_data_page_v1(n, rep_bytes, def_bytes, encoded, encoding)
1044
+ def write_data_page_v1(n, rep_bytes, def_bytes, encoded, encoding, codec)
930
1045
  body = String.new(encoding: Encoding::BINARY)
931
1046
  body << [rep_bytes.bytesize].pack("V") << rep_bytes unless rep_bytes.empty?
932
1047
  body << [def_bytes.bytesize].pack("V") << def_bytes unless def_bytes.empty?
@@ -938,7 +1053,7 @@ module Herringbone
938
1053
  definition_level_encoding: E::RLE, repetition_level_encoding: E::RLE
939
1054
  )
940
1055
  )
941
- write_page(header, body)
1056
+ write_page(header, body, codec: codec)
942
1057
  end
943
1058
 
944
1059
  # A DATA_PAGE_V2: levels without length prefixes and uncompressed, then the compressed values
@@ -950,9 +1065,10 @@ module Herringbone
950
1065
  # @param def_bytes [String] RLE-encoded definition levels, empty when the column has none
951
1066
  # @param encoded [String] encoded values
952
1067
  # @param encoding [Integer] encoding id of the values
1068
+ # @param codec [Integer] codec id to compress the values with
953
1069
  # @return [Integer] uncompressed size including the header
954
- def write_data_page_v2(n, nulls, rows, rep_bytes, def_bytes, encoded, encoding)
955
- compressed = Compression.compress(@codec, encoded)
1070
+ def write_data_page_v2(n, nulls, rows, rep_bytes, def_bytes, encoded, encoding, codec)
1071
+ compressed = Compression.compress(codec, encoded)
956
1072
  header = Format::PageHeader.new(
957
1073
  type: Format::PageType::DATA_PAGE_V2,
958
1074
  uncompressed_page_size: rep_bytes.bytesize + def_bytes.bytesize + encoded.bytesize,
@@ -960,10 +1076,10 @@ module Herringbone
960
1076
  num_values: n, num_nulls: nulls, num_rows: rows, encoding: encoding,
961
1077
  definition_levels_byte_length: def_bytes.bytesize,
962
1078
  repetition_levels_byte_length: rep_bytes.bytesize,
963
- is_compressed: @codec != Format::Codec::UNCOMPRESSED
1079
+ is_compressed: codec != Format::Codec::UNCOMPRESSED
964
1080
  )
965
1081
  )
966
- write_page(header, "".b, rep_bytes + def_bytes + compressed)
1082
+ write_page(header, "".b, rep_bytes + def_bytes + compressed, codec: codec)
967
1083
  end
968
1084
 
969
1085
  # Statistics and column index bounds longer than this are truncated, see #truncate_min
data/lib/herringbone.rb CHANGED
@@ -64,6 +64,7 @@ require_relative "herringbone/xxhash"
64
64
  require_relative "herringbone/bloom_filter"
65
65
  require_relative "herringbone/inspector"
66
66
  require_relative "herringbone/visualizer"
67
+ require_relative "herringbone/redaction"
67
68
 
68
69
  module Herringbone
69
70
  module_function
@@ -77,7 +78,7 @@ module Herringbone
77
78
  # to Writer.
78
79
  #
79
80
  # File.open("orders.parquet", "wb") { |f| Herringbone.write(f, Order.where(created_at: 1.year.ago..)) }
80
- # Herringbone.write(io, events.lazy.map(&:to_h)) { json :payload }
81
+ # Herringbone.write(io, events.lazy.map(&:to_h)) { |s| s.json :payload }
81
82
  #
82
83
  # @param io [IO, #write] destination; written sequentially, never closed
83
84
  # @param records [Enumerable<Hash, Array, Object>, Class, #find_each] rows (Hashes, Arrays in schema
@@ -97,14 +98,17 @@ module Herringbone
97
98
  # @option options [Hash{String => String}] :metadata ({}) key/value metadata for the footer
98
99
  # @option options [Boolean, Array<String>, Hash{String => Boolean, Hash}] :bloom_filters (nil)
99
100
  # columns to write split block bloom filters for, see Writer
100
- # @yield optional block for Schema.infer (Builder DSL), declaring fields that replace inferred ones;
101
- # ignored when the schema is not inferred
101
+ # @yield [s] optional, declares fields that replace inferred ones (see Schema.infer); ignored when
102
+ # the schema is not inferred
103
+ # @yieldparam s [Schema::Builder] the builder to declare fields on
104
+ # @yieldreturn [void]
102
105
  # @return [Integer] number of rows written
103
106
  # @raise [ArgumentError] when the schema has to be inferred and +records+ is empty or holds Array
104
- # rows, or an option is invalid
107
+ # rows, an option is invalid, or the block takes no parameter
105
108
  # @raise [EncodeError] when a row does not fit the schema
106
109
  # @raise [SchemaMismatch] when a row does not fit the inferred schema; the file is left unfinished
107
110
  def write(io, records, schema: nil, **options, &overrides)
111
+ Schema::Builder.check_block!(overrides, "Herringbone.write(io, rows) { |s| s.json :payload }")
108
112
  model = if records.respond_to?(:klass) then records.klass
109
113
  elsif records.respond_to?(:columns) && records.respond_to?(:find_each) then records
110
114
  end
@@ -112,7 +116,7 @@ module Herringbone
112
116
  writer = if schema
113
117
  Writer.new(io, schema, **options)
114
118
  else
115
- fix = "Herringbone.write(io, rows) { %s }"
119
+ fix = "Herringbone.write(io, rows) { |s| s.%s }"
116
120
  InferringWriter.new(io, fix: fix, **options) { |sample| Schema.infer(sample, &overrides) }
117
121
  end
118
122
  begin
@@ -129,6 +133,44 @@ module Herringbone
129
133
  writer.rows_written
130
134
  end
131
135
 
136
+ # Rewrites +input+ into +output+ with rows removed or values replaced, building the Redaction from
137
+ # the block (or taking +redaction+). A shortcut for Redaction#apply.
138
+ #
139
+ # Herringbone.redact(input, output) do |r|
140
+ # r.where(user_id: 42).delete
141
+ # r.replace(:email) { |email| email && OpenSSL::HMAC.hexdigest("SHA256", KEY, email) }
142
+ # end
143
+ #
144
+ # @param input [IO, StringIO] the Parquet file, read with #seek and #read; not closed
145
+ # @param output [IO, #write] destination, written sequentially; not closed
146
+ # @param redaction [Redaction, nil] the redaction to apply, instead of a block
147
+ # @param writer_options [Hash{Symbol => Object}] Writer options for re-encoded column chunks, and
148
+ # +metadata:+ to replace the footer key/value metadata, see Redaction#apply
149
+ # @option writer_options [Symbol] :compression (codec of each source chunk) codec for re-encoded chunks
150
+ # @option writer_options [Boolean, Array<String>, Hash{String => Boolean, Hash}] :bloom_filters (nil)
151
+ # columns whose re-encoded chunks get a bloom filter, besides those whose source chunk had one
152
+ # @option writer_options [Hash{String => String}] :metadata (the input's) footer key/value metadata
153
+ # @option writer_options [Integer] :page_bytes (1MB) approximate uncompressed data page size
154
+ # @option writer_options [Integer] :page_rows (20_000) maximum rows per data page
155
+ # @option writer_options [Integer] :data_page_version (1) 1 or 2
156
+ # @option writer_options [Boolean, Array<String>] :dictionary (true) see Writer
157
+ # @option writer_options [Hash{String => Symbol}] :encodings ({}) see Writer
158
+ # @yield [r] declares the statements on a new Redaction (see Redaction.new)
159
+ # @yieldparam r [Redaction] the redaction being built
160
+ # @yieldreturn [void]
161
+ # @return [Redaction::Report] what was done
162
+ # @raise [ArgumentError] when given both or neither of +redaction+ and a block, a block that takes
163
+ # no parameter, or when the redaction does not fit the file's schema
164
+ # @raise [EncodeError] when a replacement value cannot be written to its column
165
+ def redact(input, output, redaction = nil, **writer_options, &block)
166
+ raise ArgumentError, "Herringbone.redact takes a Redaction or a block, not both" if redaction && block
167
+ raise ArgumentError, "Herringbone.redact needs a Redaction or a block" unless redaction || block
168
+ if block&.arity&.zero?
169
+ raise ArgumentError, "The block receives the redaction: Herringbone.redact(input, output) { |r| r.where(user_id: 42).delete }"
170
+ end
171
+ (redaction || Redaction.new(&block)).apply(input, output, **writer_options)
172
+ end
173
+
132
174
  # Compression codecs this process can read and write, e.g. [:none, :snappy, :gzip, :lz4, :lz4_hadoop, :zstd].
133
175
  # :zstd and :brotli are listed when the zstd-ruby / brotli gems are loaded.
134
176
  #
metadata CHANGED
@@ -1,7 +1,7 @@
1
1
  --- !ruby/object:Gem::Specification
2
2
  name: herringbone
3
3
  version: !ruby/object:Gem::Version
4
- version: 0.4.0
4
+ version: 0.5.0
5
5
  platform: ruby
6
6
  authors:
7
7
  - Julik Tarkhanov
@@ -58,6 +58,8 @@ files:
58
58
  - lib/herringbone/reader/numo.rb
59
59
  - lib/herringbone/reader/page_stream.rb
60
60
  - lib/herringbone/reader/scan.rb
61
+ - lib/herringbone/redaction.rb
62
+ - lib/herringbone/redaction/rewriter.rb
61
63
  - lib/herringbone/schema.rb
62
64
  - lib/herringbone/simple_writer.rb
63
65
  - lib/herringbone/thrift.rb