herringbone 0.4.0 → 0.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,533 @@
1
+ # frozen_string_literal: true
2
+
3
+ module Herringbone
4
+ class Redaction
5
+ # Applies a Redaction to one file. Building it checks the redaction against the file's schema,
6
+ # so mistakes surface before anything is written.
7
+ #
8
+ # Each row group lands in one of three tiers. When no statement can match (by statistics,
9
+ # bloom filters and the page index, or by reading the condition columns), its column chunks
10
+ # are copied byte for byte. When rows match but none is deleted and only leaf columns change,
11
+ # the changed chunks are re-encoded and the others copied. Otherwise the whole row group is
12
+ # rewritten, keeping its boundary: one row group in, one (or none) out.
13
+ class Rewriter
14
+ # A replaced column, resolved against the file's schema
15
+ #
16
+ # @!attribute name
17
+ # @return [String] the column as named in the statement
18
+ # @!attribute path
19
+ # @return [Array<String>] top-level field name, then struct member names
20
+ # @!attribute field
21
+ # @return [Schema::Field] the replaced field
22
+ # @!attribute column
23
+ # @return [Schema::Column, nil] the leaf column, nil when the field is nested
24
+ Target = Struct.new(:name, :path, :field, :column)
25
+
26
+ # Writer options that make no sense here, since row groups keep their boundaries
27
+ ROW_GROUP_OPTIONS = %i[row_group_bytes row_group_rows].freeze
28
+ # Footer metadata that describes the columns, and goes stale when some are dropped
29
+ COLUMN_METADATA_KEYS = %w[ARROW:schema pandas].freeze
30
+ # Bytes read at a time when a page header runs past what was read of a chunk
31
+ READ_MORE = 64 * 1024
32
+ # What #current returns for a struct member of a null struct
33
+ ABSENT = Object.new.freeze
34
+
35
+ # @param redaction [Redaction] the statements and drops to apply
36
+ # @param io [IO, StringIO] the input file, read with #seek and #read
37
+ # @raise [ArgumentError] when a statement or drop does not fit the file's schema
38
+ # @raise [FormatError] when the footer cannot be read
39
+ def initialize(redaction, io)
40
+ @io = io
41
+ @reader = Reader.new(io)
42
+ @schema = @reader.schema
43
+ @statements = redaction.statements
44
+ @filters = @statements.map { |s| s.where && Reader::Filter.new(@schema, s.where) }
45
+ @targets = @statements.map { |s| s.targets.map { |name| target(name, s) } }
46
+ @row_blocks = @statements.map { |s| s.block && wants_row?(s.block) }
47
+ @drops = redaction.drops.uniq.map { |name| resolve(name, "drop")[1] }
48
+ check_drops!
49
+ @output_schema = output_schema
50
+ @output_columns = @output_schema.columns.map { |c| @schema.column(c.path) }
51
+ @first_rows = @reader.row_groups.each_with_object([0]) { |rg, firsts| firsts << firsts.last + rg.num_rows }
52
+ @rows_read = 0
53
+ @data = {}
54
+ end
55
+
56
+ # @return [Boolean] see Redaction#affects?
57
+ def affects?
58
+ return true unless @drops.empty?
59
+ @reader.row_groups.each_index.any? do |i|
60
+ @data = {}
61
+ candidates = candidates(i)
62
+ next false if candidates.empty?
63
+ candidates.any? { |k| @filters[k].nil? } || any_match?(i, candidates)
64
+ end
65
+ end
66
+
67
+ # @param output [IO, #write] destination
68
+ # @param options [Hash{Symbol => Object}] Writer options for re-encoded chunks, and +metadata:+
69
+ # @option options [Symbol] :compression (codec of each source chunk) codec for re-encoded chunks
70
+ # @option options [Boolean, Array<String>, Hash{String => Boolean, Hash}] :bloom_filters (nil)
71
+ # columns whose re-encoded chunks get a bloom filter, besides those whose source chunk had one
72
+ # @option options [Hash{String => String}] :metadata (the input's) footer key/value metadata
73
+ # @option options [Integer] :page_bytes (1MB) approximate uncompressed data page size
74
+ # @option options [Integer] :page_rows (20_000) maximum rows per data page
75
+ # @option options [Integer] :data_page_version (1) 1 or 2
76
+ # @option options [Boolean, Array<String>] :dictionary (true) see Writer
77
+ # @option options [Hash{String => Symbol}] :encodings ({}) see Writer
78
+ # @return [Report]
79
+ # @raise [ArgumentError] for +row_group_bytes:+ / +row_group_rows:+ or an invalid writer option
80
+ # @raise [EncodeError] when a replacement value cannot be written
81
+ def apply(output, **options)
82
+ bad = options.keys & ROW_GROUP_OPTIONS
83
+ raise ArgumentError, "#{bad.join(", ")}: a redaction keeps the row groups of the input" unless bad.empty?
84
+ @keep_codecs = !options.key?(:compression)
85
+ options = {metadata: copied_metadata}.merge(options)
86
+ writer = Writer.new(output, @output_schema, row_group_bytes: 1 << 62, **options)
87
+ @report = Report.new(rows_read: 0, rows_deleted: 0, rows_changed: 0, row_groups: {copied: 0, rewritten: 0})
88
+ begin
89
+ @reader.row_groups.each_index { |i| redact_row_group(i, writer) }
90
+ rescue Exception # rubocop:disable Lint/RescueException -- also abort on Interrupt
91
+ writer.abort
92
+ raise
93
+ end
94
+ writer.close
95
+ @report.rows_read = @rows_read
96
+ @report
97
+ end
98
+
99
+ private
100
+
101
+ # @param i [Integer] row group index
102
+ # @param writer [Writer] the output
103
+ # @return [void]
104
+ def redact_row_group(i, writer)
105
+ @data = {}
106
+ candidates = candidates(i)
107
+ if candidates.empty? || (candidates.all? { |k| @filters[k] } && !any_match?(i, candidates))
108
+ return copy_row_group(i, writer)
109
+ end
110
+ deleted, changed = run_statements(i)
111
+ if deleted.none? && changed.empty?
112
+ copy_row_group(i, writer)
113
+ elsif deleted.none? && changed.all?(&:column)
114
+ rewrite_columns(i, writer, changed)
115
+ else
116
+ rewrite_row_group(i, writer, deleted, changed)
117
+ end
118
+ end
119
+
120
+ # Statements that may touch row group +i+: those without conditions, and those whose
121
+ # conditions the statistics, bloom filters and page index do not rule out
122
+ #
123
+ # @param i [Integer] row group index
124
+ # @return [Array<Integer>] statement indexes
125
+ def candidates(i)
126
+ return [] if rows(i).zero?
127
+ @statements.each_index.select do |k|
128
+ filter = @filters[k]
129
+ filter.nil? || (filter.row_group_may_match?(@reader, i) && !filter.page_ranges(@reader, i).empty?)
130
+ end
131
+ end
132
+
133
+ # Reads the condition columns of row group +i+ and checks them against the original values.
134
+ # When no row matches any statement, no statement changes anything (later statements see
135
+ # what earlier ones changed, and none did), so the row group can be copied.
136
+ #
137
+ # @param i [Integer] row group index
138
+ # @param candidates [Array<Integer>] indexes of statements, all with conditions
139
+ # @return [Boolean] whether some row matches some statement
140
+ def any_match?(i, candidates)
141
+ fields = candidates.flat_map { |k| @filters[k].fields }.uniq
142
+ load(i, fields.map(&:name))
143
+ data = fields.map { |f| @data[f.name] }
144
+ candidates.any? { |k| !@filters[k].matching_rows(data, fields, rows(i), false).empty? }
145
+ end
146
+
147
+ # Applies the statements to every row of row group +i+, in place in @data
148
+ #
149
+ # @param i [Integer] row group index
150
+ # @return [Array(Array<Boolean>, Array<Target>)] deleted flag per row, and the targets whose
151
+ # values changed
152
+ def run_statements(i)
153
+ names = if @row_blocks.any?
154
+ @schema.fields.map(&:name)
155
+ else
156
+ @filters.compact.flat_map { |f| f.fields.map(&:name) } + @targets.flatten.map { |t| t.path.first }
157
+ end
158
+ load(i, names.uniq)
159
+ n = rows(i)
160
+ deleted = Array.new(n, false)
161
+ changed = {}
162
+ n.times do |r|
163
+ row_changed = false
164
+ @statements.each_with_index do |statement, k|
165
+ filter = @filters[k]
166
+ next if filter && !row_matches?(filter, r)
167
+ if statement.kind == :delete
168
+ deleted[r] = true
169
+ break
170
+ end
171
+ @targets[k].each do |t|
172
+ old = current(t, r)
173
+ next if old.equal?(ABSENT)
174
+ value = if @row_blocks[k] then statement.block.call(old, row_hash(r))
175
+ elsif statement.block then statement.block.call(old)
176
+ else statement.constants[t.name]
177
+ end
178
+ next unless value != old
179
+ assign(t, r, value)
180
+ changed[t.path] ||= t
181
+ row_changed = true
182
+ end
183
+ end
184
+ if deleted[r]
185
+ @report.rows_deleted += 1
186
+ elsif row_changed
187
+ @report.rows_changed += 1
188
+ end
189
+ end
190
+ [deleted, changed.values]
191
+ end
192
+
193
+ # @param filter [Reader::Filter] conditions of a statement
194
+ # @param r [Integer] row index within the row group
195
+ # @return [Boolean] whether the row, as earlier statements left it, matches all conditions
196
+ def row_matches?(filter, r)
197
+ filter.conditions.all? do |c|
198
+ value = c.path.reduce(@data[c.field.name][r]) { |h, key| h.is_a?(Hash) ? h[key] : nil }
199
+ Reader::Filter.matches?(c.test, value)
200
+ end
201
+ end
202
+
203
+ # @param t [Target] replaced column
204
+ # @param r [Integer] row index within the row group
205
+ # @return [Object] the column's current value, or ABSENT when a struct holding it is null
206
+ def current(t, r)
207
+ value = @data[t.path.first][r]
208
+ t.path.drop(1).each do |key|
209
+ return ABSENT unless value.is_a?(Hash)
210
+ value = value[key]
211
+ end
212
+ value
213
+ end
214
+
215
+ # Stores a value, copying the struct Hashes on the way so nothing else shares them
216
+ #
217
+ # @param t [Target] replaced column
218
+ # @param r [Integer] row index within the row group
219
+ # @param value [Object] the new value
220
+ # @return [void]
221
+ def assign(t, r, value)
222
+ column = @data[t.path.first]
223
+ return column[r] = value if t.path.size == 1
224
+ hash = column[r] = column[r].dup
225
+ t.path[1...-1].each { |key| hash = hash[key] = hash[key].dup }
226
+ hash[t.path.last] = value
227
+ end
228
+
229
+ # @param r [Integer] row index within the row group
230
+ # @return [Hash{String => Object}] the whole row as it stands, keyed by top-level field name
231
+ def row_hash(r)
232
+ @schema.fields.to_h { |f| [f.name, @data[f.name][r]] }
233
+ end
234
+
235
+ # Tier 1: every kept column chunk is copied as it is
236
+ #
237
+ # @param i [Integer] row group index
238
+ # @param writer [Writer] the output
239
+ # @return [void]
240
+ def copy_row_group(i, writer)
241
+ copies = @output_columns.each_with_index.to_h { |col, j| [j, copied_chunk(i, col)] }
242
+ writer.write_row_group(rows(i), copies: copies, sorting_columns: sorting_columns(i, []))
243
+ @report.row_groups[:copied] += 1
244
+ end
245
+
246
+ # Tier 2: the chunks of the changed leaf columns are encoded again, the rest are copied
247
+ #
248
+ # @param i [Integer] row group index
249
+ # @param writer [Writer] the output
250
+ # @param changed [Array<Target>] leaf targets whose values changed
251
+ # @return [void]
252
+ def rewrite_columns(i, writer, changed)
253
+ rewritten = changed.map { |t| t.column.index }
254
+ changed.map { |t| t.path.first }.uniq.each { |name| writer.buffer_field(name, @data[name]) }
255
+ copies = {}
256
+ codecs = {}
257
+ blooms = {}
258
+ @output_columns.each_with_index do |col, j|
259
+ if rewritten.include?(col.index)
260
+ chunk_settings(i, col, j, codecs, blooms)
261
+ else
262
+ copies[j] = copied_chunk(i, col)
263
+ end
264
+ end
265
+ writer.write_row_group(rows(i), copies: copies, codecs: codecs, bloom_filters: blooms,
266
+ sorting_columns: sorting_columns(i, rewritten))
267
+ @report.row_groups[:rewritten] += 1
268
+ end
269
+
270
+ # Tier 3: the remaining rows are encoded again, every column of them
271
+ #
272
+ # @param i [Integer] row group index
273
+ # @param writer [Writer] the output
274
+ # @param deleted [Array<Boolean>] deleted flag per row
275
+ # @param changed [Array<Target>] targets whose values changed
276
+ # @return [void]
277
+ def rewrite_row_group(i, writer, deleted, changed)
278
+ @report.row_groups[:rewritten] += 1
279
+ kept = deleted.count(false)
280
+ return if kept.zero?
281
+ load(i, @output_schema.fields.map(&:name))
282
+ @output_schema.fields.each do |f|
283
+ values = @data[f.name]
284
+ values = values.reject.with_index { |_, r| deleted[r] } if kept < values.size
285
+ writer.buffer_field(f.name, values)
286
+ end
287
+ codecs = {}
288
+ blooms = {}
289
+ @output_columns.each_with_index { |col, j| chunk_settings(i, col, j, codecs, blooms) }
290
+ rewritten = changed.flat_map { |t| t.field.leaves.map(&:index) }
291
+ writer.write_row_group(kept, codecs: codecs, bloom_filters: blooms, sorting_columns: sorting_columns(i, rewritten))
292
+ end
293
+
294
+ # Codec and bloom filter of a re-encoded chunk follow the source chunk: the same codec
295
+ # (unless +compression:+ was given; LZO cannot be written and becomes Snappy), and a bloom
296
+ # filter when the source chunk had one
297
+ #
298
+ # @param i [Integer] row group index
299
+ # @param col [Schema::Column] the input column
300
+ # @param j [Integer] the output column index
301
+ # @param codecs [Hash{Integer => Integer}] collects output column index => codec id
302
+ # @param blooms [Hash{Integer => Boolean}] collects output column index => true
303
+ # @return [void]
304
+ def chunk_settings(i, col, j, codecs, blooms)
305
+ meta = @reader.row_groups[i].columns.fetch(col.index).meta_data
306
+ return unless meta
307
+ codecs[j] = (meta.codec == Format::Codec::LZO) ? Format::Codec::SNAPPY : meta.codec if @keep_codecs
308
+ blooms[j] = true if meta.bloom_filter_offset
309
+ end
310
+
311
+ # The row group's sorting columns that still hold: dropped or changed columns end the list,
312
+ # since the columns after them were only sorted within their runs
313
+ #
314
+ # @param i [Integer] row group index
315
+ # @param rewritten [Array<Integer>] indexes of input columns whose values changed
316
+ # @return [Array<Format::SortingColumn>, nil]
317
+ def sorting_columns(i, rewritten)
318
+ out = []
319
+ (@reader.row_groups[i].sorting_columns || []).each do |sc|
320
+ col = @schema.columns[sc.column_idx]
321
+ j = col && !rewritten.include?(col.index) && @output_columns.index(col)
322
+ break unless j
323
+ out << Format::SortingColumn.new(column_idx: j, descending: sc.descending, nulls_first: sc.nulls_first)
324
+ end
325
+ out.empty? ? nil : out
326
+ end
327
+
328
+ # @param i [Integer] row group index
329
+ # @param col [Schema::Column] the input column
330
+ # @return [Writer::CopiedChunk] the chunk's pages, page index and bloom filter
331
+ # @raise [UnsupportedError] for an encrypted chunk or one stored in another file
332
+ # @raise [FormatError] when a page header is corrupt or the chunk overruns the file
333
+ def copied_chunk(i, col)
334
+ chunk = @reader.row_groups[i].columns.fetch(col.index)
335
+ meta = chunk.meta_data or raise UnsupportedError, "Column chunk without metadata (encrypted?)"
336
+ raise UnsupportedError, "Column chunks in external files are not supported" if chunk.file_path
337
+ column_index, offset_index = @reader.page_index(i, col)
338
+ start, bytes = chunk_bytes(col, meta, offset_index)
339
+ column_index &&= read_at(chunk.column_index_offset, chunk.column_index_length)
340
+ Writer::CopiedChunk.new(chunk, start, bytes, column_index, offset_index, bloom_bytes(i, col, meta))
341
+ end
342
+
343
+ # The chunk's pages. They are walked header by header rather than trusting
344
+ # total_compressed_size, which some writers get wrong: copying too little breaks the
345
+ # chunk, and copying too much could carry over bytes of a neighbouring chunk.
346
+ #
347
+ # @param col [Schema::Column] the input column
348
+ # @param meta [Format::ColumnMetaData] the chunk's metadata
349
+ # @param offset_index [Format::OffsetIndex, nil] the chunk's OffsetIndex, whose pages are
350
+ # included even when num_values is reached before them
351
+ # @return [Array(Integer, String)] file offset of the first page, and the pages' bytes
352
+ # @raise [FormatError] when a page header is corrupt or the chunk overruns the file
353
+ def chunk_bytes(col, meta, offset_index)
354
+ start = meta.data_page_offset
355
+ dict = meta.dictionary_page_offset
356
+ start = dict if dict&.positive? && dict < start
357
+ last = offset_index&.page_locations&.last
358
+ min_end = last ? last.offset + last.compressed_page_size - start : 0
359
+ @io.seek(start)
360
+ buf = (@io.read(meta.total_compressed_size) || "").b
361
+ pos = 0
362
+ seen = 0
363
+ while seen < meta.num_values || pos < min_end
364
+ begin
365
+ header, body = Format::PageHeader.decode(buf, pos)
366
+ rescue Thrift::Error
367
+ raise FormatError, "Corrupt page header in #{col.dotted_path}" unless read_more(buf, start, READ_MORE)
368
+ retry
369
+ end
370
+ size = header.compressed_page_size
371
+ raise FormatError, "Negative page size in #{col.dotted_path}" if size.nil? || size.negative?
372
+ short = body + size - buf.bytesize
373
+ if short.positive? && read_more(buf, start, short) < short
374
+ raise FormatError, "Column #{col.dotted_path}: page overruns the file"
375
+ end
376
+ pos = body + size
377
+ seen += header.data_page_header&.num_values || header.data_page_header_v2&.num_values || 0
378
+ end
379
+ [start, (pos == buf.bytesize) ? buf : buf.byteslice(0, pos)]
380
+ end
381
+
382
+ # Appends up to +count+ bytes that follow +buf+ in the file
383
+ #
384
+ # @param buf [String] bytes read from +start+ on; appended to
385
+ # @param start [Integer] file offset of +buf+
386
+ # @param count [Integer] bytes wanted
387
+ # @return [Integer] bytes appended, 0 at the end of the file
388
+ def read_more(buf, start, count)
389
+ @io.seek(start + buf.bytesize)
390
+ more = @io.read(count)
391
+ return 0 if more.nil?
392
+ buf << more.b
393
+ more.bytesize
394
+ end
395
+
396
+ # @param offset [Integer, nil] file offset
397
+ # @param length [Integer, nil] byte count
398
+ # @return [String, nil] the bytes, nil when absent or cut off
399
+ def read_at(offset, length)
400
+ return nil unless offset && length&.positive?
401
+ @io.seek(offset)
402
+ bytes = @io.read(length)
403
+ (bytes&.bytesize == length) ? bytes.b : nil
404
+ end
405
+
406
+ # The chunk's bloom filter as stored. Filters written without a length (older writers) are
407
+ # decoded and encoded again. A filter that cannot be read is left out, which only costs
408
+ # pruning.
409
+ #
410
+ # @param i [Integer] row group index
411
+ # @param col [Schema::Column] the input column
412
+ # @param meta [Format::ColumnMetaData] the chunk's metadata
413
+ # @return [String, nil] header and bitset
414
+ def bloom_bytes(i, col, meta)
415
+ return nil unless meta.bloom_filter_offset
416
+ return read_at(meta.bloom_filter_offset, meta.bloom_filter_length) if meta.bloom_filter_length
417
+ @reader.bloom_filter(i, col)&.encode
418
+ rescue FormatError
419
+ nil
420
+ end
421
+
422
+ # Reads top-level fields of row group +i+ into @data, unless they are there already
423
+ #
424
+ # @param i [Integer] row group index
425
+ # @param names [Array<String>] top-level field names
426
+ # @return [void]
427
+ def load(i, names)
428
+ missing = names - @data.keys
429
+ return if missing.empty?
430
+ n = rows(i)
431
+ @rows_read += n if @data.empty?
432
+ @data.merge!(@reader.read(as: :columns, columns: missing, from: @first_rows[i], limit: n))
433
+ end
434
+
435
+ # @param i [Integer] row group index
436
+ # @return [Integer] rows in the row group
437
+ def rows(i) = @reader.row_groups[i].num_rows
438
+
439
+ # @return [Hash{String => String, nil}] the input's key/value metadata, without the keys that
440
+ # describe the columns when some are dropped
441
+ def copied_metadata
442
+ meta = @reader.metadata
443
+ @drops.empty? ? meta : meta.except(*COLUMN_METADATA_KEYS)
444
+ end
445
+
446
+ # @param block [Proc] a replace block
447
+ # @return [Boolean] whether the block takes the row as a second parameter
448
+ def wants_row?(block)
449
+ params = block.parameters
450
+ params.any? { |type, _| type == :rest } || params.count { |type, _| type == :req || type == :opt } >= 2
451
+ end
452
+
453
+ # Resolves a column named in a statement: a top-level field, or a struct member by dotted path
454
+ #
455
+ # @param name [String] the column as named
456
+ # @param verb [String] "replace" or "drop", for error messages
457
+ # @return [Array(Schema::Field, Array<String>)] the field and its path of names
458
+ # @raise [ArgumentError] when there is no such column, or it is inside a list or map
459
+ def resolve(name, verb)
460
+ field = @schema.field(name)
461
+ return [field, [name]] if field
462
+ parts = name.split(".")
463
+ field = @schema.field(parts.first) or raise ArgumentError, "#{verb}: no such column #{name.inspect}"
464
+ parts.drop(1).each_with_index do |part, depth|
465
+ case field.kind
466
+ when :struct
467
+ field = field.children_by_name[part] or raise ArgumentError, "#{verb}: no such column #{name.inspect}"
468
+ when :list, :map
469
+ whole = parts.first(depth + 1).join(".")
470
+ container = (field.kind == :map) ? "Hash" : "Array"
471
+ hint = (verb == "replace") ? " with a block that receives its #{container}" : ""
472
+ raise ArgumentError, "#{verb}: #{name} is inside the #{field.kind} #{whole}; #{verb} #{whole} instead#{hint}"
473
+ else
474
+ raise ArgumentError, "#{verb}: no such column #{name.inspect}"
475
+ end
476
+ end
477
+ [field, parts]
478
+ end
479
+
480
+ # @param name [String] the column as named in the statement
481
+ # @param statement [Statement] the replace statement
482
+ # @return [Target]
483
+ # @raise [ArgumentError] when the column does not exist, is inside a list or map, or a
484
+ # constant cannot be stored in it (nil in a required column, a value of the wrong type)
485
+ def target(name, statement)
486
+ field, path = resolve(name, "replace")
487
+ if statement.block.nil?
488
+ value = statement.constants[name]
489
+ if value.nil?
490
+ raise ArgumentError, "replace: #{name} is required (null: false) and cannot be set to nil" unless field.optional
491
+ elsif field.leaf?
492
+ begin
493
+ field.column.encoder.call(value)
494
+ rescue ArgumentError, TypeError, NoMethodError, RangeError => e
495
+ raise ArgumentError, "replace: cannot write #{value.inspect} to #{name}: #{e.message}"
496
+ end
497
+ end
498
+ end
499
+ Target.new(name, path, field, field.leaf? ? field.column : nil)
500
+ end
501
+
502
+ # @return [void]
503
+ # @raise [ArgumentError] when a replaced column is dropped
504
+ def check_drops!
505
+ @targets.flatten.each do |t|
506
+ dropped = @drops.find { |path| t.path.first(path.size) == path }
507
+ raise ArgumentError, "replace: #{t.name} is dropped by drop(#{dropped.join(".").inspect})" if dropped
508
+ end
509
+ end
510
+
511
+ # The input schema without the dropped fields
512
+ #
513
+ # @return [Schema]
514
+ # @raise [ArgumentError] when every column, or every member of a struct, would be dropped
515
+ def output_schema
516
+ return @schema if @drops.empty?
517
+ copy = Schema.from_elements(@schema.to_elements)
518
+ emptied = @drops.filter_map do |path|
519
+ parent = path[0...-1].reduce(copy.root) { |node, name| node&.children&.find { |c| c.name == name } }
520
+ node = parent&.children&.find { |c| c.name == path.last }
521
+ next unless node
522
+ parent.children.delete(node)
523
+ parent if parent.children.empty?
524
+ end
525
+ emptied.each do |node|
526
+ raise ArgumentError, "drop: cannot drop every column" if node.equal?(copy.root)
527
+ raise ArgumentError, "drop: cannot drop every member of #{node.path.join(".")}; drop #{node.path.join(".")} instead"
528
+ end
529
+ Schema.new(copy.root)
530
+ end
531
+ end
532
+ end
533
+ end