herringbone 0.4.0 → 0.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +14 -0
- data/README.md +150 -16
- data/lib/herringbone/active_record.rb +1 -1
- data/lib/herringbone/byte_values.rb +9 -0
- data/lib/herringbone/inferring_writer.rb +7 -0
- data/lib/herringbone/redaction/rewriter.rb +533 -0
- data/lib/herringbone/redaction.rb +287 -0
- data/lib/herringbone/schema.rb +121 -57
- data/lib/herringbone/simple_writer.rb +14 -4
- data/lib/herringbone/version.rb +1 -1
- data/lib/herringbone/writer.rb +141 -25
- data/lib/herringbone.rb +47 -5
- metadata +3 -1
|
@@ -0,0 +1,533 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Herringbone
|
|
4
|
+
class Redaction
|
|
5
|
+
# Applies a Redaction to one file. Building it checks the redaction against the file's schema,
|
|
6
|
+
# so mistakes surface before anything is written.
|
|
7
|
+
#
|
|
8
|
+
# Each row group lands in one of three tiers. When no statement can match (by statistics,
|
|
9
|
+
# bloom filters and the page index, or by reading the condition columns), its column chunks
|
|
10
|
+
# are copied byte for byte. When rows match but none is deleted and only leaf columns change,
|
|
11
|
+
# the changed chunks are re-encoded and the others copied. Otherwise the whole row group is
|
|
12
|
+
# rewritten, keeping its boundary: one row group in, one (or none) out.
|
|
13
|
+
class Rewriter
|
|
14
|
+
# A replaced column, resolved against the file's schema
|
|
15
|
+
#
|
|
16
|
+
# @!attribute name
|
|
17
|
+
# @return [String] the column as named in the statement
|
|
18
|
+
# @!attribute path
|
|
19
|
+
# @return [Array<String>] top-level field name, then struct member names
|
|
20
|
+
# @!attribute field
|
|
21
|
+
# @return [Schema::Field] the replaced field
|
|
22
|
+
# @!attribute column
|
|
23
|
+
# @return [Schema::Column, nil] the leaf column, nil when the field is nested
|
|
24
|
+
Target = Struct.new(:name, :path, :field, :column)
|
|
25
|
+
|
|
26
|
+
# Writer options that make no sense here, since row groups keep their boundaries
|
|
27
|
+
ROW_GROUP_OPTIONS = %i[row_group_bytes row_group_rows].freeze
|
|
28
|
+
# Footer metadata that describes the columns, and goes stale when some are dropped
|
|
29
|
+
COLUMN_METADATA_KEYS = %w[ARROW:schema pandas].freeze
|
|
30
|
+
# Bytes read at a time when a page header runs past what was read of a chunk
|
|
31
|
+
READ_MORE = 64 * 1024
|
|
32
|
+
# What #current returns for a struct member of a null struct
|
|
33
|
+
ABSENT = Object.new.freeze
|
|
34
|
+
|
|
35
|
+
# @param redaction [Redaction] the statements and drops to apply
|
|
36
|
+
# @param io [IO, StringIO] the input file, read with #seek and #read
|
|
37
|
+
# @raise [ArgumentError] when a statement or drop does not fit the file's schema
|
|
38
|
+
# @raise [FormatError] when the footer cannot be read
|
|
39
|
+
def initialize(redaction, io)
|
|
40
|
+
@io = io
|
|
41
|
+
@reader = Reader.new(io)
|
|
42
|
+
@schema = @reader.schema
|
|
43
|
+
@statements = redaction.statements
|
|
44
|
+
@filters = @statements.map { |s| s.where && Reader::Filter.new(@schema, s.where) }
|
|
45
|
+
@targets = @statements.map { |s| s.targets.map { |name| target(name, s) } }
|
|
46
|
+
@row_blocks = @statements.map { |s| s.block && wants_row?(s.block) }
|
|
47
|
+
@drops = redaction.drops.uniq.map { |name| resolve(name, "drop")[1] }
|
|
48
|
+
check_drops!
|
|
49
|
+
@output_schema = output_schema
|
|
50
|
+
@output_columns = @output_schema.columns.map { |c| @schema.column(c.path) }
|
|
51
|
+
@first_rows = @reader.row_groups.each_with_object([0]) { |rg, firsts| firsts << firsts.last + rg.num_rows }
|
|
52
|
+
@rows_read = 0
|
|
53
|
+
@data = {}
|
|
54
|
+
end
|
|
55
|
+
|
|
56
|
+
# @return [Boolean] see Redaction#affects?
|
|
57
|
+
def affects?
|
|
58
|
+
return true unless @drops.empty?
|
|
59
|
+
@reader.row_groups.each_index.any? do |i|
|
|
60
|
+
@data = {}
|
|
61
|
+
candidates = candidates(i)
|
|
62
|
+
next false if candidates.empty?
|
|
63
|
+
candidates.any? { |k| @filters[k].nil? } || any_match?(i, candidates)
|
|
64
|
+
end
|
|
65
|
+
end
|
|
66
|
+
|
|
67
|
+
# @param output [IO, #write] destination
|
|
68
|
+
# @param options [Hash{Symbol => Object}] Writer options for re-encoded chunks, and +metadata:+
|
|
69
|
+
# @option options [Symbol] :compression (codec of each source chunk) codec for re-encoded chunks
|
|
70
|
+
# @option options [Boolean, Array<String>, Hash{String => Boolean, Hash}] :bloom_filters (nil)
|
|
71
|
+
# columns whose re-encoded chunks get a bloom filter, besides those whose source chunk had one
|
|
72
|
+
# @option options [Hash{String => String}] :metadata (the input's) footer key/value metadata
|
|
73
|
+
# @option options [Integer] :page_bytes (1MB) approximate uncompressed data page size
|
|
74
|
+
# @option options [Integer] :page_rows (20_000) maximum rows per data page
|
|
75
|
+
# @option options [Integer] :data_page_version (1) 1 or 2
|
|
76
|
+
# @option options [Boolean, Array<String>] :dictionary (true) see Writer
|
|
77
|
+
# @option options [Hash{String => Symbol}] :encodings ({}) see Writer
|
|
78
|
+
# @return [Report]
|
|
79
|
+
# @raise [ArgumentError] for +row_group_bytes:+ / +row_group_rows:+ or an invalid writer option
|
|
80
|
+
# @raise [EncodeError] when a replacement value cannot be written
|
|
81
|
+
def apply(output, **options)
|
|
82
|
+
bad = options.keys & ROW_GROUP_OPTIONS
|
|
83
|
+
raise ArgumentError, "#{bad.join(", ")}: a redaction keeps the row groups of the input" unless bad.empty?
|
|
84
|
+
@keep_codecs = !options.key?(:compression)
|
|
85
|
+
options = {metadata: copied_metadata}.merge(options)
|
|
86
|
+
writer = Writer.new(output, @output_schema, row_group_bytes: 1 << 62, **options)
|
|
87
|
+
@report = Report.new(rows_read: 0, rows_deleted: 0, rows_changed: 0, row_groups: {copied: 0, rewritten: 0})
|
|
88
|
+
begin
|
|
89
|
+
@reader.row_groups.each_index { |i| redact_row_group(i, writer) }
|
|
90
|
+
rescue Exception # rubocop:disable Lint/RescueException -- also abort on Interrupt
|
|
91
|
+
writer.abort
|
|
92
|
+
raise
|
|
93
|
+
end
|
|
94
|
+
writer.close
|
|
95
|
+
@report.rows_read = @rows_read
|
|
96
|
+
@report
|
|
97
|
+
end
|
|
98
|
+
|
|
99
|
+
private
|
|
100
|
+
|
|
101
|
+
# @param i [Integer] row group index
|
|
102
|
+
# @param writer [Writer] the output
|
|
103
|
+
# @return [void]
|
|
104
|
+
def redact_row_group(i, writer)
|
|
105
|
+
@data = {}
|
|
106
|
+
candidates = candidates(i)
|
|
107
|
+
if candidates.empty? || (candidates.all? { |k| @filters[k] } && !any_match?(i, candidates))
|
|
108
|
+
return copy_row_group(i, writer)
|
|
109
|
+
end
|
|
110
|
+
deleted, changed = run_statements(i)
|
|
111
|
+
if deleted.none? && changed.empty?
|
|
112
|
+
copy_row_group(i, writer)
|
|
113
|
+
elsif deleted.none? && changed.all?(&:column)
|
|
114
|
+
rewrite_columns(i, writer, changed)
|
|
115
|
+
else
|
|
116
|
+
rewrite_row_group(i, writer, deleted, changed)
|
|
117
|
+
end
|
|
118
|
+
end
|
|
119
|
+
|
|
120
|
+
# Statements that may touch row group +i+: those without conditions, and those whose
|
|
121
|
+
# conditions the statistics, bloom filters and page index do not rule out
|
|
122
|
+
#
|
|
123
|
+
# @param i [Integer] row group index
|
|
124
|
+
# @return [Array<Integer>] statement indexes
|
|
125
|
+
def candidates(i)
|
|
126
|
+
return [] if rows(i).zero?
|
|
127
|
+
@statements.each_index.select do |k|
|
|
128
|
+
filter = @filters[k]
|
|
129
|
+
filter.nil? || (filter.row_group_may_match?(@reader, i) && !filter.page_ranges(@reader, i).empty?)
|
|
130
|
+
end
|
|
131
|
+
end
|
|
132
|
+
|
|
133
|
+
# Reads the condition columns of row group +i+ and checks them against the original values.
|
|
134
|
+
# When no row matches any statement, no statement changes anything (later statements see
|
|
135
|
+
# what earlier ones changed, and none did), so the row group can be copied.
|
|
136
|
+
#
|
|
137
|
+
# @param i [Integer] row group index
|
|
138
|
+
# @param candidates [Array<Integer>] indexes of statements, all with conditions
|
|
139
|
+
# @return [Boolean] whether some row matches some statement
|
|
140
|
+
def any_match?(i, candidates)
|
|
141
|
+
fields = candidates.flat_map { |k| @filters[k].fields }.uniq
|
|
142
|
+
load(i, fields.map(&:name))
|
|
143
|
+
data = fields.map { |f| @data[f.name] }
|
|
144
|
+
candidates.any? { |k| !@filters[k].matching_rows(data, fields, rows(i), false).empty? }
|
|
145
|
+
end
|
|
146
|
+
|
|
147
|
+
# Applies the statements to every row of row group +i+, in place in @data
|
|
148
|
+
#
|
|
149
|
+
# @param i [Integer] row group index
|
|
150
|
+
# @return [Array(Array<Boolean>, Array<Target>)] deleted flag per row, and the targets whose
|
|
151
|
+
# values changed
|
|
152
|
+
def run_statements(i)
|
|
153
|
+
names = if @row_blocks.any?
|
|
154
|
+
@schema.fields.map(&:name)
|
|
155
|
+
else
|
|
156
|
+
@filters.compact.flat_map { |f| f.fields.map(&:name) } + @targets.flatten.map { |t| t.path.first }
|
|
157
|
+
end
|
|
158
|
+
load(i, names.uniq)
|
|
159
|
+
n = rows(i)
|
|
160
|
+
deleted = Array.new(n, false)
|
|
161
|
+
changed = {}
|
|
162
|
+
n.times do |r|
|
|
163
|
+
row_changed = false
|
|
164
|
+
@statements.each_with_index do |statement, k|
|
|
165
|
+
filter = @filters[k]
|
|
166
|
+
next if filter && !row_matches?(filter, r)
|
|
167
|
+
if statement.kind == :delete
|
|
168
|
+
deleted[r] = true
|
|
169
|
+
break
|
|
170
|
+
end
|
|
171
|
+
@targets[k].each do |t|
|
|
172
|
+
old = current(t, r)
|
|
173
|
+
next if old.equal?(ABSENT)
|
|
174
|
+
value = if @row_blocks[k] then statement.block.call(old, row_hash(r))
|
|
175
|
+
elsif statement.block then statement.block.call(old)
|
|
176
|
+
else statement.constants[t.name]
|
|
177
|
+
end
|
|
178
|
+
next unless value != old
|
|
179
|
+
assign(t, r, value)
|
|
180
|
+
changed[t.path] ||= t
|
|
181
|
+
row_changed = true
|
|
182
|
+
end
|
|
183
|
+
end
|
|
184
|
+
if deleted[r]
|
|
185
|
+
@report.rows_deleted += 1
|
|
186
|
+
elsif row_changed
|
|
187
|
+
@report.rows_changed += 1
|
|
188
|
+
end
|
|
189
|
+
end
|
|
190
|
+
[deleted, changed.values]
|
|
191
|
+
end
|
|
192
|
+
|
|
193
|
+
# @param filter [Reader::Filter] conditions of a statement
|
|
194
|
+
# @param r [Integer] row index within the row group
|
|
195
|
+
# @return [Boolean] whether the row, as earlier statements left it, matches all conditions
|
|
196
|
+
def row_matches?(filter, r)
|
|
197
|
+
filter.conditions.all? do |c|
|
|
198
|
+
value = c.path.reduce(@data[c.field.name][r]) { |h, key| h.is_a?(Hash) ? h[key] : nil }
|
|
199
|
+
Reader::Filter.matches?(c.test, value)
|
|
200
|
+
end
|
|
201
|
+
end
|
|
202
|
+
|
|
203
|
+
# @param t [Target] replaced column
|
|
204
|
+
# @param r [Integer] row index within the row group
|
|
205
|
+
# @return [Object] the column's current value, or ABSENT when a struct holding it is null
|
|
206
|
+
def current(t, r)
|
|
207
|
+
value = @data[t.path.first][r]
|
|
208
|
+
t.path.drop(1).each do |key|
|
|
209
|
+
return ABSENT unless value.is_a?(Hash)
|
|
210
|
+
value = value[key]
|
|
211
|
+
end
|
|
212
|
+
value
|
|
213
|
+
end
|
|
214
|
+
|
|
215
|
+
# Stores a value, copying the struct Hashes on the way so nothing else shares them
|
|
216
|
+
#
|
|
217
|
+
# @param t [Target] replaced column
|
|
218
|
+
# @param r [Integer] row index within the row group
|
|
219
|
+
# @param value [Object] the new value
|
|
220
|
+
# @return [void]
|
|
221
|
+
def assign(t, r, value)
|
|
222
|
+
column = @data[t.path.first]
|
|
223
|
+
return column[r] = value if t.path.size == 1
|
|
224
|
+
hash = column[r] = column[r].dup
|
|
225
|
+
t.path[1...-1].each { |key| hash = hash[key] = hash[key].dup }
|
|
226
|
+
hash[t.path.last] = value
|
|
227
|
+
end
|
|
228
|
+
|
|
229
|
+
# @param r [Integer] row index within the row group
|
|
230
|
+
# @return [Hash{String => Object}] the whole row as it stands, keyed by top-level field name
|
|
231
|
+
def row_hash(r)
|
|
232
|
+
@schema.fields.to_h { |f| [f.name, @data[f.name][r]] }
|
|
233
|
+
end
|
|
234
|
+
|
|
235
|
+
# Tier 1: every kept column chunk is copied as it is
|
|
236
|
+
#
|
|
237
|
+
# @param i [Integer] row group index
|
|
238
|
+
# @param writer [Writer] the output
|
|
239
|
+
# @return [void]
|
|
240
|
+
def copy_row_group(i, writer)
|
|
241
|
+
copies = @output_columns.each_with_index.to_h { |col, j| [j, copied_chunk(i, col)] }
|
|
242
|
+
writer.write_row_group(rows(i), copies: copies, sorting_columns: sorting_columns(i, []))
|
|
243
|
+
@report.row_groups[:copied] += 1
|
|
244
|
+
end
|
|
245
|
+
|
|
246
|
+
# Tier 2: the chunks of the changed leaf columns are encoded again, the rest are copied
|
|
247
|
+
#
|
|
248
|
+
# @param i [Integer] row group index
|
|
249
|
+
# @param writer [Writer] the output
|
|
250
|
+
# @param changed [Array<Target>] leaf targets whose values changed
|
|
251
|
+
# @return [void]
|
|
252
|
+
def rewrite_columns(i, writer, changed)
|
|
253
|
+
rewritten = changed.map { |t| t.column.index }
|
|
254
|
+
changed.map { |t| t.path.first }.uniq.each { |name| writer.buffer_field(name, @data[name]) }
|
|
255
|
+
copies = {}
|
|
256
|
+
codecs = {}
|
|
257
|
+
blooms = {}
|
|
258
|
+
@output_columns.each_with_index do |col, j|
|
|
259
|
+
if rewritten.include?(col.index)
|
|
260
|
+
chunk_settings(i, col, j, codecs, blooms)
|
|
261
|
+
else
|
|
262
|
+
copies[j] = copied_chunk(i, col)
|
|
263
|
+
end
|
|
264
|
+
end
|
|
265
|
+
writer.write_row_group(rows(i), copies: copies, codecs: codecs, bloom_filters: blooms,
|
|
266
|
+
sorting_columns: sorting_columns(i, rewritten))
|
|
267
|
+
@report.row_groups[:rewritten] += 1
|
|
268
|
+
end
|
|
269
|
+
|
|
270
|
+
# Tier 3: the remaining rows are encoded again, every column of them
|
|
271
|
+
#
|
|
272
|
+
# @param i [Integer] row group index
|
|
273
|
+
# @param writer [Writer] the output
|
|
274
|
+
# @param deleted [Array<Boolean>] deleted flag per row
|
|
275
|
+
# @param changed [Array<Target>] targets whose values changed
|
|
276
|
+
# @return [void]
|
|
277
|
+
def rewrite_row_group(i, writer, deleted, changed)
|
|
278
|
+
@report.row_groups[:rewritten] += 1
|
|
279
|
+
kept = deleted.count(false)
|
|
280
|
+
return if kept.zero?
|
|
281
|
+
load(i, @output_schema.fields.map(&:name))
|
|
282
|
+
@output_schema.fields.each do |f|
|
|
283
|
+
values = @data[f.name]
|
|
284
|
+
values = values.reject.with_index { |_, r| deleted[r] } if kept < values.size
|
|
285
|
+
writer.buffer_field(f.name, values)
|
|
286
|
+
end
|
|
287
|
+
codecs = {}
|
|
288
|
+
blooms = {}
|
|
289
|
+
@output_columns.each_with_index { |col, j| chunk_settings(i, col, j, codecs, blooms) }
|
|
290
|
+
rewritten = changed.flat_map { |t| t.field.leaves.map(&:index) }
|
|
291
|
+
writer.write_row_group(kept, codecs: codecs, bloom_filters: blooms, sorting_columns: sorting_columns(i, rewritten))
|
|
292
|
+
end
|
|
293
|
+
|
|
294
|
+
# Codec and bloom filter of a re-encoded chunk follow the source chunk: the same codec
|
|
295
|
+
# (unless +compression:+ was given; LZO cannot be written and becomes Snappy), and a bloom
|
|
296
|
+
# filter when the source chunk had one
|
|
297
|
+
#
|
|
298
|
+
# @param i [Integer] row group index
|
|
299
|
+
# @param col [Schema::Column] the input column
|
|
300
|
+
# @param j [Integer] the output column index
|
|
301
|
+
# @param codecs [Hash{Integer => Integer}] collects output column index => codec id
|
|
302
|
+
# @param blooms [Hash{Integer => Boolean}] collects output column index => true
|
|
303
|
+
# @return [void]
|
|
304
|
+
def chunk_settings(i, col, j, codecs, blooms)
|
|
305
|
+
meta = @reader.row_groups[i].columns.fetch(col.index).meta_data
|
|
306
|
+
return unless meta
|
|
307
|
+
codecs[j] = (meta.codec == Format::Codec::LZO) ? Format::Codec::SNAPPY : meta.codec if @keep_codecs
|
|
308
|
+
blooms[j] = true if meta.bloom_filter_offset
|
|
309
|
+
end
|
|
310
|
+
|
|
311
|
+
# The row group's sorting columns that still hold: dropped or changed columns end the list,
|
|
312
|
+
# since the columns after them were only sorted within their runs
|
|
313
|
+
#
|
|
314
|
+
# @param i [Integer] row group index
|
|
315
|
+
# @param rewritten [Array<Integer>] indexes of input columns whose values changed
|
|
316
|
+
# @return [Array<Format::SortingColumn>, nil]
|
|
317
|
+
def sorting_columns(i, rewritten)
|
|
318
|
+
out = []
|
|
319
|
+
(@reader.row_groups[i].sorting_columns || []).each do |sc|
|
|
320
|
+
col = @schema.columns[sc.column_idx]
|
|
321
|
+
j = col && !rewritten.include?(col.index) && @output_columns.index(col)
|
|
322
|
+
break unless j
|
|
323
|
+
out << Format::SortingColumn.new(column_idx: j, descending: sc.descending, nulls_first: sc.nulls_first)
|
|
324
|
+
end
|
|
325
|
+
out.empty? ? nil : out
|
|
326
|
+
end
|
|
327
|
+
|
|
328
|
+
# @param i [Integer] row group index
|
|
329
|
+
# @param col [Schema::Column] the input column
|
|
330
|
+
# @return [Writer::CopiedChunk] the chunk's pages, page index and bloom filter
|
|
331
|
+
# @raise [UnsupportedError] for an encrypted chunk or one stored in another file
|
|
332
|
+
# @raise [FormatError] when a page header is corrupt or the chunk overruns the file
|
|
333
|
+
def copied_chunk(i, col)
|
|
334
|
+
chunk = @reader.row_groups[i].columns.fetch(col.index)
|
|
335
|
+
meta = chunk.meta_data or raise UnsupportedError, "Column chunk without metadata (encrypted?)"
|
|
336
|
+
raise UnsupportedError, "Column chunks in external files are not supported" if chunk.file_path
|
|
337
|
+
column_index, offset_index = @reader.page_index(i, col)
|
|
338
|
+
start, bytes = chunk_bytes(col, meta, offset_index)
|
|
339
|
+
column_index &&= read_at(chunk.column_index_offset, chunk.column_index_length)
|
|
340
|
+
Writer::CopiedChunk.new(chunk, start, bytes, column_index, offset_index, bloom_bytes(i, col, meta))
|
|
341
|
+
end
|
|
342
|
+
|
|
343
|
+
# The chunk's pages. They are walked header by header rather than trusting
|
|
344
|
+
# total_compressed_size, which some writers get wrong: copying too little breaks the
|
|
345
|
+
# chunk, and copying too much could carry over bytes of a neighbouring chunk.
|
|
346
|
+
#
|
|
347
|
+
# @param col [Schema::Column] the input column
|
|
348
|
+
# @param meta [Format::ColumnMetaData] the chunk's metadata
|
|
349
|
+
# @param offset_index [Format::OffsetIndex, nil] the chunk's OffsetIndex, whose pages are
|
|
350
|
+
# included even when num_values is reached before them
|
|
351
|
+
# @return [Array(Integer, String)] file offset of the first page, and the pages' bytes
|
|
352
|
+
# @raise [FormatError] when a page header is corrupt or the chunk overruns the file
|
|
353
|
+
def chunk_bytes(col, meta, offset_index)
|
|
354
|
+
start = meta.data_page_offset
|
|
355
|
+
dict = meta.dictionary_page_offset
|
|
356
|
+
start = dict if dict&.positive? && dict < start
|
|
357
|
+
last = offset_index&.page_locations&.last
|
|
358
|
+
min_end = last ? last.offset + last.compressed_page_size - start : 0
|
|
359
|
+
@io.seek(start)
|
|
360
|
+
buf = (@io.read(meta.total_compressed_size) || "").b
|
|
361
|
+
pos = 0
|
|
362
|
+
seen = 0
|
|
363
|
+
while seen < meta.num_values || pos < min_end
|
|
364
|
+
begin
|
|
365
|
+
header, body = Format::PageHeader.decode(buf, pos)
|
|
366
|
+
rescue Thrift::Error
|
|
367
|
+
raise FormatError, "Corrupt page header in #{col.dotted_path}" unless read_more(buf, start, READ_MORE)
|
|
368
|
+
retry
|
|
369
|
+
end
|
|
370
|
+
size = header.compressed_page_size
|
|
371
|
+
raise FormatError, "Negative page size in #{col.dotted_path}" if size.nil? || size.negative?
|
|
372
|
+
short = body + size - buf.bytesize
|
|
373
|
+
if short.positive? && read_more(buf, start, short) < short
|
|
374
|
+
raise FormatError, "Column #{col.dotted_path}: page overruns the file"
|
|
375
|
+
end
|
|
376
|
+
pos = body + size
|
|
377
|
+
seen += header.data_page_header&.num_values || header.data_page_header_v2&.num_values || 0
|
|
378
|
+
end
|
|
379
|
+
[start, (pos == buf.bytesize) ? buf : buf.byteslice(0, pos)]
|
|
380
|
+
end
|
|
381
|
+
|
|
382
|
+
# Appends up to +count+ bytes that follow +buf+ in the file
|
|
383
|
+
#
|
|
384
|
+
# @param buf [String] bytes read from +start+ on; appended to
|
|
385
|
+
# @param start [Integer] file offset of +buf+
|
|
386
|
+
# @param count [Integer] bytes wanted
|
|
387
|
+
# @return [Integer] bytes appended, 0 at the end of the file
|
|
388
|
+
def read_more(buf, start, count)
|
|
389
|
+
@io.seek(start + buf.bytesize)
|
|
390
|
+
more = @io.read(count)
|
|
391
|
+
return 0 if more.nil?
|
|
392
|
+
buf << more.b
|
|
393
|
+
more.bytesize
|
|
394
|
+
end
|
|
395
|
+
|
|
396
|
+
# @param offset [Integer, nil] file offset
|
|
397
|
+
# @param length [Integer, nil] byte count
|
|
398
|
+
# @return [String, nil] the bytes, nil when absent or cut off
|
|
399
|
+
def read_at(offset, length)
|
|
400
|
+
return nil unless offset && length&.positive?
|
|
401
|
+
@io.seek(offset)
|
|
402
|
+
bytes = @io.read(length)
|
|
403
|
+
(bytes&.bytesize == length) ? bytes.b : nil
|
|
404
|
+
end
|
|
405
|
+
|
|
406
|
+
# The chunk's bloom filter as stored. Filters written without a length (older writers) are
|
|
407
|
+
# decoded and encoded again. A filter that cannot be read is left out, which only costs
|
|
408
|
+
# pruning.
|
|
409
|
+
#
|
|
410
|
+
# @param i [Integer] row group index
|
|
411
|
+
# @param col [Schema::Column] the input column
|
|
412
|
+
# @param meta [Format::ColumnMetaData] the chunk's metadata
|
|
413
|
+
# @return [String, nil] header and bitset
|
|
414
|
+
def bloom_bytes(i, col, meta)
|
|
415
|
+
return nil unless meta.bloom_filter_offset
|
|
416
|
+
return read_at(meta.bloom_filter_offset, meta.bloom_filter_length) if meta.bloom_filter_length
|
|
417
|
+
@reader.bloom_filter(i, col)&.encode
|
|
418
|
+
rescue FormatError
|
|
419
|
+
nil
|
|
420
|
+
end
|
|
421
|
+
|
|
422
|
+
# Reads top-level fields of row group +i+ into @data, unless they are there already
|
|
423
|
+
#
|
|
424
|
+
# @param i [Integer] row group index
|
|
425
|
+
# @param names [Array<String>] top-level field names
|
|
426
|
+
# @return [void]
|
|
427
|
+
def load(i, names)
|
|
428
|
+
missing = names - @data.keys
|
|
429
|
+
return if missing.empty?
|
|
430
|
+
n = rows(i)
|
|
431
|
+
@rows_read += n if @data.empty?
|
|
432
|
+
@data.merge!(@reader.read(as: :columns, columns: missing, from: @first_rows[i], limit: n))
|
|
433
|
+
end
|
|
434
|
+
|
|
435
|
+
# @param i [Integer] row group index
|
|
436
|
+
# @return [Integer] rows in the row group
|
|
437
|
+
def rows(i) = @reader.row_groups[i].num_rows
|
|
438
|
+
|
|
439
|
+
# @return [Hash{String => String, nil}] the input's key/value metadata, without the keys that
|
|
440
|
+
# describe the columns when some are dropped
|
|
441
|
+
def copied_metadata
|
|
442
|
+
meta = @reader.metadata
|
|
443
|
+
@drops.empty? ? meta : meta.except(*COLUMN_METADATA_KEYS)
|
|
444
|
+
end
|
|
445
|
+
|
|
446
|
+
# @param block [Proc] a replace block
|
|
447
|
+
# @return [Boolean] whether the block takes the row as a second parameter
|
|
448
|
+
def wants_row?(block)
|
|
449
|
+
params = block.parameters
|
|
450
|
+
params.any? { |type, _| type == :rest } || params.count { |type, _| type == :req || type == :opt } >= 2
|
|
451
|
+
end
|
|
452
|
+
|
|
453
|
+
# Resolves a column named in a statement: a top-level field, or a struct member by dotted path
|
|
454
|
+
#
|
|
455
|
+
# @param name [String] the column as named
|
|
456
|
+
# @param verb [String] "replace" or "drop", for error messages
|
|
457
|
+
# @return [Array(Schema::Field, Array<String>)] the field and its path of names
|
|
458
|
+
# @raise [ArgumentError] when there is no such column, or it is inside a list or map
|
|
459
|
+
def resolve(name, verb)
|
|
460
|
+
field = @schema.field(name)
|
|
461
|
+
return [field, [name]] if field
|
|
462
|
+
parts = name.split(".")
|
|
463
|
+
field = @schema.field(parts.first) or raise ArgumentError, "#{verb}: no such column #{name.inspect}"
|
|
464
|
+
parts.drop(1).each_with_index do |part, depth|
|
|
465
|
+
case field.kind
|
|
466
|
+
when :struct
|
|
467
|
+
field = field.children_by_name[part] or raise ArgumentError, "#{verb}: no such column #{name.inspect}"
|
|
468
|
+
when :list, :map
|
|
469
|
+
whole = parts.first(depth + 1).join(".")
|
|
470
|
+
container = (field.kind == :map) ? "Hash" : "Array"
|
|
471
|
+
hint = (verb == "replace") ? " with a block that receives its #{container}" : ""
|
|
472
|
+
raise ArgumentError, "#{verb}: #{name} is inside the #{field.kind} #{whole}; #{verb} #{whole} instead#{hint}"
|
|
473
|
+
else
|
|
474
|
+
raise ArgumentError, "#{verb}: no such column #{name.inspect}"
|
|
475
|
+
end
|
|
476
|
+
end
|
|
477
|
+
[field, parts]
|
|
478
|
+
end
|
|
479
|
+
|
|
480
|
+
# @param name [String] the column as named in the statement
|
|
481
|
+
# @param statement [Statement] the replace statement
|
|
482
|
+
# @return [Target]
|
|
483
|
+
# @raise [ArgumentError] when the column does not exist, is inside a list or map, or a
|
|
484
|
+
# constant cannot be stored in it (nil in a required column, a value of the wrong type)
|
|
485
|
+
def target(name, statement)
|
|
486
|
+
field, path = resolve(name, "replace")
|
|
487
|
+
if statement.block.nil?
|
|
488
|
+
value = statement.constants[name]
|
|
489
|
+
if value.nil?
|
|
490
|
+
raise ArgumentError, "replace: #{name} is required (null: false) and cannot be set to nil" unless field.optional
|
|
491
|
+
elsif field.leaf?
|
|
492
|
+
begin
|
|
493
|
+
field.column.encoder.call(value)
|
|
494
|
+
rescue ArgumentError, TypeError, NoMethodError, RangeError => e
|
|
495
|
+
raise ArgumentError, "replace: cannot write #{value.inspect} to #{name}: #{e.message}"
|
|
496
|
+
end
|
|
497
|
+
end
|
|
498
|
+
end
|
|
499
|
+
Target.new(name, path, field, field.leaf? ? field.column : nil)
|
|
500
|
+
end
|
|
501
|
+
|
|
502
|
+
# @return [void]
|
|
503
|
+
# @raise [ArgumentError] when a replaced column is dropped
|
|
504
|
+
def check_drops!
|
|
505
|
+
@targets.flatten.each do |t|
|
|
506
|
+
dropped = @drops.find { |path| t.path.first(path.size) == path }
|
|
507
|
+
raise ArgumentError, "replace: #{t.name} is dropped by drop(#{dropped.join(".").inspect})" if dropped
|
|
508
|
+
end
|
|
509
|
+
end
|
|
510
|
+
|
|
511
|
+
# The input schema without the dropped fields
|
|
512
|
+
#
|
|
513
|
+
# @return [Schema]
|
|
514
|
+
# @raise [ArgumentError] when every column, or every member of a struct, would be dropped
|
|
515
|
+
def output_schema
|
|
516
|
+
return @schema if @drops.empty?
|
|
517
|
+
copy = Schema.from_elements(@schema.to_elements)
|
|
518
|
+
emptied = @drops.filter_map do |path|
|
|
519
|
+
parent = path[0...-1].reduce(copy.root) { |node, name| node&.children&.find { |c| c.name == name } }
|
|
520
|
+
node = parent&.children&.find { |c| c.name == path.last }
|
|
521
|
+
next unless node
|
|
522
|
+
parent.children.delete(node)
|
|
523
|
+
parent if parent.children.empty?
|
|
524
|
+
end
|
|
525
|
+
emptied.each do |node|
|
|
526
|
+
raise ArgumentError, "drop: cannot drop every column" if node.equal?(copy.root)
|
|
527
|
+
raise ArgumentError, "drop: cannot drop every member of #{node.path.join(".")}; drop #{node.path.join(".")} instead"
|
|
528
|
+
end
|
|
529
|
+
Schema.new(copy.root)
|
|
530
|
+
end
|
|
531
|
+
end
|
|
532
|
+
end
|
|
533
|
+
end
|