herringbone 0.1.0 → 0.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/README.md +146 -74
- data/bin/herringbone +38 -26
- data/lib/herringbone/active_record.rb +10 -13
- data/lib/herringbone/bloom_filter.rb +270 -0
- data/lib/herringbone/compression.rb +73 -16
- data/lib/herringbone/encodings/delta.rb +7 -7
- data/lib/herringbone/encodings/plain.rb +7 -7
- data/lib/herringbone/encodings/rle.rb +2 -2
- data/lib/herringbone/format.rb +53 -0
- data/lib/herringbone/inspector.rb +1388 -0
- data/lib/herringbone/reader/column_chunk_reader.rb +296 -0
- data/lib/herringbone/reader/column_cursor.rb +181 -0
- data/lib/herringbone/reader/page_stream.rb +339 -0
- data/lib/herringbone/reader/scan.rb +256 -0
- data/lib/herringbone/reader.rb +294 -272
- data/lib/herringbone/schema.rb +21 -76
- data/lib/herringbone/types.rb +4 -4
- data/lib/herringbone/version.rb +1 -1
- data/lib/herringbone/visualizer.rb +1096 -0
- data/lib/herringbone/writer.rb +258 -90
- data/lib/herringbone/xxhash.rb +319 -0
- data/lib/herringbone.rb +36 -9
- metadata +12 -32
|
@@ -0,0 +1,1388 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require "json"
|
|
4
|
+
require "zlib"
|
|
5
|
+
|
|
6
|
+
module Herringbone
|
|
7
|
+
# Examines a Parquet file using only its footer, page headers and page indexes. Values are
|
|
8
|
+
# never decompressed or decoded, so this works for files whose codecs are not installed and
|
|
9
|
+
# stays fast for big files (it seeks from page header to page header).
|
|
10
|
+
#
|
|
11
|
+
# File.open("data.parquet", "rb") do |io|
|
|
12
|
+
# inspector = Herringbone::Inspector.new(io)
|
|
13
|
+
# inspector.summary # => { file_size:, num_rows:, codecs:, ... }
|
|
14
|
+
# inspector.row_groups[0].column("name").pages
|
|
15
|
+
# inspector.to_h # everything, JSON-serializable
|
|
16
|
+
# puts inspector.report # readable text (bin/herringbone inspect)
|
|
17
|
+
# html = inspector.to_html # self-contained HTML page (see Visualizer)
|
|
18
|
+
# end
|
|
19
|
+
#
|
|
20
|
+
# Page headers and indexes are read lazily; #load_all reads them all up front, after which the
|
|
21
|
+
# inspector no longer needs the IO. The IO is never closed. After #verify_checksums, page CRC
|
|
22
|
+
# results are included in every output.
|
|
23
|
+
class Inspector
|
|
24
|
+
T = Format::Type
|
|
25
|
+
MAGIC = "PAR1"
|
|
26
|
+
WINDOW = 64 * 1024
|
|
27
|
+
|
|
28
|
+
# Just enough of parquet.thrift's BloomFilterHeader to learn the filter's size
|
|
29
|
+
class BloomFilterHeader < Thrift::Struct
|
|
30
|
+
field 1, :num_bytes, :i32
|
|
31
|
+
end
|
|
32
|
+
|
|
33
|
+
# Decoded min/max statistics (from a column chunk, a page header or a page index entry)
|
|
34
|
+
Stats = Struct.new(:min, :max, :null_count, :distinct_count, :min_exact, :max_exact, :source, :caveat,
|
|
35
|
+
keyword_init: true) do
|
|
36
|
+
def to_h = Inspector.jsonable(super.compact)
|
|
37
|
+
end
|
|
38
|
+
|
|
39
|
+
# One page header of a column chunk. +checksum+ is nil until the CRCs are verified
|
|
40
|
+
# (Inspector#verify_checksums), then :ok, :mismatch or :absent (the page has no CRC).
|
|
41
|
+
PageInfo = Struct.new(:index, :type, :offset, :header_size, :compressed_size, :uncompressed_size,
|
|
42
|
+
:num_values, :num_nulls, :num_rows, :first_row_index, :encoding, :definition_level_encoding,
|
|
43
|
+
:repetition_level_encoding, :definition_levels_byte_length, :repetition_levels_byte_length,
|
|
44
|
+
:is_compressed, :is_sorted, :statistics, :crc, :checksum, keyword_init: true) do
|
|
45
|
+
def total_size = header_size + compressed_size
|
|
46
|
+
def end_offset = offset + total_size
|
|
47
|
+
def body_offset = offset + header_size
|
|
48
|
+
def dictionary? = type == :DICTIONARY_PAGE
|
|
49
|
+
def data? = type == :DATA_PAGE || type == :DATA_PAGE_V2
|
|
50
|
+
|
|
51
|
+
# The CRC from the header as an unsigned 32-bit value (Thrift stores it as a signed i32)
|
|
52
|
+
def expected_crc = crc && (crc & 0xFFFF_FFFF)
|
|
53
|
+
|
|
54
|
+
def to_h
|
|
55
|
+
h = super
|
|
56
|
+
h[:statistics] = statistics&.to_h
|
|
57
|
+
h[:crc] = !crc.nil?
|
|
58
|
+
Inspector.jsonable(h.compact)
|
|
59
|
+
end
|
|
60
|
+
end
|
|
61
|
+
|
|
62
|
+
# ColumnIndex of one column chunk, with min/max decoded per page (nil for all-null pages)
|
|
63
|
+
ColumnIndexInfo = Struct.new(:offset, :length, :null_pages, :min_values, :max_values, :boundary_order,
|
|
64
|
+
:null_counts, :repetition_level_histograms, :definition_level_histograms, keyword_init: true) do
|
|
65
|
+
def to_h = Inspector.jsonable(super.compact)
|
|
66
|
+
end
|
|
67
|
+
|
|
68
|
+
PageLocation = Struct.new(:offset, :compressed_page_size, :first_row_index, keyword_init: true)
|
|
69
|
+
|
|
70
|
+
OffsetIndexInfo = Struct.new(:offset, :length, :page_locations, :unencoded_byte_array_data_bytes,
|
|
71
|
+
keyword_init: true) do
|
|
72
|
+
def to_h
|
|
73
|
+
h = super
|
|
74
|
+
h[:page_locations] = page_locations.map(&:to_h)
|
|
75
|
+
h.compact
|
|
76
|
+
end
|
|
77
|
+
end
|
|
78
|
+
|
|
79
|
+
# One column chunk of a row group
|
|
80
|
+
class ColumnChunkInfo
|
|
81
|
+
attr_reader :inspector, :row_group, :column, :chunk, :meta, :error
|
|
82
|
+
|
|
83
|
+
def initialize(inspector, row_group, column, chunk)
|
|
84
|
+
@inspector = inspector
|
|
85
|
+
@row_group = row_group
|
|
86
|
+
@column = column
|
|
87
|
+
@chunk = chunk
|
|
88
|
+
@meta = chunk.meta_data
|
|
89
|
+
end
|
|
90
|
+
|
|
91
|
+
def path = @column.dotted_path
|
|
92
|
+
def column_index_number = @column.index
|
|
93
|
+
def codec = Format::Codec::NAMES[@meta.codec] || @meta.codec
|
|
94
|
+
def encodings = (@meta.encodings || []).map { |e| Inspector.encoding_name(e) }
|
|
95
|
+
def num_values = @meta.num_values
|
|
96
|
+
def compressed_size = @meta.total_compressed_size
|
|
97
|
+
def uncompressed_size = @meta.total_uncompressed_size
|
|
98
|
+
def data_page_offset = @meta.data_page_offset
|
|
99
|
+
def external_file = @chunk.file_path
|
|
100
|
+
|
|
101
|
+
def compression_ratio
|
|
102
|
+
compressed_size.to_i.positive? ? uncompressed_size.to_f / compressed_size : nil
|
|
103
|
+
end
|
|
104
|
+
|
|
105
|
+
# Some writers store 0 when there is no dictionary page, and some store a data_page_offset
|
|
106
|
+
# of 0 for empty chunks (which would point at the magic bytes)
|
|
107
|
+
def dictionary_page_offset
|
|
108
|
+
d = @meta.dictionary_page_offset
|
|
109
|
+
d && d >= 4 && (d < @meta.data_page_offset.to_i || @meta.data_page_offset.to_i < 4) ? d : nil
|
|
110
|
+
end
|
|
111
|
+
|
|
112
|
+
# Where the chunk's first page starts
|
|
113
|
+
def start_offset = dictionary_page_offset || data_page_offset
|
|
114
|
+
|
|
115
|
+
# The end according to the metadata; some writers under-report it (see #end_offset)
|
|
116
|
+
def declared_end_offset = start_offset + compressed_size
|
|
117
|
+
|
|
118
|
+
# The end of the last page actually found (the declared end if the pages could not be walked)
|
|
119
|
+
def end_offset
|
|
120
|
+
last = pages.last
|
|
121
|
+
last ? [last.end_offset, declared_end_offset].max : declared_end_offset
|
|
122
|
+
end
|
|
123
|
+
|
|
124
|
+
def encoding_stats
|
|
125
|
+
(@meta.encoding_stats || []).map do |s|
|
|
126
|
+
{ page_type: Format::PageType::NAMES[s.page_type] || s.page_type,
|
|
127
|
+
encoding: Inspector.encoding_name(s.encoding), count: s.count }
|
|
128
|
+
end
|
|
129
|
+
end
|
|
130
|
+
|
|
131
|
+
def statistics
|
|
132
|
+
return @statistics if defined?(@statistics)
|
|
133
|
+
@statistics = @inspector.decode_statistics(@meta.statistics, @column)
|
|
134
|
+
end
|
|
135
|
+
|
|
136
|
+
def size_statistics
|
|
137
|
+
s = @meta.size_statistics
|
|
138
|
+
s&.to_h
|
|
139
|
+
end
|
|
140
|
+
|
|
141
|
+
def key_value_metadata
|
|
142
|
+
(@meta.key_value_metadata || []).to_h { |kv| [kv.key, kv.value] }
|
|
143
|
+
end
|
|
144
|
+
|
|
145
|
+
def bloom_filter_offset = @meta.bloom_filter_offset
|
|
146
|
+
|
|
147
|
+
# Bytes taken by the bloom filter (header + bitset); read from its header when the footer has no length
|
|
148
|
+
def bloom_filter_length
|
|
149
|
+
return nil unless bloom_filter_offset
|
|
150
|
+
return @meta.bloom_filter_length if @meta.bloom_filter_length
|
|
151
|
+
return @bloom_filter_length if defined?(@bloom_filter_length)
|
|
152
|
+
@bloom_filter_length = @inspector.bloom_filter_size(bloom_filter_offset)
|
|
153
|
+
end
|
|
154
|
+
|
|
155
|
+
def column_index_range
|
|
156
|
+
o = @chunk.column_index_offset
|
|
157
|
+
o && @chunk.column_index_length ? [o, @chunk.column_index_length] : nil
|
|
158
|
+
end
|
|
159
|
+
|
|
160
|
+
def offset_index_range
|
|
161
|
+
o = @chunk.offset_index_offset
|
|
162
|
+
o && @chunk.offset_index_length ? [o, @chunk.offset_index_length] : nil
|
|
163
|
+
end
|
|
164
|
+
|
|
165
|
+
# All page headers, in file order. Errors while walking (corrupt or truncated headers) end
|
|
166
|
+
# the walk; #error then says what went wrong and #pages holds the pages found before it.
|
|
167
|
+
def pages
|
|
168
|
+
@pages ||= begin
|
|
169
|
+
list, @error = @inspector.walk_pages(self)
|
|
170
|
+
apply_offset_index(list)
|
|
171
|
+
list
|
|
172
|
+
end
|
|
173
|
+
end
|
|
174
|
+
|
|
175
|
+
def data_pages = pages.select(&:data?)
|
|
176
|
+
def dictionary_page = pages.find(&:dictionary?)
|
|
177
|
+
|
|
178
|
+
# Number of entries in the dictionary (from its page header)
|
|
179
|
+
def dictionary_size = dictionary_page&.num_values
|
|
180
|
+
|
|
181
|
+
def column_index
|
|
182
|
+
return @column_index if defined?(@column_index)
|
|
183
|
+
@column_index = @inspector.read_column_index(self)
|
|
184
|
+
end
|
|
185
|
+
|
|
186
|
+
def offset_index
|
|
187
|
+
return @offset_index if defined?(@offset_index)
|
|
188
|
+
@offset_index = @inspector.read_offset_index(self)
|
|
189
|
+
end
|
|
190
|
+
|
|
191
|
+
def null_count
|
|
192
|
+
return statistics.null_count if statistics&.null_count
|
|
193
|
+
counts = data_pages.map(&:num_nulls)
|
|
194
|
+
counts.all? ? counts.sum : nil
|
|
195
|
+
end
|
|
196
|
+
|
|
197
|
+
# Reads every page body (compressed bytes, as stored) and checks it against the CRC in its
|
|
198
|
+
# page header. Sets PageInfo#checksum on each page and returns the pages' statuses.
|
|
199
|
+
def verify_checksums
|
|
200
|
+
pages.map { |p| p.checksum = @inspector.page_checksum(p) }
|
|
201
|
+
end
|
|
202
|
+
|
|
203
|
+
# Page statistics (from page headers) that disagree with the ColumnIndex entry for the
|
|
204
|
+
# same page. Each: { page:, data_page:, field:, page_value:, index_value: } where +page+
|
|
205
|
+
# is the page's position in #pages, +data_page+ its position among the data pages (and
|
|
206
|
+
# in the ColumnIndex), +field+ one of :min, :max, :null_count, :null_page, :page_count.
|
|
207
|
+
# A ColumnIndex bound that is wider than the page's (e.g. a truncated string prefix) is
|
|
208
|
+
# allowed; one that is narrower, or a differing null count, is reported. Empty when the
|
|
209
|
+
# chunk has no ColumnIndex or its pages carry no statistics.
|
|
210
|
+
def index_mismatches
|
|
211
|
+
@index_mismatches ||= @inspector.compare_page_index(self)
|
|
212
|
+
end
|
|
213
|
+
|
|
214
|
+
def to_h
|
|
215
|
+
Inspector.jsonable({
|
|
216
|
+
path: path,
|
|
217
|
+
column: @column.index,
|
|
218
|
+
type: Inspector.type_name(@column),
|
|
219
|
+
codec: codec,
|
|
220
|
+
encodings: encodings,
|
|
221
|
+
encoding_stats: encoding_stats,
|
|
222
|
+
num_values: num_values,
|
|
223
|
+
null_count: null_count,
|
|
224
|
+
compressed_size: compressed_size,
|
|
225
|
+
uncompressed_size: uncompressed_size,
|
|
226
|
+
compression_ratio: compression_ratio&.round(4),
|
|
227
|
+
file_offset: @chunk.file_offset,
|
|
228
|
+
start_offset: start_offset,
|
|
229
|
+
end_offset: end_offset,
|
|
230
|
+
data_page_offset: data_page_offset,
|
|
231
|
+
dictionary_page_offset: dictionary_page_offset,
|
|
232
|
+
dictionary: dictionary_page && { offset: dictionary_page.offset, num_values: dictionary_page.num_values,
|
|
233
|
+
compressed_size: dictionary_page.compressed_size, uncompressed_size: dictionary_page.uncompressed_size,
|
|
234
|
+
is_sorted: dictionary_page.is_sorted },
|
|
235
|
+
statistics: statistics&.to_h,
|
|
236
|
+
size_statistics: size_statistics,
|
|
237
|
+
key_value_metadata: key_value_metadata.empty? ? nil : key_value_metadata,
|
|
238
|
+
bloom_filter: bloom_filter_offset && { offset: bloom_filter_offset, length: bloom_filter_length },
|
|
239
|
+
column_index_offset: column_index_range&.first,
|
|
240
|
+
column_index_length: column_index_range&.last,
|
|
241
|
+
offset_index_offset: offset_index_range&.first,
|
|
242
|
+
offset_index_length: offset_index_range&.last,
|
|
243
|
+
external_file: external_file,
|
|
244
|
+
num_pages: pages.size,
|
|
245
|
+
num_data_pages: data_pages.size,
|
|
246
|
+
index_mismatches: index_mismatches.empty? ? nil : index_mismatches,
|
|
247
|
+
error: error,
|
|
248
|
+
pages: pages.map(&:to_h),
|
|
249
|
+
column_index: column_index&.to_h,
|
|
250
|
+
offset_index: offset_index&.to_h
|
|
251
|
+
}.compact)
|
|
252
|
+
end
|
|
253
|
+
|
|
254
|
+
def inspect
|
|
255
|
+
"#<#{self.class.name} #{path} rg=#{@row_group.index} #{codec} #{compressed_size}/#{uncompressed_size} bytes>"
|
|
256
|
+
end
|
|
257
|
+
|
|
258
|
+
private
|
|
259
|
+
|
|
260
|
+
# Page row counts come from the offset index when there is one (v1 pages don't carry them)
|
|
261
|
+
def apply_offset_index(list)
|
|
262
|
+
oi = offset_index or return
|
|
263
|
+
by_offset = list.select(&:data?).to_h { |p| [p.offset, p] }
|
|
264
|
+
locs = oi.page_locations
|
|
265
|
+
locs.each_with_index do |loc, i|
|
|
266
|
+
page = by_offset[loc.offset] or next
|
|
267
|
+
page.first_row_index = loc.first_row_index
|
|
268
|
+
next_first = locs[i + 1]&.first_row_index || @row_group.num_rows
|
|
269
|
+
page.num_rows ||= next_first - loc.first_row_index
|
|
270
|
+
end
|
|
271
|
+
end
|
|
272
|
+
end
|
|
273
|
+
|
|
274
|
+
# One row group
|
|
275
|
+
class RowGroupInfo
|
|
276
|
+
attr_reader :index, :row_group, :columns
|
|
277
|
+
|
|
278
|
+
def initialize(inspector, index, row_group, first_row)
|
|
279
|
+
@index = index
|
|
280
|
+
@row_group = row_group
|
|
281
|
+
@first_row = first_row
|
|
282
|
+
leaves = inspector.schema.columns
|
|
283
|
+
@columns = (row_group.columns || []).each_with_index.map do |cc, i|
|
|
284
|
+
column = leaves[i] or raise FormatError, "Row group #{index} has more column chunks than the schema has columns"
|
|
285
|
+
ColumnChunkInfo.new(inspector, self, column, cc)
|
|
286
|
+
end
|
|
287
|
+
end
|
|
288
|
+
|
|
289
|
+
def num_rows = @row_group.num_rows
|
|
290
|
+
def first_row = @first_row
|
|
291
|
+
def total_byte_size = @row_group.total_byte_size
|
|
292
|
+
def compressed_size = @row_group.total_compressed_size || @columns.sum(&:compressed_size)
|
|
293
|
+
def uncompressed_size = @columns.sum { |c| c.uncompressed_size.to_i }
|
|
294
|
+
def start_offset = @columns.map(&:start_offset).min
|
|
295
|
+
def end_offset = @columns.map(&:end_offset).max
|
|
296
|
+
def column(path) = @columns.find { |c| c.path == path.to_s || c.column.path == Array(path) }
|
|
297
|
+
|
|
298
|
+
def sorting_columns
|
|
299
|
+
(@row_group.sorting_columns || []).map do |s|
|
|
300
|
+
{ column: @columns[s.column_idx]&.path || s.column_idx, descending: s.descending, nulls_first: s.nulls_first }
|
|
301
|
+
end
|
|
302
|
+
end
|
|
303
|
+
|
|
304
|
+
def to_h
|
|
305
|
+
Inspector.jsonable({
|
|
306
|
+
index: index,
|
|
307
|
+
ordinal: @row_group.ordinal,
|
|
308
|
+
num_rows: num_rows,
|
|
309
|
+
first_row: first_row,
|
|
310
|
+
total_byte_size: total_byte_size,
|
|
311
|
+
compressed_size: compressed_size,
|
|
312
|
+
uncompressed_size: uncompressed_size,
|
|
313
|
+
file_offset: @row_group.file_offset,
|
|
314
|
+
start_offset: start_offset,
|
|
315
|
+
end_offset: end_offset,
|
|
316
|
+
sorting_columns: sorting_columns,
|
|
317
|
+
columns: @columns.map(&:to_h)
|
|
318
|
+
}.compact)
|
|
319
|
+
end
|
|
320
|
+
|
|
321
|
+
def inspect
|
|
322
|
+
"#<#{self.class.name} #{index} rows=#{num_rows} columns=#{@columns.size}>"
|
|
323
|
+
end
|
|
324
|
+
end
|
|
325
|
+
|
|
326
|
+
attr_reader :metadata, :schema, :file_size, :footer_size
|
|
327
|
+
|
|
328
|
+
# A label for the file (the basename of the IO's path, when it has one)
|
|
329
|
+
attr_reader :name
|
|
330
|
+
|
|
331
|
+
# +io+ is a random-access IO (responds to #seek and #read, e.g. File.open(path, "rb")). It is
|
|
332
|
+
# left open.
|
|
333
|
+
def initialize(io)
|
|
334
|
+
@io = io
|
|
335
|
+
unless @io.respond_to?(:read) && @io.respond_to?(:seek)
|
|
336
|
+
raise ArgumentError, "Herringbone::Inspector expects an IO that supports #seek and #read " \
|
|
337
|
+
"(e.g. File.open(path, \"rb\")), got #{io.class}"
|
|
338
|
+
end
|
|
339
|
+
@name = @io.respond_to?(:path) && @io.path ? File.basename(@io.path.to_s) : nil
|
|
340
|
+
read_footer
|
|
341
|
+
@schema = Schema.from_elements(@metadata.schema)
|
|
342
|
+
end
|
|
343
|
+
|
|
344
|
+
def num_rows = @metadata.num_rows
|
|
345
|
+
def created_by = @metadata.created_by
|
|
346
|
+
def version = @metadata.version
|
|
347
|
+
def columns = @schema.columns
|
|
348
|
+
def footer_offset = @file_size - 8 - @footer_size
|
|
349
|
+
|
|
350
|
+
def row_groups
|
|
351
|
+
@row_groups ||= begin
|
|
352
|
+
first = 0
|
|
353
|
+
(@metadata.row_groups || []).each_with_index.map do |rg, i|
|
|
354
|
+
info = RowGroupInfo.new(self, i, rg, first)
|
|
355
|
+
first += rg.num_rows.to_i
|
|
356
|
+
info
|
|
357
|
+
end
|
|
358
|
+
end
|
|
359
|
+
end
|
|
360
|
+
|
|
361
|
+
def column_chunks = row_groups.flat_map(&:columns)
|
|
362
|
+
|
|
363
|
+
# Walks every page header and page index now (e.g. before closing the file)
|
|
364
|
+
def load_all
|
|
365
|
+
column_chunks.each do |c|
|
|
366
|
+
c.pages
|
|
367
|
+
c.column_index
|
|
368
|
+
c.offset_index
|
|
369
|
+
c.bloom_filter_length
|
|
370
|
+
end
|
|
371
|
+
self
|
|
372
|
+
end
|
|
373
|
+
|
|
374
|
+
def page_index? = column_chunks.any? { |c| c.column_index_range || c.offset_index_range }
|
|
375
|
+
def bloom_filters? = column_chunks.any?(&:bloom_filter_offset)
|
|
376
|
+
|
|
377
|
+
def key_value_metadata
|
|
378
|
+
(@metadata.key_value_metadata || []).map { |kv| describe_key_value(kv.key, kv.value) }
|
|
379
|
+
end
|
|
380
|
+
|
|
381
|
+
def column_orders
|
|
382
|
+
orders = @metadata.column_orders or return nil
|
|
383
|
+
orders.map { |o| o.type_order ? "TYPE_DEFINED_ORDER" : "UNKNOWN" }
|
|
384
|
+
end
|
|
385
|
+
|
|
386
|
+
def summary
|
|
387
|
+
codecs = column_chunks.map(&:codec).uniq
|
|
388
|
+
{
|
|
389
|
+
name: @name,
|
|
390
|
+
file_size: @file_size,
|
|
391
|
+
footer_size: @footer_size,
|
|
392
|
+
footer_offset: footer_offset,
|
|
393
|
+
format_version: version,
|
|
394
|
+
created_by: created_by,
|
|
395
|
+
num_rows: num_rows,
|
|
396
|
+
num_row_groups: row_groups.size,
|
|
397
|
+
num_columns: columns.size,
|
|
398
|
+
codecs: codecs,
|
|
399
|
+
compressed_size: column_chunks.sum { |c| c.compressed_size.to_i },
|
|
400
|
+
uncompressed_size: column_chunks.sum { |c| c.uncompressed_size.to_i },
|
|
401
|
+
page_index: page_index?,
|
|
402
|
+
bloom_filters: bloom_filters?,
|
|
403
|
+
column_orders: column_orders
|
|
404
|
+
}.tap { |h| h[:checksums] = checksum_summary.except(:mismatches) if checksums_verified? }
|
|
405
|
+
end
|
|
406
|
+
|
|
407
|
+
# Reads every page body and checks it against the CRC32 in its page header (the CRC covers
|
|
408
|
+
# the page data as stored: compressed, and for v2 pages the levels plus the compressed
|
|
409
|
+
# values). Nothing is decompressed. Sets PageInfo#checksum on every page and returns
|
|
410
|
+
# #checksum_summary. Needs the IO, so call it before closing the file.
|
|
411
|
+
def verify_checksums
|
|
412
|
+
column_chunks.each(&:verify_checksums)
|
|
413
|
+
@checksums_verified = true
|
|
414
|
+
checksum_summary
|
|
415
|
+
end
|
|
416
|
+
|
|
417
|
+
def checksums_verified? = @checksums_verified == true
|
|
418
|
+
|
|
419
|
+
# After #verify_checksums: { ok:, mismatch:, absent:, mismatches: [{ row_group:, column:, page:, type:, offset:, crc:, actual: }] }
|
|
420
|
+
def checksum_summary
|
|
421
|
+
return nil unless checksums_verified?
|
|
422
|
+
all = column_chunks.flat_map { |c| c.pages.map { |p| [c, p] } }
|
|
423
|
+
tally = all.map { |_, p| p.checksum }.tally
|
|
424
|
+
{
|
|
425
|
+
ok: tally.fetch(:ok, 0), mismatch: tally.fetch(:mismatch, 0), absent: tally.fetch(:absent, 0),
|
|
426
|
+
mismatches: all.select { |_, p| p.checksum == :mismatch }.map do |c, p|
|
|
427
|
+
{ row_group: c.row_group.index, column: c.path, page: p.index, type: p.type, offset: p.offset,
|
|
428
|
+
crc: p.expected_crc, actual: page_crc(p) }
|
|
429
|
+
end
|
|
430
|
+
}
|
|
431
|
+
end
|
|
432
|
+
|
|
433
|
+
# Every disagreement between page header statistics and the ColumnIndex, across the file,
|
|
434
|
+
# each with :row_group and :column added (see ColumnChunkInfo#index_mismatches)
|
|
435
|
+
def index_mismatches
|
|
436
|
+
column_chunks.flat_map do |c|
|
|
437
|
+
c.index_mismatches.map { |m| { row_group: c.row_group.index, column: c.path }.merge(m) }
|
|
438
|
+
end
|
|
439
|
+
end
|
|
440
|
+
|
|
441
|
+
# The decoded ARROW:schema key/value (see ArrowSchema.decode); nil when the file has none or
|
|
442
|
+
# it could not be decoded, and #arrow_schema_error then says why
|
|
443
|
+
def arrow_schema
|
|
444
|
+
return @arrow_schema if defined?(@arrow_schema)
|
|
445
|
+
@arrow_schema_error = nil
|
|
446
|
+
kv = (@metadata.key_value_metadata || []).find { |x| x.key == "ARROW:schema" }
|
|
447
|
+
@arrow_schema = kv && begin
|
|
448
|
+
ArrowSchema.decode(kv.value)
|
|
449
|
+
rescue StandardError => e
|
|
450
|
+
@arrow_schema_error = "could not decode ARROW:schema: #{e.message}"
|
|
451
|
+
nil
|
|
452
|
+
end
|
|
453
|
+
end
|
|
454
|
+
|
|
455
|
+
def arrow_schema_error
|
|
456
|
+
arrow_schema
|
|
457
|
+
@arrow_schema_error
|
|
458
|
+
end
|
|
459
|
+
|
|
460
|
+
# Schema tree: Hashes with name, repetition, types, levels (leaves) and children (groups)
|
|
461
|
+
def schema_tree
|
|
462
|
+
leaf_by_node = @schema.columns.to_h { |c| [c.node, c] }
|
|
463
|
+
build = lambda do |node|
|
|
464
|
+
h = { name: node.name, repetition: node.repetition }
|
|
465
|
+
if node.leaf?
|
|
466
|
+
col = leaf_by_node[node]
|
|
467
|
+
h[:physical_type] = T::NAMES[node.type]&.to_s
|
|
468
|
+
h[:type_length] = node.type_length
|
|
469
|
+
h[:column] = col.index
|
|
470
|
+
h[:path] = col.dotted_path
|
|
471
|
+
h[:max_definition_level] = col.max_definition_level
|
|
472
|
+
h[:max_repetition_level] = col.max_repetition_level
|
|
473
|
+
end
|
|
474
|
+
h[:logical_type] = Inspector.logical_type_name(node)
|
|
475
|
+
h[:converted_type] = Format::ConvertedType::NAMES[node.converted_type]&.to_s
|
|
476
|
+
h[:field_id] = node.field_id
|
|
477
|
+
h[:children] = node.children.map { |c| build.call(c) } if node.group?
|
|
478
|
+
h.compact
|
|
479
|
+
end
|
|
480
|
+
tree = @schema.root.children.map { |c| build.call(c) }
|
|
481
|
+
annotate_arrow_types(tree, arrow_schema[:fields]) if arrow_schema
|
|
482
|
+
tree
|
|
483
|
+
end
|
|
484
|
+
|
|
485
|
+
# Per leaf column: sums over all row groups, plus overall min/max where comparable
|
|
486
|
+
def column_totals
|
|
487
|
+
columns.map do |col|
|
|
488
|
+
chunks = row_groups.map { |rg| rg.columns[col.index] }.compact
|
|
489
|
+
compressed = chunks.sum { |c| c.compressed_size.to_i }
|
|
490
|
+
uncompressed = chunks.sum { |c| c.uncompressed_size.to_i }
|
|
491
|
+
nulls = chunks.map(&:null_count)
|
|
492
|
+
stats = chunks.map(&:statistics)
|
|
493
|
+
mins = stats.map { |s| s&.min }
|
|
494
|
+
maxes = stats.map { |s| s&.max }
|
|
495
|
+
{
|
|
496
|
+
column: col.index,
|
|
497
|
+
path: col.dotted_path,
|
|
498
|
+
type: Inspector.type_name(col),
|
|
499
|
+
codecs: chunks.map(&:codec).uniq,
|
|
500
|
+
encodings: chunks.flat_map(&:encodings).uniq,
|
|
501
|
+
num_values: chunks.sum { |c| c.num_values.to_i },
|
|
502
|
+
null_count: nulls.all? ? nulls.sum : nil,
|
|
503
|
+
compressed_size: compressed,
|
|
504
|
+
uncompressed_size: uncompressed,
|
|
505
|
+
compression_ratio: compressed.positive? ? (uncompressed.to_f / compressed).round(4) : nil,
|
|
506
|
+
num_pages: chunks.sum { |c| c.pages.size },
|
|
507
|
+
num_data_pages: chunks.sum { |c| c.data_pages.size },
|
|
508
|
+
dictionary_pages: chunks.count(&:dictionary_page),
|
|
509
|
+
dictionary_bytes: chunks.sum { |c| c.dictionary_page&.total_size.to_i },
|
|
510
|
+
min: mins.all? ? safe_extreme(mins, :min) : nil,
|
|
511
|
+
max: maxes.all? ? safe_extreme(maxes, :max) : nil
|
|
512
|
+
}.compact
|
|
513
|
+
end
|
|
514
|
+
end
|
|
515
|
+
|
|
516
|
+
# Byte ranges of the whole file, in offset order: magic, pages (or whole chunks when their pages
|
|
517
|
+
# could not be walked), bloom filters, page indexes, footer. Gaps right after a chunk at its
|
|
518
|
+
# file_offset are inline :column_metadata copies; other gaps are reported as :unknown. Each entry: { kind:, start:, length:, row_group:, column:, page: }
|
|
519
|
+
def layout
|
|
520
|
+
segs = [{ kind: :magic, start: 0, length: 4 }]
|
|
521
|
+
column_chunks.each do |c|
|
|
522
|
+
rg = c.row_group.index
|
|
523
|
+
col = c.column.index
|
|
524
|
+
if c.pages.empty?
|
|
525
|
+
segs << { kind: :chunk, start: c.start_offset, length: c.compressed_size, row_group: rg, column: col }
|
|
526
|
+
else
|
|
527
|
+
c.pages.each_with_index do |p, i|
|
|
528
|
+
segs << { kind: p.dictionary? ? :dictionary_page : :data_page, start: p.offset, length: p.total_size,
|
|
529
|
+
row_group: rg, column: col, page: i }
|
|
530
|
+
end
|
|
531
|
+
end
|
|
532
|
+
if c.bloom_filter_offset
|
|
533
|
+
segs << { kind: :bloom_filter, start: c.bloom_filter_offset, length: c.bloom_filter_length.to_i,
|
|
534
|
+
row_group: rg, column: col }
|
|
535
|
+
end
|
|
536
|
+
if (r = c.column_index_range)
|
|
537
|
+
segs << { kind: :column_index, start: r[0], length: r[1], row_group: rg, column: col }
|
|
538
|
+
end
|
|
539
|
+
if (r = c.offset_index_range)
|
|
540
|
+
segs << { kind: :offset_index, start: r[0], length: r[1], row_group: rg, column: col }
|
|
541
|
+
end
|
|
542
|
+
end
|
|
543
|
+
segs << { kind: :footer, start: footer_offset, length: @footer_size }
|
|
544
|
+
segs << { kind: :footer_length, start: @file_size - 8, length: 4 }
|
|
545
|
+
segs << { kind: :magic, start: @file_size - 4, length: 4 }
|
|
546
|
+
segs.sort_by! { |s| s[:start] }
|
|
547
|
+
# Some writers (old parquet-rs, parquet-mr for a while) store a copy of the ColumnMetaData
|
|
548
|
+
# right after the chunk, at ColumnChunk.file_offset
|
|
549
|
+
meta_copies = column_chunks.each_with_object({}) do |c, h|
|
|
550
|
+
fo = c.chunk.file_offset
|
|
551
|
+
h[fo] = c if fo && fo >= c.end_offset
|
|
552
|
+
end
|
|
553
|
+
out = []
|
|
554
|
+
pos = 0
|
|
555
|
+
segs.each do |s|
|
|
556
|
+
if s[:start] > pos
|
|
557
|
+
c = meta_copies[pos]
|
|
558
|
+
out << if c
|
|
559
|
+
{ kind: :column_metadata, start: pos, length: s[:start] - pos, row_group: c.row_group.index, column: c.column.index }
|
|
560
|
+
else
|
|
561
|
+
{ kind: :unknown, start: pos, length: s[:start] - pos }
|
|
562
|
+
end
|
|
563
|
+
end
|
|
564
|
+
out << s
|
|
565
|
+
pos = [pos, s[:start] + s[:length]].max
|
|
566
|
+
end
|
|
567
|
+
out
|
|
568
|
+
end
|
|
569
|
+
|
|
570
|
+
def to_h
|
|
571
|
+
Inspector.jsonable({
|
|
572
|
+
summary: summary,
|
|
573
|
+
key_value_metadata: key_value_metadata,
|
|
574
|
+
schema: schema_tree,
|
|
575
|
+
row_groups: row_groups.map(&:to_h),
|
|
576
|
+
column_totals: column_totals,
|
|
577
|
+
checksum_mismatches: checksum_summary&.fetch(:mismatches),
|
|
578
|
+
index_mismatches: index_mismatches
|
|
579
|
+
}.compact)
|
|
580
|
+
end
|
|
581
|
+
|
|
582
|
+
def to_json(*args) = to_h.to_json(*args)
|
|
583
|
+
|
|
584
|
+
# A self-contained HTML page showing the file's layout, see Visualizer
|
|
585
|
+
def to_html = Visualizer.new(self).to_html
|
|
586
|
+
|
|
587
|
+
# Readable text summary. With pages: true, lists every page header too.
|
|
588
|
+
def report(pages: false)
|
|
589
|
+
s = summary
|
|
590
|
+
out = []
|
|
591
|
+
out << "file: #{@name || "(IO)"}"
|
|
592
|
+
out << "size: #{Inspector.human_bytes(s[:file_size])} (#{s[:file_size]} bytes), footer #{s[:footer_size]} bytes at #{s[:footer_offset]}"
|
|
593
|
+
out << "rows: #{s[:num_rows]}, row groups: #{s[:num_row_groups]}, columns: #{s[:num_columns]}, format version #{s[:format_version]}"
|
|
594
|
+
out << "created by: #{s[:created_by]}"
|
|
595
|
+
out << "codecs: #{s[:codecs].join(", ")}; data #{Inspector.human_bytes(s[:compressed_size])} compressed, " \
|
|
596
|
+
"#{Inspector.human_bytes(s[:uncompressed_size])} uncompressed#{ratio_text(s[:uncompressed_size], s[:compressed_size])}"
|
|
597
|
+
out << "page index: #{s[:page_index] ? "yes" : "no"}, bloom filters: #{s[:bloom_filters] ? "yes" : "no"}"
|
|
598
|
+
if (cs = checksum_summary)
|
|
599
|
+
out << "page CRCs: #{cs[:ok]} ok, #{cs[:mismatch]} mismatched, #{cs[:absent]} without a CRC"
|
|
600
|
+
cs[:mismatches].each do |m|
|
|
601
|
+
out << " CRC MISMATCH: row group #{m[:row_group]} #{m[:column]} page #{m[:page]} (#{m[:type]} @#{m[:offset]}): " \
|
|
602
|
+
"header says #{format("%08x", m[:crc])}, data has #{format("%08x", m[:actual])}"
|
|
603
|
+
end
|
|
604
|
+
end
|
|
605
|
+
mismatches = index_mismatches
|
|
606
|
+
unless mismatches.empty?
|
|
607
|
+
out << "page statistics vs column index: #{mismatches.size} disagreement#{mismatches.size == 1 ? "" : "s"}"
|
|
608
|
+
mismatches.first(50).each { |m| out << " #{index_mismatch_text(m)}" }
|
|
609
|
+
out << " ..." if mismatches.size > 50
|
|
610
|
+
end
|
|
611
|
+
kvs = key_value_metadata
|
|
612
|
+
unless kvs.empty?
|
|
613
|
+
out << "key/value metadata:"
|
|
614
|
+
kvs.each do |kv|
|
|
615
|
+
out << " #{kv[:key]} (#{kv[:format]}, #{kv[:bytesize]} bytes): #{kv[:summary] || kv[:value].to_s[0, 80].inspect}"
|
|
616
|
+
out << " #{kv[:arrow_error]}" if kv[:arrow_error]
|
|
617
|
+
ArrowSchema.lines(kv[:arrow_schema][:fields]).each { |l| out << " #{l}" } if kv[:arrow_schema]
|
|
618
|
+
end
|
|
619
|
+
end
|
|
620
|
+
out << "schema:"
|
|
621
|
+
walk = lambda do |n, depth|
|
|
622
|
+
type = n[:children] ? "group" : [n[:physical_type], n[:type_length] && "(#{n[:type_length]})"].compact.join
|
|
623
|
+
ann = n[:logical_type] || n[:converted_type]
|
|
624
|
+
levels = n[:children] ? "" : " [def #{n[:max_definition_level]}, rep #{n[:max_repetition_level]}]"
|
|
625
|
+
arrow = n[:arrow_type] ? " arrow: #{n[:arrow_type]}" : ""
|
|
626
|
+
out << "#{" " * depth}#{n[:repetition]} #{type} #{n[:name]}#{ann ? " (#{ann})" : ""}#{levels}#{arrow}"
|
|
627
|
+
(n[:children] || []).each { |c| walk.call(c, depth + 1) }
|
|
628
|
+
end
|
|
629
|
+
schema_tree.each { |n| walk.call(n, 1) }
|
|
630
|
+
out << "columns:"
|
|
631
|
+
column_totals.each do |t|
|
|
632
|
+
range = t.key?(:min) ? " [#{Inspector.display(t[:min])} .. #{Inspector.display(t[:max])}]" : ""
|
|
633
|
+
out << " #{t[:path]}: #{t[:type]} #{t[:codecs].join(",")} #{t[:encodings].join(",")} " \
|
|
634
|
+
"#{Inspector.human_bytes(t[:compressed_size])}/#{Inspector.human_bytes(t[:uncompressed_size])}" \
|
|
635
|
+
"#{ratio_text(t[:uncompressed_size], t[:compressed_size])}, #{t[:num_values]} values" \
|
|
636
|
+
"#{t[:null_count] ? ", #{t[:null_count]} nulls" : ""}, #{t[:num_data_pages]} data pages#{range}"
|
|
637
|
+
end
|
|
638
|
+
row_groups.each do |rg|
|
|
639
|
+
sorting = rg.sorting_columns.map { |c| "#{c[:column]}#{c[:descending] ? " desc" : ""}" }
|
|
640
|
+
out << "row group #{rg.index}: #{rg.num_rows} rows, #{Inspector.human_bytes(rg.compressed_size)} " \
|
|
641
|
+
"at #{rg.start_offset}..#{rg.end_offset}#{sorting.empty? ? "" : ", sorted by #{sorting.join(", ")}"}"
|
|
642
|
+
rg.columns.each do |c|
|
|
643
|
+
st = c.statistics
|
|
644
|
+
range = st && (st.min || st.max) ? " [#{Inspector.display(st.min)} .. #{Inspector.display(st.max)}]" : ""
|
|
645
|
+
extras = []
|
|
646
|
+
extras << "dict #{c.dictionary_size} entries" if c.dictionary_page
|
|
647
|
+
extras << "column index" if c.column_index_range
|
|
648
|
+
extras << "offset index" if c.offset_index_range
|
|
649
|
+
extras << "bloom filter" if c.bloom_filter_offset
|
|
650
|
+
extras << "ERROR: #{c.error}" if c.error
|
|
651
|
+
out << " #{c.path}: #{c.codec} #{c.encodings.join(",")} " \
|
|
652
|
+
"#{c.compressed_size}/#{c.uncompressed_size} bytes#{ratio_text(c.uncompressed_size, c.compressed_size)}, " \
|
|
653
|
+
"#{c.num_values} values, #{c.pages.size} pages#{extras.empty? ? "" : ", #{extras.join(", ")}"}#{range}"
|
|
654
|
+
next unless pages
|
|
655
|
+
c.pages.each do |p|
|
|
656
|
+
st = p.statistics
|
|
657
|
+
out << " #{p.index}: #{p.type} @#{p.offset} header #{p.header_size} + #{p.compressed_size}/#{p.uncompressed_size} bytes, " \
|
|
658
|
+
"#{p.num_values} values#{p.num_nulls ? ", #{p.num_nulls} nulls" : ""}#{p.num_rows ? ", #{p.num_rows} rows" : ""}" \
|
|
659
|
+
"#{p.encoding ? " #{p.encoding}" : ""}#{crc_text(p)}" \
|
|
660
|
+
"#{st && (st.min || st.max) ? " [#{Inspector.display(st.min)} .. #{Inspector.display(st.max)}]" : ""}"
|
|
661
|
+
end
|
|
662
|
+
end
|
|
663
|
+
end
|
|
664
|
+
out.join("\n")
|
|
665
|
+
end
|
|
666
|
+
|
|
667
|
+
def inspect
|
|
668
|
+
"#<#{self.class.name} #{@name || "(IO)"} rows=#{num_rows} row_groups=#{row_groups.size} size=#{@file_size}>"
|
|
669
|
+
end
|
|
670
|
+
|
|
671
|
+
# ---- used by the info objects ----
|
|
672
|
+
|
|
673
|
+
# Walks page headers from the chunk's first page. Returns [pages, error_message_or_nil].
|
|
674
|
+
# Mirrors the reader's tolerance: a chunk may extend past its declared total_compressed_size.
|
|
675
|
+
def walk_pages(chunk)
|
|
676
|
+
return [[], "column chunk stored in external file #{chunk.external_file}"] if chunk.external_file
|
|
677
|
+
pages = []
|
|
678
|
+
pos = chunk.start_offset
|
|
679
|
+
limit = footer_offset
|
|
680
|
+
total = chunk.num_values.to_i
|
|
681
|
+
seen = 0
|
|
682
|
+
declared_end = chunk.declared_end_offset
|
|
683
|
+
# Walk until all values are accounted for (reading past the declared end if a writer
|
|
684
|
+
# under-reported it), then pick up any trailing pages still inside the declared range
|
|
685
|
+
while pos < limit && (seen < total || pos < declared_end)
|
|
686
|
+
trailing = seen >= total
|
|
687
|
+
begin
|
|
688
|
+
header, header_size = read_page_header(pos, limit)
|
|
689
|
+
page = page_info(pages.size, header, pos, header_size, chunk.column)
|
|
690
|
+
rescue Thrift::Error, FormatError => e
|
|
691
|
+
break if trailing
|
|
692
|
+
return [pages, "corrupt page header at #{pos}: #{e.message}"]
|
|
693
|
+
end
|
|
694
|
+
if page.end_offset > limit
|
|
695
|
+
break if trailing
|
|
696
|
+
return [pages, "page #{pages.size} at #{pos} overruns the data section (#{page.end_offset} > #{limit})"]
|
|
697
|
+
end
|
|
698
|
+
pages << page
|
|
699
|
+
seen += page.num_values.to_i if page.data?
|
|
700
|
+
pos = page.end_offset
|
|
701
|
+
end
|
|
702
|
+
[pages, seen < total ? "found #{seen} of #{total} values in page headers" : nil]
|
|
703
|
+
end
|
|
704
|
+
|
|
705
|
+
def read_column_index(chunk)
|
|
706
|
+
offset, length = chunk.column_index_range
|
|
707
|
+
return nil unless offset && length.positive?
|
|
708
|
+
ci = Format::ColumnIndex.decode(read_at(offset, length)).first
|
|
709
|
+
col = chunk.column
|
|
710
|
+
mins = ci.min_values || []
|
|
711
|
+
maxes = ci.max_values || []
|
|
712
|
+
nulls = ci.null_pages || []
|
|
713
|
+
ColumnIndexInfo.new(
|
|
714
|
+
offset: offset, length: length,
|
|
715
|
+
null_pages: nulls,
|
|
716
|
+
min_values: mins.each_with_index.map { |v, i| nulls[i] ? nil : decode_value(v, col) },
|
|
717
|
+
max_values: maxes.each_with_index.map { |v, i| nulls[i] ? nil : decode_value(v, col) },
|
|
718
|
+
boundary_order: Format::BoundaryOrder::NAMES[ci.boundary_order]&.to_s || ci.boundary_order,
|
|
719
|
+
null_counts: ci.null_counts,
|
|
720
|
+
repetition_level_histograms: ci.repetition_level_histograms,
|
|
721
|
+
definition_level_histograms: ci.definition_level_histograms
|
|
722
|
+
)
|
|
723
|
+
rescue Thrift::Error
|
|
724
|
+
nil
|
|
725
|
+
end
|
|
726
|
+
|
|
727
|
+
def read_offset_index(chunk)
|
|
728
|
+
offset, length = chunk.offset_index_range
|
|
729
|
+
return nil unless offset && length.positive?
|
|
730
|
+
oi = Format::OffsetIndex.decode(read_at(offset, length)).first
|
|
731
|
+
OffsetIndexInfo.new(
|
|
732
|
+
offset: offset, length: length,
|
|
733
|
+
page_locations: (oi.page_locations || []).map do |l|
|
|
734
|
+
PageLocation.new(offset: l.offset, compressed_page_size: l.compressed_page_size, first_row_index: l.first_row_index)
|
|
735
|
+
end,
|
|
736
|
+
unencoded_byte_array_data_bytes: oi.unencoded_byte_array_data_bytes
|
|
737
|
+
)
|
|
738
|
+
rescue Thrift::Error
|
|
739
|
+
nil
|
|
740
|
+
end
|
|
741
|
+
|
|
742
|
+
# Size in bytes of the bloom filter at +offset+ (its Thrift header plus the bitset)
|
|
743
|
+
def bloom_filter_size(offset)
|
|
744
|
+
buf = read_at(offset, 64)
|
|
745
|
+
header, size = BloomFilterHeader.decode(buf)
|
|
746
|
+
header.num_bytes ? size + header.num_bytes : nil
|
|
747
|
+
rescue Thrift::Error
|
|
748
|
+
nil
|
|
749
|
+
end
|
|
750
|
+
|
|
751
|
+
# :ok, :mismatch or :absent for one page (reads its body)
|
|
752
|
+
def page_checksum(page)
|
|
753
|
+
return :absent unless page.crc
|
|
754
|
+
page_crc(page) == page.expected_crc ? :ok : :mismatch
|
|
755
|
+
end
|
|
756
|
+
|
|
757
|
+
# CRC32 of a page's body as stored
|
|
758
|
+
def page_crc(page)
|
|
759
|
+
Zlib.crc32(read_at(page.body_offset, page.compressed_size))
|
|
760
|
+
end
|
|
761
|
+
|
|
762
|
+
# See ColumnChunkInfo#index_mismatches
|
|
763
|
+
def compare_page_index(chunk)
|
|
764
|
+
ci = chunk.column_index or return []
|
|
765
|
+
data = chunk.data_pages
|
|
766
|
+
if ci.null_pages.size != data.size
|
|
767
|
+
return [{ page: nil, data_page: nil, field: :page_count, page_value: data.size, index_value: ci.null_pages.size }]
|
|
768
|
+
end
|
|
769
|
+
order = Inspector.sort_order(chunk.column)
|
|
770
|
+
out = []
|
|
771
|
+
data.each_with_index do |p, k|
|
|
772
|
+
report = ->(field, pv, iv) { out << { page: p.index, data_page: k, field: field, page_value: pv, index_value: iv } }
|
|
773
|
+
idx_nulls = ci.null_counts&.[](k)
|
|
774
|
+
report.call(:null_count, p.num_nulls, idx_nulls) if idx_nulls && p.num_nulls && idx_nulls != p.num_nulls
|
|
775
|
+
st = p.statistics
|
|
776
|
+
# Legacy min/max were computed with another ordering, so they can't be compared
|
|
777
|
+
next if st.nil? || st.caveat || (st.min.nil? && st.max.nil?)
|
|
778
|
+
if ci.null_pages[k]
|
|
779
|
+
report.call(:null_page, "has min/max", "null page")
|
|
780
|
+
next
|
|
781
|
+
end
|
|
782
|
+
imin = ci.min_values[k]
|
|
783
|
+
imax = ci.max_values[k]
|
|
784
|
+
report.call(:min, st.min, imin) if st.min_exact != false && narrower?(imin, st.min, order, :min)
|
|
785
|
+
report.call(:max, st.max, imax) if st.max_exact != false && narrower?(imax, st.max, order, :max)
|
|
786
|
+
end
|
|
787
|
+
out
|
|
788
|
+
end
|
|
789
|
+
|
|
790
|
+
# Decodes a Format::Statistics into Ruby values via the column's type converter
|
|
791
|
+
def decode_statistics(st, column)
|
|
792
|
+
return nil unless st
|
|
793
|
+
order = Inspector.sort_order(column)
|
|
794
|
+
if st.min_value || st.max_value
|
|
795
|
+
min_raw = st.min_value
|
|
796
|
+
max_raw = st.max_value
|
|
797
|
+
source = "min_value/max_value"
|
|
798
|
+
elsif st.min || st.max
|
|
799
|
+
min_raw = st.min
|
|
800
|
+
max_raw = st.max
|
|
801
|
+
source = "min/max (legacy)"
|
|
802
|
+
binary = column.type == T::BYTE_ARRAY || column.type == T::FIXED_LEN_BYTE_ARRAY
|
|
803
|
+
if order == :unsigned
|
|
804
|
+
caveat = "legacy min/max were computed with signed comparison, which is wrong for unsigned-ordered " \
|
|
805
|
+
"values (strings, binary, unsigned integers); most readers ignore them"
|
|
806
|
+
elsif binary
|
|
807
|
+
caveat = "legacy min/max of binary-backed values were compared bytewise; most readers ignore them"
|
|
808
|
+
end
|
|
809
|
+
end
|
|
810
|
+
caveat ||= "sort order of #{T::NAMES[column.type]} is undefined; min/max may be meaningless" if order == :unknown && source
|
|
811
|
+
Stats.new(
|
|
812
|
+
min: decode_value(min_raw, column),
|
|
813
|
+
max: decode_value(max_raw, column),
|
|
814
|
+
null_count: st.null_count,
|
|
815
|
+
distinct_count: st.distinct_count,
|
|
816
|
+
min_exact: st.is_min_value_exact,
|
|
817
|
+
max_exact: st.is_max_value_exact,
|
|
818
|
+
source: source,
|
|
819
|
+
caveat: caveat
|
|
820
|
+
)
|
|
821
|
+
end
|
|
822
|
+
|
|
823
|
+
# Decodes one PLAIN-encoded statistics value (no length prefix for byte arrays)
|
|
824
|
+
def decode_value(bytes, column)
|
|
825
|
+
return nil if bytes.nil?
|
|
826
|
+
bytes = bytes.b
|
|
827
|
+
raw = case column.type
|
|
828
|
+
when T::BOOLEAN
|
|
829
|
+
return nil if bytes.empty?
|
|
830
|
+
bytes.getbyte(0) != 0
|
|
831
|
+
when T::BYTE_ARRAY, T::FIXED_LEN_BYTE_ARRAY
|
|
832
|
+
bytes
|
|
833
|
+
when T::INT96
|
|
834
|
+
return Inspector.hex(bytes) unless bytes.bytesize == 12
|
|
835
|
+
bytes.unpack("Q<L<")
|
|
836
|
+
else
|
|
837
|
+
width = Encodings::Plain::FORMATS[column.type][1]
|
|
838
|
+
return Inspector.hex(bytes) unless bytes.bytesize == width
|
|
839
|
+
Encodings::Plain.decode(bytes, 0, 1, column.type).first.first
|
|
840
|
+
end
|
|
841
|
+
conv = column.converter
|
|
842
|
+
conv ? conv.call(raw) : raw
|
|
843
|
+
rescue StandardError
|
|
844
|
+
Inspector.hex(bytes)
|
|
845
|
+
end
|
|
846
|
+
|
|
847
|
+
# Decodes the ARROW:schema key/value that Arrow writers (pyarrow, arrow-rs, DuckDB...) store:
|
|
848
|
+
# base64 of an Arrow IPC message whose header is a flatbuffer Schema (Arrow's Message.fbs and
|
|
849
|
+
# Schema.fbs). Pure Ruby and read-only; type names follow pyarrow's (str(field.type)).
|
|
850
|
+
#
|
|
851
|
+
# Inspector::ArrowSchema.decode(value)
|
|
852
|
+
# # => { endianness: "little", metadata: {...}, fields: [{ name: "a", type: "int32", nullable: true, ... }] }
|
|
853
|
+
#
|
|
854
|
+
# Fields carry :name, :type, :nullable and, when present, :children, :dictionary
|
|
855
|
+
# ({ index_type:, ordered:, id: }), :extension (ARROW:extension:name) and :metadata.
|
|
856
|
+
module ArrowSchema
|
|
857
|
+
class Error < StandardError; end
|
|
858
|
+
|
|
859
|
+
MAX_DEPTH = 64
|
|
860
|
+
MAX_FIELDS = 100_000
|
|
861
|
+
TIME_UNITS = %w[s ms us ns].freeze
|
|
862
|
+
MESSAGE_SCHEMA = 1
|
|
863
|
+
|
|
864
|
+
# A minimal flatbuffer reader: tables (through their vtables), scalars, strings, vectors of
|
|
865
|
+
# scalars and tables, and unions. Every read is bounds-checked; malformed input raises Error.
|
|
866
|
+
class FlatBuffer
|
|
867
|
+
def initialize(bytes)
|
|
868
|
+
@b = bytes.b
|
|
869
|
+
end
|
|
870
|
+
|
|
871
|
+
def root = table_at(u32(0))
|
|
872
|
+
def table_at(pos) = Table.new(self, pos)
|
|
873
|
+
|
|
874
|
+
def check(pos, len)
|
|
875
|
+
return if pos >= 0 && len >= 0 && pos + len <= @b.bytesize
|
|
876
|
+
raise Error, "flatbuffer read of #{len} bytes at #{pos} is out of bounds (#{@b.bytesize} bytes)"
|
|
877
|
+
end
|
|
878
|
+
|
|
879
|
+
def read(pos, len, fmt)
|
|
880
|
+
check(pos, len)
|
|
881
|
+
@b.byteslice(pos, len).unpack1(fmt)
|
|
882
|
+
end
|
|
883
|
+
|
|
884
|
+
def u8(pos) = read(pos, 1, "C")
|
|
885
|
+
def u16(pos) = read(pos, 2, "S<")
|
|
886
|
+
def i16(pos) = read(pos, 2, "s<")
|
|
887
|
+
def u32(pos) = read(pos, 4, "L<")
|
|
888
|
+
def i32(pos) = read(pos, 4, "l<")
|
|
889
|
+
def i64(pos) = read(pos, 8, "q<")
|
|
890
|
+
|
|
891
|
+
# Offsets are relative to where they are stored
|
|
892
|
+
def deref(pos) = pos + u32(pos)
|
|
893
|
+
|
|
894
|
+
def string(pos)
|
|
895
|
+
len = u32(pos)
|
|
896
|
+
check(pos + 4, len)
|
|
897
|
+
@b.byteslice(pos + 4, len).force_encoding(Encoding::UTF_8)
|
|
898
|
+
end
|
|
899
|
+
|
|
900
|
+
# [start, length] of the vector at +pos+ with +size+-byte elements
|
|
901
|
+
def vector(pos, size)
|
|
902
|
+
len = u32(pos)
|
|
903
|
+
check(pos + 4, len * size)
|
|
904
|
+
[pos + 4, len]
|
|
905
|
+
end
|
|
906
|
+
end
|
|
907
|
+
|
|
908
|
+
# One flatbuffer table; fields are addressed by their slot (declaration order in the .fbs)
|
|
909
|
+
class Table
|
|
910
|
+
def initialize(fb, pos)
|
|
911
|
+
@fb = fb
|
|
912
|
+
@pos = pos
|
|
913
|
+
@vtable = pos - fb.i32(pos)
|
|
914
|
+
@vtable_size = fb.u16(@vtable)
|
|
915
|
+
raise Error, "bad flatbuffer vtable at #{@vtable}" if @vtable_size < 4 || @vtable_size.odd?
|
|
916
|
+
fb.check(@vtable, @vtable_size)
|
|
917
|
+
end
|
|
918
|
+
|
|
919
|
+
# Absolute position of a field's value, nil when absent
|
|
920
|
+
def field(slot)
|
|
921
|
+
o = 4 + (slot * 2)
|
|
922
|
+
return nil if o + 2 > @vtable_size
|
|
923
|
+
off = @fb.u16(@vtable + o)
|
|
924
|
+
off.zero? ? nil : @pos + off
|
|
925
|
+
end
|
|
926
|
+
|
|
927
|
+
def u8(slot, default = 0) = (p = field(slot)) ? @fb.u8(p) : default
|
|
928
|
+
def bool(slot, default = false) = (p = field(slot)) ? @fb.u8(p) != 0 : default
|
|
929
|
+
def i16(slot, default = 0) = (p = field(slot)) ? @fb.i16(p) : default
|
|
930
|
+
def i32(slot, default = 0) = (p = field(slot)) ? @fb.i32(p) : default
|
|
931
|
+
def i64(slot, default = 0) = (p = field(slot)) ? @fb.i64(p) : default
|
|
932
|
+
def string(slot) = (p = field(slot)) && @fb.string(@fb.deref(p))
|
|
933
|
+
def table(slot) = (p = field(slot)) && @fb.table_at(@fb.deref(p))
|
|
934
|
+
|
|
935
|
+
def tables(slot)
|
|
936
|
+
p = field(slot) or return []
|
|
937
|
+
start, len = @fb.vector(@fb.deref(p), 4)
|
|
938
|
+
Array.new(len) { |i| @fb.table_at(@fb.deref(start + (4 * i))) }
|
|
939
|
+
end
|
|
940
|
+
|
|
941
|
+
def i32s(slot)
|
|
942
|
+
p = field(slot) or return []
|
|
943
|
+
start, len = @fb.vector(@fb.deref(p), 4)
|
|
944
|
+
Array.new(len) { |i| @fb.i32(start + (4 * i)) }
|
|
945
|
+
end
|
|
946
|
+
end
|
|
947
|
+
|
|
948
|
+
module_function
|
|
949
|
+
|
|
950
|
+
# Decodes the base64 ARROW:schema value; raises ArrowSchema::Error when it can't
|
|
951
|
+
def decode(b64)
|
|
952
|
+
bytes = b64.to_s.unpack1("m")
|
|
953
|
+
raise Error, "empty value" if bytes.empty?
|
|
954
|
+
fb = FlatBuffer.new(message_bytes(bytes))
|
|
955
|
+
message = fb.root
|
|
956
|
+
header_type = message.u8(1)
|
|
957
|
+
raise Error, "IPC message holds a #{header_type} header, not a Schema" unless header_type == MESSAGE_SCHEMA
|
|
958
|
+
schema = message.table(2) or raise Error, "IPC message has no Schema"
|
|
959
|
+
count = [0]
|
|
960
|
+
{
|
|
961
|
+
endianness: schema.i16(0).zero? ? "little" : "big",
|
|
962
|
+
fields: schema.tables(1).map { |f| field(f, 0, count) },
|
|
963
|
+
metadata: key_values(schema.tables(2))
|
|
964
|
+
}.compact
|
|
965
|
+
rescue ArgumentError, TypeError, RangeError => e
|
|
966
|
+
raise Error, e.message
|
|
967
|
+
end
|
|
968
|
+
|
|
969
|
+
# The flatbuffer inside an encapsulated IPC message: [0xFFFFFFFF] int32 length, flatbuffer
|
|
970
|
+
# (the continuation marker is missing in files from before Arrow 0.15)
|
|
971
|
+
def message_bytes(bytes)
|
|
972
|
+
raise Error, "too short for an IPC message (#{bytes.bytesize} bytes)" if bytes.bytesize < 8
|
|
973
|
+
len = bytes.unpack1("l<")
|
|
974
|
+
start = 4
|
|
975
|
+
if len == -1
|
|
976
|
+
len = bytes.byteslice(4, 4).unpack1("l<")
|
|
977
|
+
start = 8
|
|
978
|
+
end
|
|
979
|
+
raise Error, "IPC message length #{len} does not fit in #{bytes.bytesize} bytes" if len <= 0 || start + len > bytes.bytesize
|
|
980
|
+
bytes.byteslice(start, len)
|
|
981
|
+
end
|
|
982
|
+
|
|
983
|
+
def field(t, depth, count)
|
|
984
|
+
raise Error, "fields nested deeper than #{MAX_DEPTH} levels" if depth > MAX_DEPTH
|
|
985
|
+
raise Error, "more than #{MAX_FIELDS} fields" if (count[0] += 1) > MAX_FIELDS
|
|
986
|
+
children = t.tables(5).map { |c| field(c, depth + 1, count) }
|
|
987
|
+
metadata = key_values(t.tables(6))
|
|
988
|
+
type = type_name(t.u8(2), t.table(3), children)
|
|
989
|
+
h = { name: t.string(0).to_s, type: type, nullable: t.bool(1) }
|
|
990
|
+
if (d = t.table(4))
|
|
991
|
+
index = d.table(1)
|
|
992
|
+
index_type = index ? int_name(index) : "int32"
|
|
993
|
+
ordered = d.bool(2)
|
|
994
|
+
h[:dictionary] = { index_type: index_type, ordered: ordered, id: d.i64(0) }
|
|
995
|
+
h[:type] = "dictionary<values=#{type}, indices=#{index_type}, ordered=#{ordered ? 1 : 0}>"
|
|
996
|
+
end
|
|
997
|
+
h[:children] = children unless children.empty?
|
|
998
|
+
if metadata
|
|
999
|
+
h[:extension] = metadata["ARROW:extension:name"] if metadata["ARROW:extension:name"]
|
|
1000
|
+
h[:metadata] = metadata
|
|
1001
|
+
end
|
|
1002
|
+
h
|
|
1003
|
+
end
|
|
1004
|
+
|
|
1005
|
+
def key_values(tables)
|
|
1006
|
+
return nil if tables.empty?
|
|
1007
|
+
tables.to_h { |kv| [kv.string(0).to_s, kv.string(1).to_s] }
|
|
1008
|
+
end
|
|
1009
|
+
|
|
1010
|
+
# A child as pyarrow prints it inside a nested type: "name: type" plus " not null"
|
|
1011
|
+
def child_text(c) = "#{c[:name]}: #{c[:type]}#{c[:nullable] ? "" : " not null"}"
|
|
1012
|
+
|
|
1013
|
+
def int_name(t) = "#{t.bool(1) ? "" : "u"}int#{t.i32(0)}"
|
|
1014
|
+
|
|
1015
|
+
def unit(u) = TIME_UNITS[u] || "unit#{u}"
|
|
1016
|
+
|
|
1017
|
+
# Type names follow Arrow's DataType::ToString (what pyarrow prints)
|
|
1018
|
+
def type_name(kind, t, children)
|
|
1019
|
+
case kind
|
|
1020
|
+
when 1 then "null"
|
|
1021
|
+
when 2 then t ? int_name(t) : "int"
|
|
1022
|
+
when 3 then %w[halffloat float double][t ? t.i16(0) : 0] || "float?"
|
|
1023
|
+
when 4 then "binary"
|
|
1024
|
+
when 5 then "string"
|
|
1025
|
+
when 6 then "bool"
|
|
1026
|
+
when 7
|
|
1027
|
+
raise Error, "decimal type without parameters" unless t
|
|
1028
|
+
bits = t.i32(2, 128)
|
|
1029
|
+
"decimal#{bits}(#{t.i32(0)}, #{t.i32(1)})"
|
|
1030
|
+
when 8 then t&.i16(0, 1)&.zero? ? "date32[day]" : "date64[ms]"
|
|
1031
|
+
when 9
|
|
1032
|
+
bits = t ? t.i32(1, 32) : 32
|
|
1033
|
+
"time#{bits}[#{unit(t ? t.i16(0, 1) : 1)}]"
|
|
1034
|
+
when 10
|
|
1035
|
+
tz = t&.string(1)
|
|
1036
|
+
"timestamp[#{unit(t ? t.i16(0) : 0)}#{tz ? ", tz=#{tz}" : ""}]"
|
|
1037
|
+
when 11 then %w[month_interval day_time_interval month_day_nano_interval][t ? t.i16(0) : 0] || "interval"
|
|
1038
|
+
when 12 then "list<#{children.map { |c| child_text(c) }.join(", ")}>"
|
|
1039
|
+
when 13 then "struct<#{children.map { |c| child_text(c) }.join(", ")}>"
|
|
1040
|
+
when 14
|
|
1041
|
+
mode = t&.i16(0)&.positive? ? "dense" : "sparse"
|
|
1042
|
+
ids = t ? t.i32s(1) : []
|
|
1043
|
+
members = children.each_with_index.map { |c, i| "#{child_text(c)}=#{ids[i] || i}" }
|
|
1044
|
+
"#{mode}_union<#{members.join(", ")}>"
|
|
1045
|
+
when 15 then "fixed_size_binary[#{t ? t.i32(0) : 0}]"
|
|
1046
|
+
when 16 then "fixed_size_list<#{children.map { |c| child_text(c) }.join(", ")}>[#{t ? t.i32(0) : 0}]"
|
|
1047
|
+
when 17 then map_name(t, children)
|
|
1048
|
+
when 18 then "duration[#{unit(t ? t.i16(0, 1) : 1)}]"
|
|
1049
|
+
when 19 then "large_binary"
|
|
1050
|
+
when 20 then "large_string"
|
|
1051
|
+
when 21 then "large_list<#{children.map { |c| child_text(c) }.join(", ")}>"
|
|
1052
|
+
when 22 then "run_end_encoded<#{children.map { |c| "#{c[:name] == "values" ? "values" : "run_ends"}: #{c[:type]}" }.join(", ")}>"
|
|
1053
|
+
when 23 then "binary_view"
|
|
1054
|
+
when 24 then "string_view"
|
|
1055
|
+
when 25 then "list_view<#{children.map { |c| child_text(c) }.join(", ")}>"
|
|
1056
|
+
when 26 then "large_list_view<#{children.map { |c| child_text(c) }.join(", ")}>"
|
|
1057
|
+
else "unknown type #{kind}"
|
|
1058
|
+
end
|
|
1059
|
+
end
|
|
1060
|
+
|
|
1061
|
+
# map<key, value> with non-standard field names in parentheses, as Arrow prints it
|
|
1062
|
+
def map_name(t, children)
|
|
1063
|
+
entries = children.first
|
|
1064
|
+
kv = entries && entries[:children] || []
|
|
1065
|
+
named = ->(f, std) { f ? "#{f[:type]}#{f[:name] == std ? "" : " ('#{f[:name]}')"}" : "?" }
|
|
1066
|
+
sorted = t&.bool(0) ? ", keys_sorted" : ""
|
|
1067
|
+
entries_name = entries && entries[:name] != "entries" ? " ('#{entries[:name]}')" : ""
|
|
1068
|
+
"map<#{named.call(kv[0], "key")}, #{named.call(kv[1], "value")}#{sorted}#{entries_name}>"
|
|
1069
|
+
end
|
|
1070
|
+
|
|
1071
|
+
# "name: type" lines for a field and its children, indented, for text output
|
|
1072
|
+
def lines(fields, depth = 0, out = [])
|
|
1073
|
+
fields.each do |f|
|
|
1074
|
+
notes = []
|
|
1075
|
+
notes << "not null" unless f[:nullable]
|
|
1076
|
+
notes << "extension #{f[:extension]}" if f[:extension]
|
|
1077
|
+
meta = (f[:metadata] || {}).reject { |k, _| k.start_with?("ARROW:extension:") }
|
|
1078
|
+
notes << "metadata #{meta.map { |k, v| "#{k}=#{v.to_s[0, 60].inspect}" }.join(", ")}" unless meta.empty?
|
|
1079
|
+
out << "#{" " * depth}#{f[:name]}: #{f[:type]}#{notes.empty? ? "" : " (#{notes.join("; ")})"}"
|
|
1080
|
+
lines(f[:children], depth + 1, out) if f[:children] && depth < 8
|
|
1081
|
+
end
|
|
1082
|
+
out
|
|
1083
|
+
end
|
|
1084
|
+
end
|
|
1085
|
+
|
|
1086
|
+
# ---- class helpers ----
|
|
1087
|
+
|
|
1088
|
+
def self.encoding_name(e) = Format::Encoding::NAMES[e]&.to_s || e.to_s
|
|
1089
|
+
|
|
1090
|
+
def self.type_name(column)
|
|
1091
|
+
node = column.node
|
|
1092
|
+
phys = T::NAMES[node.type].to_s
|
|
1093
|
+
phys += "(#{node.type_length})" if node.type == T::FIXED_LEN_BYTE_ARRAY && node.type_length
|
|
1094
|
+
logical = logical_type_name(node)
|
|
1095
|
+
logical ? "#{phys} #{logical}" : phys
|
|
1096
|
+
end
|
|
1097
|
+
|
|
1098
|
+
def self.logical_type_name(node)
|
|
1099
|
+
if (kind = node.logical_type&.kind)
|
|
1100
|
+
name, payload = kind
|
|
1101
|
+
case name
|
|
1102
|
+
when :integer then "INTEGER(#{payload.bit_width}, #{payload.is_signed ? "signed" : "unsigned"})"
|
|
1103
|
+
when :decimal then "DECIMAL(#{payload.precision}, #{payload.scale})"
|
|
1104
|
+
when :timestamp, :time
|
|
1105
|
+
"#{name.upcase}(#{payload.unit&.to_sym&.upcase}, #{payload.is_adjusted_to_utc ? "UTC" : "local"})"
|
|
1106
|
+
when :unknown then "NULL"
|
|
1107
|
+
else name.to_s.upcase
|
|
1108
|
+
end
|
|
1109
|
+
elsif node.converted_type
|
|
1110
|
+
c = Format::ConvertedType::NAMES[node.converted_type].to_s
|
|
1111
|
+
c == "DECIMAL" ? "DECIMAL(#{node.precision}, #{node.scale || 0})" : c
|
|
1112
|
+
end
|
|
1113
|
+
end
|
|
1114
|
+
|
|
1115
|
+
# :signed, :unsigned or :unknown, per the Parquet sort order rules for the column's type
|
|
1116
|
+
def self.sort_order(column)
|
|
1117
|
+
kind, _a, signed = Types.logical_of(column.node)
|
|
1118
|
+
case kind
|
|
1119
|
+
when :integer then signed == false ? :unsigned : :signed
|
|
1120
|
+
when :decimal, :date, :time, :timestamp, :float16 then :signed
|
|
1121
|
+
when :string, :enum, :json, :bson, :uuid then :unsigned
|
|
1122
|
+
else
|
|
1123
|
+
case column.type
|
|
1124
|
+
when T::BYTE_ARRAY, T::FIXED_LEN_BYTE_ARRAY then :unsigned
|
|
1125
|
+
when T::INT96 then :unknown
|
|
1126
|
+
else :signed
|
|
1127
|
+
end
|
|
1128
|
+
end
|
|
1129
|
+
end
|
|
1130
|
+
|
|
1131
|
+
# Converts Ruby values (Time, BigDecimal, binary Strings, non-finite Floats...) to JSON-safe ones
|
|
1132
|
+
def self.jsonable(v)
|
|
1133
|
+
case v
|
|
1134
|
+
when Hash then v.each_with_object({}) { |(k, x), h| h[k.is_a?(Symbol) ? k : k.to_s] = jsonable(x) }
|
|
1135
|
+
when Array then v.map { |x| jsonable(x) }
|
|
1136
|
+
when Struct then v.respond_to?(:to_h) ? jsonable(v.to_h) : v.to_s
|
|
1137
|
+
when String then text(v)
|
|
1138
|
+
when Symbol then v.to_s
|
|
1139
|
+
when Float then v.finite? ? v : v.to_s
|
|
1140
|
+
when Time then v.utc.strftime(v.nsec.zero? ? "%Y-%m-%dT%H:%M:%SZ" : "%Y-%m-%dT%H:%M:%S.%NZ")
|
|
1141
|
+
when Date then v.iso8601
|
|
1142
|
+
when BigDecimal then v.to_s("F")
|
|
1143
|
+
when Integer, true, false, nil then v
|
|
1144
|
+
else v.to_s
|
|
1145
|
+
end
|
|
1146
|
+
end
|
|
1147
|
+
|
|
1148
|
+
# A String as readable text when it is valid UTF-8 without control characters, else as hex
|
|
1149
|
+
def self.text(s)
|
|
1150
|
+
return s if s.encoding == Encoding::UTF_8 && s.valid_encoding?
|
|
1151
|
+
u = s.dup.force_encoding(Encoding::UTF_8)
|
|
1152
|
+
return u if u.valid_encoding? && !u.match?(/[\x00-\x08\x0e-\x1f\x7f]/)
|
|
1153
|
+
hex(s)
|
|
1154
|
+
end
|
|
1155
|
+
|
|
1156
|
+
def self.hex(bytes)
|
|
1157
|
+
b = bytes.b
|
|
1158
|
+
b.bytesize > 64 ? "0x#{b.byteslice(0, 64).unpack1("H*")}… (#{b.bytesize} bytes)" : "0x#{b.unpack1("H*")}"
|
|
1159
|
+
end
|
|
1160
|
+
|
|
1161
|
+
# A short display form of a decoded value
|
|
1162
|
+
def self.display(v, max: 40)
|
|
1163
|
+
s = case v
|
|
1164
|
+
when String then v.encoding == Encoding::BINARY ? text(v) : v
|
|
1165
|
+
when nil then "null"
|
|
1166
|
+
else jsonable(v).to_s
|
|
1167
|
+
end
|
|
1168
|
+
s = s.inspect if v.is_a?(String)
|
|
1169
|
+
s.size > max ? "#{s[0, max - 1]}…" : s
|
|
1170
|
+
end
|
|
1171
|
+
|
|
1172
|
+
def self.human_bytes(n)
|
|
1173
|
+
return "?" unless n
|
|
1174
|
+
units = %w[B KB MB GB TB]
|
|
1175
|
+
f = n.to_f
|
|
1176
|
+
i = 0
|
|
1177
|
+
while f >= 1024 && i < units.size - 1
|
|
1178
|
+
f /= 1024
|
|
1179
|
+
i += 1
|
|
1180
|
+
end
|
|
1181
|
+
i.zero? ? "#{n} B" : format("%.#{f < 10 ? 2 : 1}f %s", f, units[i])
|
|
1182
|
+
end
|
|
1183
|
+
|
|
1184
|
+
private
|
|
1185
|
+
|
|
1186
|
+
def ratio_text(uncompressed, compressed)
|
|
1187
|
+
compressed.to_i.positive? && uncompressed ? format(" (%.2fx)", uncompressed.to_f / compressed) : ""
|
|
1188
|
+
end
|
|
1189
|
+
|
|
1190
|
+
def crc_text(page)
|
|
1191
|
+
case page.checksum
|
|
1192
|
+
when :ok then " crc ok"
|
|
1193
|
+
when :mismatch then " CRC MISMATCH"
|
|
1194
|
+
else page.crc ? " crc" : ""
|
|
1195
|
+
end
|
|
1196
|
+
end
|
|
1197
|
+
|
|
1198
|
+
def index_mismatch_text(m)
|
|
1199
|
+
where = "row group #{m[:row_group]} #{m[:column]}"
|
|
1200
|
+
if m[:field] == :page_count
|
|
1201
|
+
"#{where}: #{m[:page_value]} data pages but #{m[:index_value]} column index entries"
|
|
1202
|
+
else
|
|
1203
|
+
"#{where} page #{m[:page]}: #{m[:field]} in page header #{Inspector.display(m[:page_value])}, " \
|
|
1204
|
+
"in column index #{Inspector.display(m[:index_value])}"
|
|
1205
|
+
end
|
|
1206
|
+
end
|
|
1207
|
+
|
|
1208
|
+
# Adds :arrow_type to schema nodes with a same-named Arrow field (top level, and struct members)
|
|
1209
|
+
def annotate_arrow_types(nodes, fields, depth = 0)
|
|
1210
|
+
by_name = fields.to_h { |f| [f[:name], f] }
|
|
1211
|
+
nodes.each do |n|
|
|
1212
|
+
f = by_name[n[:name]] or next
|
|
1213
|
+
n[:arrow_type] = f[:type]
|
|
1214
|
+
if n[:children] && f[:children] && f[:type].start_with?("struct<") && depth < 32
|
|
1215
|
+
annotate_arrow_types(n[:children], f[:children], depth + 1)
|
|
1216
|
+
end
|
|
1217
|
+
end
|
|
1218
|
+
end
|
|
1219
|
+
|
|
1220
|
+
# Whether the index bound +idx+ excludes values the page's bound +page+ says are present
|
|
1221
|
+
# (index min above the page min, or index max below the page max). Truncated binary bounds
|
|
1222
|
+
# (one a prefix of the other) and values that can't be compared are never reported.
|
|
1223
|
+
def narrower?(idx, page, order, which)
|
|
1224
|
+
return false if idx.nil? || page.nil? || order == :unknown
|
|
1225
|
+
a, b = which == :min ? [page, idx] : [idx, page] # true when a < b
|
|
1226
|
+
if a.is_a?(String) && b.is_a?(String)
|
|
1227
|
+
a = a.b
|
|
1228
|
+
b = b.b
|
|
1229
|
+
return false if a.start_with?(b) || b.start_with?(a)
|
|
1230
|
+
return order == :unsigned ? a < b : false
|
|
1231
|
+
end
|
|
1232
|
+
return false if a.is_a?(Float) && a.nan? || b.is_a?(Float) && b.nan?
|
|
1233
|
+
a = a ? 1 : 0 if a == true || a == false
|
|
1234
|
+
b = b ? 1 : 0 if b == true || b == false
|
|
1235
|
+
(a <=> b) == -1
|
|
1236
|
+
rescue StandardError
|
|
1237
|
+
false
|
|
1238
|
+
end
|
|
1239
|
+
|
|
1240
|
+
def safe_extreme(values, which)
|
|
1241
|
+
vals = values.compact
|
|
1242
|
+
return nil if vals.empty?
|
|
1243
|
+
if vals.all? { |v| v == true || v == false }
|
|
1244
|
+
return which == :min ? vals.all? : vals.any?
|
|
1245
|
+
end
|
|
1246
|
+
return nil unless vals.map(&:class).uniq.size == 1
|
|
1247
|
+
which == :min ? vals.min : vals.max
|
|
1248
|
+
rescue ArgumentError, NoMethodError
|
|
1249
|
+
nil
|
|
1250
|
+
end
|
|
1251
|
+
|
|
1252
|
+
def read_footer
|
|
1253
|
+
@io.seek(0, IO::SEEK_END)
|
|
1254
|
+
@file_size = @io.pos
|
|
1255
|
+
raise FormatError, "File too small to be Parquet (#{@file_size} bytes)" if @file_size < 12
|
|
1256
|
+
tail = read_at(@file_size - 8, 8)
|
|
1257
|
+
raise UnsupportedError, "Encrypted Parquet files are not supported" if tail.byteslice(4, 4) == "PARE"
|
|
1258
|
+
raise FormatError, "Missing PAR1 footer magic" unless tail.byteslice(4, 4) == MAGIC
|
|
1259
|
+
@footer_size = tail.unpack1("V")
|
|
1260
|
+
raise FormatError, "Footer length #{@footer_size} exceeds file size" if @footer_size + 12 > @file_size
|
|
1261
|
+
@metadata = Format::FileMetaData.decode(read_at(footer_offset, @footer_size)).first
|
|
1262
|
+
rescue Thrift::Error => e
|
|
1263
|
+
raise FormatError, "Corrupt file metadata: #{e.message}"
|
|
1264
|
+
end
|
|
1265
|
+
|
|
1266
|
+
# Reads +len+ bytes at +pos+ through a small read-ahead window, so walking many small pages
|
|
1267
|
+
# does not cost a syscall per header
|
|
1268
|
+
def read_at(pos, len)
|
|
1269
|
+
if @window && pos >= @window_pos && pos + len <= @window_pos + @window.bytesize
|
|
1270
|
+
return @window.byteslice(pos - @window_pos, len)
|
|
1271
|
+
end
|
|
1272
|
+
@io.seek(pos)
|
|
1273
|
+
if len >= WINDOW
|
|
1274
|
+
(@io.read(len) || "".b).b
|
|
1275
|
+
else
|
|
1276
|
+
@window_pos = pos
|
|
1277
|
+
@window = (@io.read(WINDOW) || "".b).b
|
|
1278
|
+
@window.byteslice(0, len)
|
|
1279
|
+
end
|
|
1280
|
+
end
|
|
1281
|
+
|
|
1282
|
+
# Decodes the page header at +pos+, reading more bytes when it is larger than the first guess
|
|
1283
|
+
# (page statistics of long strings can make headers big)
|
|
1284
|
+
def read_page_header(pos, limit)
|
|
1285
|
+
want = 256
|
|
1286
|
+
while true
|
|
1287
|
+
avail = [want, limit - pos].min
|
|
1288
|
+
raise FormatError, "no room for a page header" if avail <= 0
|
|
1289
|
+
buf = read_at(pos, avail)
|
|
1290
|
+
begin
|
|
1291
|
+
header, size = Format::PageHeader.decode(buf)
|
|
1292
|
+
return [header, size]
|
|
1293
|
+
rescue Thrift::Error
|
|
1294
|
+
raise if avail < want || want >= 16 * 1024 * 1024
|
|
1295
|
+
want *= 8
|
|
1296
|
+
end
|
|
1297
|
+
end
|
|
1298
|
+
end
|
|
1299
|
+
|
|
1300
|
+
def page_info(index, h, pos, header_size, column)
|
|
1301
|
+
info = PageInfo.new(
|
|
1302
|
+
index: index,
|
|
1303
|
+
type: Format::PageType::NAMES[h.type] || h.type,
|
|
1304
|
+
offset: pos,
|
|
1305
|
+
header_size: header_size,
|
|
1306
|
+
compressed_size: h.compressed_page_size.to_i,
|
|
1307
|
+
uncompressed_size: h.uncompressed_page_size.to_i,
|
|
1308
|
+
crc: h.crc
|
|
1309
|
+
)
|
|
1310
|
+
raise FormatError, "negative page size #{info.compressed_size}" if info.compressed_size.negative?
|
|
1311
|
+
if (d = h.data_page_header)
|
|
1312
|
+
info.num_values = d.num_values
|
|
1313
|
+
info.encoding = Inspector.encoding_name(d.encoding)
|
|
1314
|
+
info.definition_level_encoding = Inspector.encoding_name(d.definition_level_encoding) if d.definition_level_encoding
|
|
1315
|
+
info.repetition_level_encoding = Inspector.encoding_name(d.repetition_level_encoding) if d.repetition_level_encoding
|
|
1316
|
+
info.statistics = decode_statistics(d.statistics, column)
|
|
1317
|
+
info.num_nulls = info.statistics&.null_count
|
|
1318
|
+
# Without repetition every value is a row
|
|
1319
|
+
info.num_rows = d.num_values if column.max_repetition_level.zero?
|
|
1320
|
+
elsif (d = h.data_page_header_v2)
|
|
1321
|
+
info.num_values = d.num_values
|
|
1322
|
+
info.num_nulls = d.num_nulls
|
|
1323
|
+
info.num_rows = d.num_rows
|
|
1324
|
+
info.encoding = Inspector.encoding_name(d.encoding)
|
|
1325
|
+
info.definition_levels_byte_length = d.definition_levels_byte_length
|
|
1326
|
+
info.repetition_levels_byte_length = d.repetition_levels_byte_length
|
|
1327
|
+
info.is_compressed = d.is_compressed.nil? ? true : d.is_compressed
|
|
1328
|
+
info.statistics = decode_statistics(d.statistics, column)
|
|
1329
|
+
elsif (d = h.dictionary_page_header)
|
|
1330
|
+
info.num_values = d.num_values
|
|
1331
|
+
info.encoding = Inspector.encoding_name(d.encoding)
|
|
1332
|
+
info.is_sorted = d.is_sorted
|
|
1333
|
+
end
|
|
1334
|
+
info
|
|
1335
|
+
end
|
|
1336
|
+
|
|
1337
|
+
def describe_key_value(key, value)
|
|
1338
|
+
value = value.to_s
|
|
1339
|
+
h = { key: key, bytesize: value.bytesize }
|
|
1340
|
+
if key == "ARROW:schema"
|
|
1341
|
+
h[:format] = "arrow_schema"
|
|
1342
|
+
h[:value] = value.size > 120 ? "#{value[0, 120]}…" : value
|
|
1343
|
+
if (arrow = arrow_schema)
|
|
1344
|
+
h[:summary] = "Arrow schema, #{arrow[:fields].size} field#{arrow[:fields].size == 1 ? "" : "s"}"
|
|
1345
|
+
h[:arrow_fields] = arrow[:fields].map { |f| f[:name] }
|
|
1346
|
+
meta = arrow[:metadata]&.transform_values { |v| v.size > 4000 ? "#{v[0, 4000]}…" : v }
|
|
1347
|
+
h[:arrow_schema] = arrow.merge(metadata: meta).compact
|
|
1348
|
+
else
|
|
1349
|
+
h[:summary] = "Arrow IPC schema message, base64-encoded (#{value.bytesize} bytes)"
|
|
1350
|
+
h[:arrow_error] = arrow_schema_error
|
|
1351
|
+
names = arrow_field_names(value)
|
|
1352
|
+
h[:arrow_fields] = names if names
|
|
1353
|
+
end
|
|
1354
|
+
elsif value.lstrip.start_with?("{", "[") && value.bytesize < 4 * 1024 * 1024
|
|
1355
|
+
begin
|
|
1356
|
+
h[:json] = JSON.parse(value)
|
|
1357
|
+
h[:format] = "json"
|
|
1358
|
+
h[:summary] = key == "pandas" ? pandas_summary(h[:json]) : "JSON"
|
|
1359
|
+
rescue JSON::ParserError
|
|
1360
|
+
h[:format] = "text"
|
|
1361
|
+
end
|
|
1362
|
+
h[:value] = value.size > 4000 ? "#{value[0, 4000]}…" : value
|
|
1363
|
+
else
|
|
1364
|
+
t = Inspector.text(value.b)
|
|
1365
|
+
h[:format] = t.start_with?("0x") && value.bytesize.positive? ? "binary" : "text"
|
|
1366
|
+
h[:value] = t.size > 4000 ? "#{t[0, 4000]}…" : t
|
|
1367
|
+
end
|
|
1368
|
+
h
|
|
1369
|
+
end
|
|
1370
|
+
|
|
1371
|
+
def pandas_summary(json)
|
|
1372
|
+
return "pandas metadata" unless json.is_a?(Hash)
|
|
1373
|
+
cols = json["columns"]&.size
|
|
1374
|
+
"pandas #{json["pandas_version"]} metadata, #{cols || "?"} columns"
|
|
1375
|
+
end
|
|
1376
|
+
|
|
1377
|
+
# Field names from an Arrow IPC schema message, found by scanning for its flatbuffer strings.
|
|
1378
|
+
# Best effort: only used to label the blob, nil when nothing sensible is found.
|
|
1379
|
+
def arrow_field_names(b64)
|
|
1380
|
+
bytes = b64.unpack1("m")
|
|
1381
|
+
schema_cols = columns.map { |c| c.path.first }.uniq
|
|
1382
|
+
found = schema_cols.select { |n| bytes.include?(n.b) }
|
|
1383
|
+
found.empty? ? nil : found
|
|
1384
|
+
rescue ArgumentError
|
|
1385
|
+
nil
|
|
1386
|
+
end
|
|
1387
|
+
end
|
|
1388
|
+
end
|