herringbone 0.1.0 → 0.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,1784 @@
1
+ # frozen_string_literal: true
2
+
3
+ require "json"
4
+ require "zlib"
5
+
6
+ module Herringbone
7
+ # Examines a Parquet file using only its footer, page headers and page indexes. Values are
8
+ # never decompressed or decoded, so this works for files whose codecs are not installed and
9
+ # stays fast for big files (it seeks from page header to page header).
10
+ #
11
+ # File.open("data.parquet", "rb") do |io|
12
+ # inspector = Herringbone::Inspector.new(io)
13
+ # inspector.summary # => { file_size:, num_rows:, codecs:, ... }
14
+ # inspector.row_groups[0].column("name").pages
15
+ # inspector.to_h # everything, JSON-serializable
16
+ # puts inspector.report # readable text (bin/herringbone inspect)
17
+ # html = inspector.to_html # self-contained HTML page (see Visualizer)
18
+ # end
19
+ #
20
+ # Page headers and indexes are read lazily; #load_all reads them all up front, after which the
21
+ # inspector no longer needs the IO. The IO is never closed. After #verify_checksums, page CRC
22
+ # results are included in every output.
23
+ class Inspector
24
+ # Shorthand for the Parquet physical type constants
25
+ T = Format::Type
26
+ # Magic bytes at the start and end of every (unencrypted) Parquet file
27
+ MAGIC = "PAR1"
28
+ # Size of the read-ahead window used by #read_at, in bytes
29
+ WINDOW = 64 * 1024
30
+
31
+ # Just enough of parquet.thrift's BloomFilterHeader to learn the filter's size
32
+ class BloomFilterHeader < Thrift::Struct
33
+ field 1, :num_bytes, :i32
34
+ end
35
+
36
+ # Decoded min/max statistics (from a column chunk, a page header or a page index entry).
37
+ # +min+/+max+ are converted with the column's type converter (hex Strings when they could
38
+ # not be decoded); +min_exact+/+max_exact+ mirror is_min_value_exact/is_max_value_exact;
39
+ # +source+ says which Thrift fields they came from and +caveat+ why they may be unreliable.
40
+ Stats = Struct.new(:min, :max, :null_count, :distinct_count, :min_exact, :max_exact, :source, :caveat,
41
+ keyword_init: true) do
42
+ # @return [Hash{Symbol => Object}] the set members, JSON-safe (see Inspector.jsonable)
43
+ def to_h = Inspector.jsonable(super.compact)
44
+ end
45
+
46
+ # One page header of a column chunk. +checksum+ is nil until the CRCs are verified
47
+ # (Inspector#verify_checksums), then :ok, :mismatch or :absent (the page has no CRC);
48
+ # +actual_crc+ is then the CRC32 of the page's stored bytes (nil when it has no CRC).
49
+ # +index+ is the page's position in the chunk, +offset+ where its header starts; +type+ is
50
+ # a PageType name such as :DATA_PAGE (the raw Integer when unknown). +first_row_index+ and,
51
+ # for v1 pages of repeated columns, +num_rows+ are only known from the OffsetIndex.
52
+ PageInfo = Struct.new(:index, :type, :offset, :header_size, :compressed_size, :uncompressed_size,
53
+ :num_values, :num_nulls, :num_rows, :first_row_index, :encoding, :definition_level_encoding,
54
+ :repetition_level_encoding, :definition_levels_byte_length, :repetition_levels_byte_length,
55
+ :is_compressed, :is_sorted, :statistics, :crc, :checksum, :actual_crc, keyword_init: true) do
56
+ # @return [Integer] bytes taken by the page in the file: header plus compressed body
57
+ def total_size = header_size + compressed_size
58
+
59
+ # @return [Integer] file offset just past the page's body
60
+ def end_offset = offset + total_size
61
+
62
+ # @return [Integer] file offset where the page's body (after the header) starts
63
+ def body_offset = offset + header_size
64
+
65
+ # @return [Boolean] whether this is a dictionary page
66
+ def dictionary? = type == :DICTIONARY_PAGE
67
+
68
+ # @return [Boolean] whether this is a v1 or v2 data page
69
+ def data? = type == :DATA_PAGE || type == :DATA_PAGE_V2
70
+
71
+ # The CRC from the header as an unsigned 32-bit value (Thrift stores it as a signed i32)
72
+ # @return [Integer, nil] nil when the header has no CRC
73
+ def expected_crc = crc && (crc & 0xFFFF_FFFF)
74
+
75
+ # @return [Hash{Symbol => Object}] the set members, JSON-safe; +:crc+ becomes a Boolean
76
+ # saying whether the header has a CRC, +:actual_crc+ is left out
77
+ def to_h
78
+ h = super
79
+ h.delete(:actual_crc)
80
+ h[:statistics] = statistics&.to_h
81
+ h[:crc] = !crc.nil?
82
+ Inspector.jsonable(h.compact)
83
+ end
84
+ end
85
+
86
+ # ColumnIndex of one column chunk, with min/max decoded per page (nil for all-null pages)
87
+ ColumnIndexInfo = Struct.new(:offset, :length, :null_pages, :min_values, :max_values, :boundary_order,
88
+ :null_counts, :repetition_level_histograms, :definition_level_histograms, keyword_init: true) do
89
+ # @return [Hash{Symbol => Object}] the set members, JSON-safe (see Inspector.jsonable)
90
+ def to_h = Inspector.jsonable(super.compact)
91
+ end
92
+
93
+ # One OffsetIndex entry: where a data page starts, its size (header included) and the index
94
+ # of its first row within the row group
95
+ PageLocation = Struct.new(:offset, :compressed_page_size, :first_row_index, keyword_init: true)
96
+
97
+ # OffsetIndex of one column chunk: its own +offset+/+length+ in the file, the PageLocations
98
+ # of its data pages and the optional per-page unencoded BYTE_ARRAY sizes
99
+ OffsetIndexInfo = Struct.new(:offset, :length, :page_locations, :unencoded_byte_array_data_bytes,
100
+ keyword_init: true) do
101
+ # @return [Hash{Symbol => Object}] the set members, with +:page_locations+ as Hashes
102
+ def to_h
103
+ h = super
104
+ h[:page_locations] = page_locations.map(&:to_h)
105
+ h.compact
106
+ end
107
+ end
108
+
109
+ # One column chunk of a row group
110
+ class ColumnChunkInfo
111
+ attr_reader :inspector, :row_group, :column, :chunk, :meta, :error
112
+
113
+ # @param inspector [Inspector] owner, used to read page headers and indexes lazily
114
+ # @param row_group [RowGroupInfo] row group the chunk belongs to
115
+ # @param column [Schema::Column] leaf column the chunk stores
116
+ # @param chunk [Format::ColumnChunk] chunk as decoded from the footer
117
+ def initialize(inspector, row_group, column, chunk)
118
+ @inspector = inspector
119
+ @row_group = row_group
120
+ @column = column
121
+ @chunk = chunk
122
+ @meta = chunk.meta_data
123
+ end
124
+
125
+ # @return [String] dotted path of the column, e.g. +"address.city"+
126
+ def path = @column.dotted_path
127
+
128
+ # @return [Integer] position of the column among the schema's leaf columns
129
+ def column_index_number = @column.index
130
+
131
+ # @return [Symbol, Integer] codec name such as :SNAPPY (the raw Integer when unknown)
132
+ def codec = Format::Codec::NAMES[@meta.codec] || @meta.codec
133
+
134
+ # @return [Array<String>] encoding names listed in the column metadata
135
+ def encodings = (@meta.encodings || []).map { |e| Inspector.encoding_name(e) }
136
+
137
+ # @return [Integer] values in the chunk, nulls and repeated entries included
138
+ def num_values = @meta.num_values
139
+
140
+ # @return [Integer] total_compressed_size from the column metadata (page headers included)
141
+ def compressed_size = @meta.total_compressed_size
142
+
143
+ # @return [Integer] total_uncompressed_size from the column metadata (page headers included)
144
+ def uncompressed_size = @meta.total_uncompressed_size
145
+
146
+ # @return [Integer] offset of the first data page, as declared in the column metadata
147
+ def data_page_offset = @meta.data_page_offset
148
+
149
+ # @return [String, nil] path of the file holding the chunk when it is not stored in this one
150
+ def external_file = @chunk.file_path
151
+
152
+ # @return [Float, nil] uncompressed size divided by compressed size; nil when nothing is compressed
153
+ def compression_ratio
154
+ compressed_size.to_i.positive? ? uncompressed_size.to_f / compressed_size : nil
155
+ end
156
+
157
+ # Some writers store 0 when there is no dictionary page, and some store a data_page_offset
158
+ # of 0 for empty chunks (which would point at the magic bytes)
159
+ # @return [Integer, nil] the declared dictionary page offset, nil when absent or implausible
160
+ def dictionary_page_offset
161
+ d = @meta.dictionary_page_offset
162
+ (d && d >= 4 && (d < @meta.data_page_offset.to_i || @meta.data_page_offset.to_i < 4)) ? d : nil
163
+ end
164
+
165
+ # Where the chunk's first page starts
166
+ # @return [Integer] file offset
167
+ def start_offset = dictionary_page_offset || data_page_offset
168
+
169
+ # The end according to the metadata; some writers under-report it (see #end_offset)
170
+ # @return [Integer] file offset just past the chunk
171
+ def declared_end_offset = start_offset + compressed_size
172
+
173
+ # The end of the last page actually found (the declared end if the pages could not be walked)
174
+ # @return [Integer] file offset just past the chunk, never before #declared_end_offset
175
+ def end_offset
176
+ last = pages.last
177
+ last ? [last.end_offset, declared_end_offset].max : declared_end_offset
178
+ end
179
+
180
+ # Page counts per page type and encoding, from the column metadata's encoding_stats
181
+ # @return [Array<Hash{Symbol => Object}>] each { page_type:, encoding:, count: }; empty when
182
+ # the writer stored none
183
+ def encoding_stats
184
+ (@meta.encoding_stats || []).map do |s|
185
+ {page_type: Format::PageType::NAMES[s.page_type] || s.page_type,
186
+ encoding: Inspector.encoding_name(s.encoding), count: s.count}
187
+ end
188
+ end
189
+
190
+ # Chunk-level statistics from the column metadata, decoded (memoized)
191
+ # @return [Stats, nil] nil when the chunk has no statistics
192
+ def statistics
193
+ return @statistics if defined?(@statistics)
194
+ @statistics = @inspector.decode_statistics(@meta.statistics, @column)
195
+ end
196
+
197
+ # @return [Hash{Symbol => Object}, nil] the SizeStatistics (unencoded byte array sizes and
198
+ # level histograms) as a Hash, nil when absent
199
+ def size_statistics
200
+ s = @meta.size_statistics
201
+ s&.to_h
202
+ end
203
+
204
+ # @return [Hash{String => String}] the chunk's own key/value metadata (rarely used by writers)
205
+ def key_value_metadata
206
+ (@meta.key_value_metadata || []).to_h { |kv| [kv.key, kv.value] }
207
+ end
208
+
209
+ # @return [Integer, nil] file offset of the chunk's bloom filter, nil when it has none
210
+ def bloom_filter_offset = @meta.bloom_filter_offset
211
+
212
+ # Bytes taken by the bloom filter (header + bitset); read from its header when the footer has no length
213
+ # @return [Integer, nil] nil when there is no bloom filter or its header can't be decoded
214
+ def bloom_filter_length
215
+ return nil unless bloom_filter_offset
216
+ return @meta.bloom_filter_length if @meta.bloom_filter_length
217
+ return @bloom_filter_length if defined?(@bloom_filter_length)
218
+ @bloom_filter_length = @inspector.bloom_filter_size(bloom_filter_offset)
219
+ end
220
+
221
+ # @return [Array(Integer, Integer), nil] [offset, length] of the chunk's ColumnIndex, nil when
222
+ # it has none
223
+ def column_index_range
224
+ o = @chunk.column_index_offset
225
+ (o && @chunk.column_index_length) ? [o, @chunk.column_index_length] : nil
226
+ end
227
+
228
+ # @return [Array(Integer, Integer), nil] [offset, length] of the chunk's OffsetIndex, nil when
229
+ # it has none
230
+ def offset_index_range
231
+ o = @chunk.offset_index_offset
232
+ (o && @chunk.offset_index_length) ? [o, @chunk.offset_index_length] : nil
233
+ end
234
+
235
+ # All page headers, in file order. Errors while walking (corrupt or truncated headers) end
236
+ # the walk; #error then says what went wrong and #pages holds the pages found before it.
237
+ # @return [Array<PageInfo>]
238
+ def pages
239
+ @pages ||= begin
240
+ list, @error = @inspector.walk_pages(self)
241
+ apply_offset_index(list)
242
+ list
243
+ end
244
+ end
245
+
246
+ # @return [Array<PageInfo>] the v1 and v2 data pages, in file order
247
+ def data_pages = pages.select(&:data?)
248
+
249
+ # @return [PageInfo, nil] the dictionary page, nil when the chunk has none
250
+ def dictionary_page = pages.find(&:dictionary?)
251
+
252
+ # Number of entries in the dictionary (from its page header)
253
+ # @return [Integer, nil] nil when the chunk has no dictionary page
254
+ def dictionary_size = dictionary_page&.num_values
255
+
256
+ # The chunk's ColumnIndex, read and decoded on first use
257
+ # @return [ColumnIndexInfo, nil] nil when there is none or it is corrupt
258
+ def column_index
259
+ return @column_index if defined?(@column_index)
260
+ @column_index = @inspector.read_column_index(self)
261
+ end
262
+
263
+ # The chunk's OffsetIndex, read and decoded on first use
264
+ # @return [OffsetIndexInfo, nil] nil when there is none or it is corrupt
265
+ def offset_index
266
+ return @offset_index if defined?(@offset_index)
267
+ @offset_index = @inspector.read_offset_index(self)
268
+ end
269
+
270
+ # Null count from the chunk statistics, else the sum over the data page headers
271
+ # @return [Integer, nil] nil when neither the statistics nor every data page has it
272
+ def null_count
273
+ return statistics.null_count if statistics&.null_count
274
+ counts = data_pages.map(&:num_nulls)
275
+ counts.all? ? counts.sum : nil
276
+ end
277
+
278
+ # Reads every page body (compressed bytes, as stored) and checks it against the CRC in its
279
+ # page header. Sets PageInfo#checksum on each page and returns the pages' statuses.
280
+ # @return [Array<Symbol>] :ok, :mismatch or :absent per page, in #pages order
281
+ def verify_checksums
282
+ pages.map { |p| p.checksum = @inspector.page_checksum(p) }
283
+ end
284
+
285
+ # Page statistics (from page headers) that disagree with the ColumnIndex entry for the
286
+ # same page. Each: { page:, data_page:, field:, page_value:, index_value: } where +page+
287
+ # is the page's position in #pages, +data_page+ its position among the data pages (and
288
+ # in the ColumnIndex), +field+ one of :min, :max, :null_count, :null_page, :page_count.
289
+ # A ColumnIndex bound that is wider than the page's (e.g. a truncated string prefix) is
290
+ # allowed; one that is narrower, or a differing null count, is reported. Empty when the
291
+ # chunk has no ColumnIndex or its pages carry no statistics.
292
+ # @return [Array<Hash{Symbol => Object}>]
293
+ def index_mismatches
294
+ @index_mismatches ||= @inspector.compare_page_index(self)
295
+ end
296
+
297
+ # Everything known about the chunk, including its pages and page indexes (reads them if
298
+ # they were not read yet)
299
+ # @return [Hash{Symbol => Object}] JSON-safe; keys without a value are left out
300
+ def to_h
301
+ Inspector.jsonable({
302
+ path: path,
303
+ column: @column.index,
304
+ type: Inspector.type_name(@column),
305
+ codec: codec,
306
+ encodings: encodings,
307
+ encoding_stats: encoding_stats,
308
+ num_values: num_values,
309
+ null_count: null_count,
310
+ compressed_size: compressed_size,
311
+ uncompressed_size: uncompressed_size,
312
+ compression_ratio: compression_ratio&.round(4),
313
+ file_offset: @chunk.file_offset,
314
+ start_offset: start_offset,
315
+ end_offset: end_offset,
316
+ data_page_offset: data_page_offset,
317
+ dictionary_page_offset: dictionary_page_offset,
318
+ dictionary: dictionary_page && {offset: dictionary_page.offset, num_values: dictionary_page.num_values,
319
+ compressed_size: dictionary_page.compressed_size, uncompressed_size: dictionary_page.uncompressed_size,
320
+ is_sorted: dictionary_page.is_sorted},
321
+ statistics: statistics&.to_h,
322
+ size_statistics: size_statistics,
323
+ key_value_metadata: key_value_metadata.empty? ? nil : key_value_metadata,
324
+ bloom_filter: bloom_filter_offset && {offset: bloom_filter_offset, length: bloom_filter_length},
325
+ column_index_offset: column_index_range&.first,
326
+ column_index_length: column_index_range&.last,
327
+ offset_index_offset: offset_index_range&.first,
328
+ offset_index_length: offset_index_range&.last,
329
+ external_file: external_file,
330
+ num_pages: pages.size,
331
+ num_data_pages: data_pages.size,
332
+ index_mismatches: index_mismatches.empty? ? nil : index_mismatches,
333
+ error: error,
334
+ pages: pages.map(&:to_h),
335
+ column_index: column_index&.to_h,
336
+ offset_index: offset_index&.to_h
337
+ }.compact)
338
+ end
339
+
340
+ # @return [String] short description: path, row group, codec and sizes
341
+ def inspect
342
+ "#<#{self.class.name} #{path} rg=#{@row_group.index} #{codec} #{compressed_size}/#{uncompressed_size} bytes>"
343
+ end
344
+
345
+ private
346
+
347
+ # Page row counts come from the offset index when there is one (v1 pages don't carry them)
348
+ # Sets +first_row_index+ on the data pages the index points at, and +num_rows+ where the
349
+ # page header did not provide it.
350
+ # @param list [Array<PageInfo>] pages of this chunk, updated in place
351
+ # @return [void]
352
+ def apply_offset_index(list)
353
+ oi = offset_index or return
354
+ by_offset = list.select(&:data?).to_h { |p| [p.offset, p] }
355
+ locs = oi.page_locations
356
+ locs.each_with_index do |loc, i|
357
+ page = by_offset[loc.offset] or next
358
+ page.first_row_index = loc.first_row_index
359
+ next_first = locs[i + 1]&.first_row_index || @row_group.num_rows
360
+ page.num_rows ||= next_first - loc.first_row_index
361
+ end
362
+ end
363
+ end
364
+
365
+ # One row group
366
+ class RowGroupInfo
367
+ attr_reader :index, :row_group, :columns
368
+
369
+ # @param inspector [Inspector] owner, passed on to the column chunks
370
+ # @param index [Integer] position of the row group in the file
371
+ # @param row_group [Format::RowGroup] row group as decoded from the footer
372
+ # @param first_row [Integer] file-wide index of the row group's first row
373
+ # @raise [FormatError] when the row group has more column chunks than the schema has leaves
374
+ def initialize(inspector, index, row_group, first_row)
375
+ @index = index
376
+ @row_group = row_group
377
+ @first_row = first_row
378
+ leaves = inspector.schema.columns
379
+ @columns = (row_group.columns || []).each_with_index.map do |cc, i|
380
+ column = leaves[i] or raise FormatError, "Row group #{index} has more column chunks than the schema has columns"
381
+ ColumnChunkInfo.new(inspector, self, column, cc)
382
+ end
383
+ end
384
+
385
+ # @return [Integer] rows in the row group
386
+ def num_rows = @row_group.num_rows
387
+ # @return [Integer] file-wide index of the row group's first row
388
+ attr_reader :first_row
389
+
390
+ # @return [Integer] total_byte_size from the footer (uncompressed size of all column data)
391
+ def total_byte_size = @row_group.total_byte_size
392
+
393
+ # @return [Integer] total_compressed_size from the footer, else the sum over the chunks
394
+ def compressed_size = @row_group.total_compressed_size || @columns.sum(&:compressed_size)
395
+
396
+ # @return [Integer] sum of the chunks' total_uncompressed_size
397
+ def uncompressed_size = @columns.sum { |c| c.uncompressed_size.to_i }
398
+
399
+ # @return [Integer, nil] lowest start offset of the row group's chunks; nil without chunks
400
+ def start_offset = @columns.map(&:start_offset).min
401
+
402
+ # @return [Integer, nil] highest end offset of the row group's chunks; nil without chunks
403
+ def end_offset = @columns.map(&:end_offset).max
404
+
405
+ # @param path [String, Array<String>] dotted path (+"a.b"+) or path segments (+["a", "b"]+)
406
+ # @return [ColumnChunkInfo, nil] the chunk of that column, nil when there is none
407
+ def column(path) = @columns.find { |c| c.path == path.to_s || c.column.path == Array(path) }
408
+
409
+ # The row group's declared sort order
410
+ # @return [Array<Hash{Symbol => Object}>] each { column:, descending:, nulls_first: }, where
411
+ # +column+ is the column's path (its index when out of range)
412
+ def sorting_columns
413
+ (@row_group.sorting_columns || []).map do |s|
414
+ {column: @columns[s.column_idx]&.path || s.column_idx, descending: s.descending, nulls_first: s.nulls_first}
415
+ end
416
+ end
417
+
418
+ # Everything known about the row group, with each column chunk's #to_h
419
+ # @return [Hash{Symbol => Object}] JSON-safe; keys without a value are left out
420
+ def to_h
421
+ Inspector.jsonable({
422
+ index: index,
423
+ ordinal: @row_group.ordinal,
424
+ num_rows: num_rows,
425
+ first_row: first_row,
426
+ total_byte_size: total_byte_size,
427
+ compressed_size: compressed_size,
428
+ uncompressed_size: uncompressed_size,
429
+ file_offset: @row_group.file_offset,
430
+ start_offset: start_offset,
431
+ end_offset: end_offset,
432
+ sorting_columns: sorting_columns,
433
+ columns: @columns.map(&:to_h)
434
+ }.compact)
435
+ end
436
+
437
+ # @return [String] short description: index, row count and number of columns
438
+ def inspect
439
+ "#<#{self.class.name} #{index} rows=#{num_rows} columns=#{@columns.size}>"
440
+ end
441
+ end
442
+
443
+ attr_reader :metadata, :schema, :file_size, :footer_size
444
+
445
+ # A label for the file (the basename of the IO's path, when it has one)
446
+ attr_reader :name
447
+
448
+ # +io+ is a random-access IO (responds to #seek and #read, e.g. File.open(path, "rb")). It is
449
+ # left open.
450
+ # @param io [IO] random-access IO positioned anywhere; only the footer is read here
451
+ # @raise [ArgumentError] when +io+ does not support #seek and #read
452
+ # @raise [FormatError] when the file is too small, lacks the magic bytes or has a corrupt footer
453
+ # @raise [UnsupportedError] when the file is encrypted (PARE magic)
454
+ def initialize(io)
455
+ @io = io
456
+ unless @io.respond_to?(:read) && @io.respond_to?(:seek)
457
+ raise ArgumentError, "Herringbone::Inspector expects an IO that supports #seek and #read " \
458
+ "(e.g. File.open(path, \"rb\")), got #{io.class}"
459
+ end
460
+ @name = (@io.respond_to?(:path) && @io.path) ? File.basename(@io.path.to_s) : nil
461
+ read_footer
462
+ @schema = Schema.from_elements(@metadata.schema)
463
+ end
464
+
465
+ # @return [Integer] row count declared in the footer
466
+ def num_rows = @metadata.num_rows
467
+
468
+ # @return [String, nil] the writer's created_by string, e.g. +"parquet-cpp-arrow version 15.0.0"+
469
+ def created_by = @metadata.created_by
470
+
471
+ # @return [Integer] format version from the footer (1 or 2; says little about the features used)
472
+ def version = @metadata.version
473
+
474
+ # @return [Array<Schema::Column>] the schema's leaf columns
475
+ def columns = @schema.columns
476
+
477
+ # @return [Integer] file offset where the Thrift-encoded FileMetaData starts
478
+ def footer_offset = @file_size - 8 - @footer_size
479
+
480
+ # @return [Array<RowGroupInfo>] the row groups, in footer order (built on first use)
481
+ def row_groups
482
+ @row_groups ||= begin
483
+ first = 0
484
+ (@metadata.row_groups || []).each_with_index.map do |rg, i|
485
+ info = RowGroupInfo.new(self, i, rg, first)
486
+ first += rg.num_rows.to_i
487
+ info
488
+ end
489
+ end
490
+ end
491
+
492
+ # @return [Array<ColumnChunkInfo>] the column chunks of all row groups, row group by row group
493
+ def column_chunks = row_groups.flat_map(&:columns)
494
+
495
+ # Walks every page header and page index now (e.g. before closing the file)
496
+ # @return [Inspector] self
497
+ def load_all
498
+ column_chunks.each do |c|
499
+ c.pages
500
+ c.column_index
501
+ c.offset_index
502
+ c.bloom_filter_length
503
+ end
504
+ self
505
+ end
506
+
507
+ # @return [Boolean] whether any column chunk has a ColumnIndex or an OffsetIndex
508
+ def page_index? = column_chunks.any? { |c| c.column_index_range || c.offset_index_range }
509
+
510
+ # @return [Boolean] whether any column chunk has a bloom filter
511
+ def bloom_filters? = column_chunks.any?(&:bloom_filter_offset)
512
+
513
+ # The file-level key/value metadata, each entry described: ARROW:schema is decoded, JSON
514
+ # values are parsed (pandas metadata summarized), binary values are shown as hex
515
+ # @return [Array<Hash{Symbol => Object}>] each with :key, :bytesize, :format ("arrow_schema",
516
+ # "json", "text" or "binary"), :value (truncated) and, depending on the format, :summary,
517
+ # :json, :arrow_schema, :arrow_fields, :arrow_error
518
+ def key_value_metadata
519
+ (@metadata.key_value_metadata || []).map { |kv| describe_key_value(kv.key, kv.value) }
520
+ end
521
+
522
+ # @return [Array<String>, nil] "TYPE_DEFINED_ORDER" or "UNKNOWN" per leaf column; nil when the
523
+ # footer has no column_orders
524
+ def column_orders
525
+ orders = @metadata.column_orders or return nil
526
+ orders.map { |o| o.type_order ? "TYPE_DEFINED_ORDER" : "UNKNOWN" }
527
+ end
528
+
529
+ # File-wide facts: sizes, row and column counts, codecs, whether page indexes and bloom
530
+ # filters are present, and (after #verify_checksums) the CRC tallies under :checksums.
531
+ # Walks no page headers.
532
+ # @return [Hash{Symbol => Object}]
533
+ def summary
534
+ codecs = column_chunks.map(&:codec).uniq
535
+ {
536
+ name: @name,
537
+ file_size: @file_size,
538
+ footer_size: @footer_size,
539
+ footer_offset: footer_offset,
540
+ format_version: version,
541
+ created_by: created_by,
542
+ num_rows: num_rows,
543
+ num_row_groups: row_groups.size,
544
+ num_columns: columns.size,
545
+ codecs: codecs,
546
+ compressed_size: column_chunks.sum { |c| c.compressed_size.to_i },
547
+ uncompressed_size: column_chunks.sum { |c| c.uncompressed_size.to_i },
548
+ page_index: page_index?,
549
+ bloom_filters: bloom_filters?,
550
+ column_orders: column_orders
551
+ }.tap { |h| h[:checksums] = checksum_summary.except(:mismatches) if checksums_verified? }
552
+ end
553
+
554
+ # Reads every page body and checks it against the CRC32 in its page header (the CRC covers
555
+ # the page data as stored: compressed, and for v2 pages the levels plus the compressed
556
+ # values). Nothing is decompressed. Sets PageInfo#checksum on every page and returns
557
+ # #checksum_summary. Needs the IO, so call it before closing the file.
558
+ # @return [Hash{Symbol => Object}] see #checksum_summary
559
+ def verify_checksums
560
+ column_chunks.each(&:verify_checksums)
561
+ @checksums_verified = true
562
+ checksum_summary
563
+ end
564
+
565
+ # @return [Boolean] whether #verify_checksums has run
566
+ def checksums_verified? = @checksums_verified == true
567
+
568
+ # After #verify_checksums: { ok:, mismatch:, absent:, mismatches: [{ row_group:, column:, page:, type:, offset:, crc:, actual: }] }
569
+ # The counts are pages per status; in each mismatch +crc+ is the CRC from the page header and
570
+ # +actual+ the CRC32 of the stored bytes (kept from the verification, so the IO is not needed).
571
+ # @return [Hash{Symbol => Object}, nil] nil before #verify_checksums
572
+ def checksum_summary
573
+ return nil unless checksums_verified?
574
+ all = column_chunks.flat_map { |c| c.pages.map { |p| [c, p] } }
575
+ tally = all.map { |_, p| p.checksum }.tally
576
+ {
577
+ ok: tally.fetch(:ok, 0), mismatch: tally.fetch(:mismatch, 0), absent: tally.fetch(:absent, 0),
578
+ mismatches: all.select { |_, p| p.checksum == :mismatch }.map do |c, p|
579
+ {row_group: c.row_group.index, column: c.path, page: p.index, type: p.type, offset: p.offset,
580
+ crc: p.expected_crc, actual: p.actual_crc}
581
+ end
582
+ }
583
+ end
584
+
585
+ # Every disagreement between page header statistics and the ColumnIndex, across the file,
586
+ # each with :row_group and :column added (see ColumnChunkInfo#index_mismatches)
587
+ # @return [Array<Hash{Symbol => Object}>]
588
+ def index_mismatches
589
+ column_chunks.flat_map do |c|
590
+ c.index_mismatches.map { |m| {row_group: c.row_group.index, column: c.path}.merge(m) }
591
+ end
592
+ end
593
+
594
+ # The decoded ARROW:schema key/value (see ArrowSchema.decode); nil when the file has none or
595
+ # it could not be decoded, and #arrow_schema_error then says why
596
+ # @return [Hash{Symbol => Object}, nil] see ArrowSchema.decode
597
+ def arrow_schema
598
+ return @arrow_schema if defined?(@arrow_schema)
599
+ @arrow_schema_error = nil
600
+ kv = (@metadata.key_value_metadata || []).find { |x| x.key == "ARROW:schema" }
601
+ @arrow_schema = kv && begin
602
+ ArrowSchema.decode(kv.value)
603
+ rescue => e
604
+ @arrow_schema_error = "could not decode ARROW:schema: #{e.message}"
605
+ nil
606
+ end
607
+ end
608
+
609
+ # @return [String, nil] why the ARROW:schema value could not be decoded; nil when it decoded
610
+ # fine or the file has none
611
+ def arrow_schema_error
612
+ arrow_schema
613
+ @arrow_schema_error
614
+ end
615
+
616
+ # Schema tree: Hashes with name, repetition, types, levels (leaves) and children (groups)
617
+ # Nodes that match a field of the ARROW:schema also get its type as :arrow_type.
618
+ # @return [Array<Hash{Symbol => Object}>] the root's children
619
+ def schema_tree
620
+ leaf_by_node = @schema.columns.to_h { |c| [c.node, c] }
621
+ build = lambda do |node|
622
+ h = {name: node.name, repetition: node.repetition}
623
+ if node.leaf?
624
+ col = leaf_by_node[node]
625
+ h[:physical_type] = T::NAMES[node.type]&.to_s
626
+ h[:type_length] = node.type_length
627
+ h[:column] = col.index
628
+ h[:path] = col.dotted_path
629
+ h[:max_definition_level] = col.max_definition_level
630
+ h[:max_repetition_level] = col.max_repetition_level
631
+ end
632
+ h[:logical_type] = Inspector.logical_type_name(node)
633
+ h[:converted_type] = Format::ConvertedType::NAMES[node.converted_type]&.to_s
634
+ h[:field_id] = node.field_id
635
+ h[:children] = node.children.map { |c| build.call(c) } if node.group?
636
+ h.compact
637
+ end
638
+ tree = @schema.root.children.map { |c| build.call(c) }
639
+ annotate_arrow_types(tree, arrow_schema[:fields]) if arrow_schema
640
+ tree
641
+ end
642
+
643
+ # Per leaf column: sums over all row groups, plus overall min/max where comparable
644
+ # Walks every page header (for the page counts).
645
+ # @return [Array<Hash{Symbol => Object}>] one per leaf column, in schema order
646
+ def column_totals
647
+ columns.map do |col|
648
+ chunks = row_groups.map { |rg| rg.columns[col.index] }.compact
649
+ compressed = chunks.sum { |c| c.compressed_size.to_i }
650
+ uncompressed = chunks.sum { |c| c.uncompressed_size.to_i }
651
+ nulls = chunks.map(&:null_count)
652
+ stats = chunks.map(&:statistics)
653
+ mins = stats.map { |s| s&.min }
654
+ maxes = stats.map { |s| s&.max }
655
+ {
656
+ column: col.index,
657
+ path: col.dotted_path,
658
+ type: Inspector.type_name(col),
659
+ codecs: chunks.map(&:codec).uniq,
660
+ encodings: chunks.flat_map(&:encodings).uniq,
661
+ num_values: chunks.sum { |c| c.num_values.to_i },
662
+ null_count: nulls.all? ? nulls.sum : nil,
663
+ compressed_size: compressed,
664
+ uncompressed_size: uncompressed,
665
+ compression_ratio: compressed.positive? ? (uncompressed.to_f / compressed).round(4) : nil,
666
+ num_pages: chunks.sum { |c| c.pages.size },
667
+ num_data_pages: chunks.sum { |c| c.data_pages.size },
668
+ dictionary_pages: chunks.count(&:dictionary_page),
669
+ dictionary_bytes: chunks.sum { |c| c.dictionary_page&.total_size.to_i },
670
+ min: mins.include?(nil) ? nil : safe_extreme(mins, :min),
671
+ max: maxes.include?(nil) ? nil : safe_extreme(maxes, :max)
672
+ }.compact
673
+ end
674
+ end
675
+
676
+ # Byte ranges of the whole file, in offset order: magic, pages (or whole chunks when their pages
677
+ # could not be walked), bloom filters, page indexes, footer. Gaps right after a chunk at its
678
+ # file_offset are inline :column_metadata copies; other gaps are reported as :unknown. Each entry: { kind:, start:, length:, row_group:, column:, page: }
679
+ # +row_group+, +column+ and +page+ are indexes, present only where they apply. Segments can
680
+ # overlap when the file is damaged; gaps are only computed past the furthest end seen so far.
681
+ # @return [Array<Hash{Symbol => Object}>]
682
+ def layout
683
+ segs = [{kind: :magic, start: 0, length: 4}]
684
+ column_chunks.each do |c|
685
+ rg = c.row_group.index
686
+ col = c.column.index
687
+ if c.pages.empty?
688
+ segs << {kind: :chunk, start: c.start_offset, length: c.compressed_size, row_group: rg, column: col}
689
+ else
690
+ c.pages.each_with_index do |p, i|
691
+ segs << {kind: p.dictionary? ? :dictionary_page : :data_page, start: p.offset, length: p.total_size,
692
+ row_group: rg, column: col, page: i}
693
+ end
694
+ end
695
+ if c.bloom_filter_offset
696
+ segs << {kind: :bloom_filter, start: c.bloom_filter_offset, length: c.bloom_filter_length.to_i,
697
+ row_group: rg, column: col}
698
+ end
699
+ if (r = c.column_index_range)
700
+ segs << {kind: :column_index, start: r[0], length: r[1], row_group: rg, column: col}
701
+ end
702
+ if (r = c.offset_index_range)
703
+ segs << {kind: :offset_index, start: r[0], length: r[1], row_group: rg, column: col}
704
+ end
705
+ end
706
+ segs << {kind: :footer, start: footer_offset, length: @footer_size}
707
+ segs << {kind: :footer_length, start: @file_size - 8, length: 4}
708
+ segs << {kind: :magic, start: @file_size - 4, length: 4}
709
+ segs.sort_by! { |s| s[:start] }
710
+ # Some writers (old parquet-rs, parquet-mr for a while) store a copy of the ColumnMetaData
711
+ # right after the chunk, at ColumnChunk.file_offset
712
+ meta_copies = column_chunks.each_with_object({}) do |c, h|
713
+ fo = c.chunk.file_offset
714
+ h[fo] = c if fo && fo >= c.end_offset
715
+ end
716
+ out = []
717
+ pos = 0
718
+ segs.each do |s|
719
+ if s[:start] > pos
720
+ c = meta_copies[pos]
721
+ out << if c
722
+ {kind: :column_metadata, start: pos, length: s[:start] - pos, row_group: c.row_group.index, column: c.column.index}
723
+ else
724
+ {kind: :unknown, start: pos, length: s[:start] - pos}
725
+ end
726
+ end
727
+ out << s
728
+ pos = [pos, s[:start] + s[:length]].max
729
+ end
730
+ out
731
+ end
732
+
733
+ # Everything: summary, key/value metadata, schema tree, row groups with their chunks and
734
+ # pages, column totals, and any checksum and page index mismatches. Walks every page header.
735
+ # @return [Hash{Symbol => Object}] JSON-safe; keys without a value are left out
736
+ def to_h
737
+ Inspector.jsonable({
738
+ summary: summary,
739
+ key_value_metadata: key_value_metadata,
740
+ schema: schema_tree,
741
+ row_groups: row_groups.map(&:to_h),
742
+ column_totals: column_totals,
743
+ checksum_mismatches: checksum_summary&.fetch(:mismatches),
744
+ index_mismatches: index_mismatches
745
+ }.compact)
746
+ end
747
+
748
+ # @param args [Array] passed on to Hash#to_json (e.g. a JSON::State)
749
+ # @return [String] #to_h as JSON
750
+ def to_json(*args) = to_h.to_json(*args)
751
+
752
+ # A self-contained HTML page showing the file's layout, see Visualizer
753
+ # @return [String] HTML document
754
+ def to_html = Visualizer.new(self).to_html
755
+
756
+ # Readable text summary. With pages: true, lists every page header too.
757
+ # @param pages [Boolean] whether to add a line per page header under each column chunk
758
+ # @return [String] multi-line text, as printed by +bin/herringbone inspect+
759
+ def report(pages: false)
760
+ s = summary
761
+ out = []
762
+ out << "file: #{@name || "(IO)"}"
763
+ out << "size: #{Inspector.human_bytes(s[:file_size])} (#{s[:file_size]} bytes), footer #{s[:footer_size]} bytes at #{s[:footer_offset]}"
764
+ out << "rows: #{s[:num_rows]}, row groups: #{s[:num_row_groups]}, columns: #{s[:num_columns]}, format version #{s[:format_version]}"
765
+ out << "created by: #{s[:created_by]}"
766
+ out << "codecs: #{s[:codecs].join(", ")}; data #{Inspector.human_bytes(s[:compressed_size])} compressed, " \
767
+ "#{Inspector.human_bytes(s[:uncompressed_size])} uncompressed#{ratio_text(s[:uncompressed_size], s[:compressed_size])}"
768
+ out << "page index: #{s[:page_index] ? "yes" : "no"}, bloom filters: #{s[:bloom_filters] ? "yes" : "no"}"
769
+ if (cs = checksum_summary)
770
+ out << "page CRCs: #{cs[:ok]} ok, #{cs[:mismatch]} mismatched, #{cs[:absent]} without a CRC"
771
+ cs[:mismatches].each do |m|
772
+ out << " CRC MISMATCH: row group #{m[:row_group]} #{m[:column]} page #{m[:page]} (#{m[:type]} @#{m[:offset]}): " \
773
+ "header says #{format("%08x", m[:crc])}, data has #{format("%08x", m[:actual])}"
774
+ end
775
+ end
776
+ mismatches = index_mismatches
777
+ unless mismatches.empty?
778
+ out << "page statistics vs column index: #{mismatches.size} disagreement#{"s" unless mismatches.size == 1}"
779
+ mismatches.first(50).each { |m| out << " #{index_mismatch_text(m)}" }
780
+ out << " ..." if mismatches.size > 50
781
+ end
782
+ kvs = key_value_metadata
783
+ unless kvs.empty?
784
+ out << "key/value metadata:"
785
+ kvs.each do |kv|
786
+ out << " #{kv[:key]} (#{kv[:format]}, #{kv[:bytesize]} bytes): #{kv[:summary] || kv[:value].to_s[0, 80].inspect}"
787
+ out << " #{kv[:arrow_error]}" if kv[:arrow_error]
788
+ ArrowSchema.lines(kv[:arrow_schema][:fields]).each { |l| out << " #{l}" } if kv[:arrow_schema]
789
+ end
790
+ end
791
+ out << "schema:"
792
+ walk = lambda do |n, depth|
793
+ type = n[:children] ? "group" : [n[:physical_type], n[:type_length] && "(#{n[:type_length]})"].compact.join
794
+ ann = n[:logical_type] || n[:converted_type]
795
+ levels = n[:children] ? "" : " [def #{n[:max_definition_level]}, rep #{n[:max_repetition_level]}]"
796
+ arrow = n[:arrow_type] ? " arrow: #{n[:arrow_type]}" : ""
797
+ out << "#{" " * depth}#{n[:repetition]} #{type} #{n[:name]}#{" (#{ann})" if ann}#{levels}#{arrow}"
798
+ (n[:children] || []).each { |c| walk.call(c, depth + 1) }
799
+ end
800
+ schema_tree.each { |n| walk.call(n, 1) }
801
+ out << "columns:"
802
+ column_totals.each do |t|
803
+ range = t.key?(:min) ? " [#{Inspector.display(t[:min])} .. #{Inspector.display(t[:max])}]" : ""
804
+ out << " #{t[:path]}: #{t[:type]} #{t[:codecs].join(",")} #{t[:encodings].join(",")} " \
805
+ "#{Inspector.human_bytes(t[:compressed_size])}/#{Inspector.human_bytes(t[:uncompressed_size])}" \
806
+ "#{ratio_text(t[:uncompressed_size], t[:compressed_size])}, #{t[:num_values]} values" \
807
+ "#{", #{t[:null_count]} nulls" if t[:null_count]}, #{t[:num_data_pages]} data pages#{range}"
808
+ end
809
+ row_groups.each do |rg|
810
+ sorting = rg.sorting_columns.map { |c| "#{c[:column]}#{" desc" if c[:descending]}" }
811
+ out << "row group #{rg.index}: #{rg.num_rows} rows, #{Inspector.human_bytes(rg.compressed_size)} " \
812
+ "at #{rg.start_offset}..#{rg.end_offset}#{", sorted by #{sorting.join(", ")}" unless sorting.empty?}"
813
+ rg.columns.each do |c|
814
+ st = c.statistics
815
+ range = (st && !(st.min.nil? && st.max.nil?)) ? " [#{Inspector.display(st.min)} .. #{Inspector.display(st.max)}]" : ""
816
+ extras = []
817
+ extras << "dict #{c.dictionary_size} entries" if c.dictionary_page
818
+ extras << "column index" if c.column_index_range
819
+ extras << "offset index" if c.offset_index_range
820
+ extras << "bloom filter" if c.bloom_filter_offset
821
+ extras << "ERROR: #{c.error}" if c.error
822
+ out << " #{c.path}: #{c.codec} #{c.encodings.join(",")} " \
823
+ "#{c.compressed_size}/#{c.uncompressed_size} bytes#{ratio_text(c.uncompressed_size, c.compressed_size)}, " \
824
+ "#{c.num_values} values, #{c.pages.size} pages#{", #{extras.join(", ")}" unless extras.empty?}#{range}"
825
+ next unless pages
826
+ c.pages.each do |p|
827
+ st = p.statistics
828
+ out << " #{p.index}: #{p.type} @#{p.offset} header #{p.header_size} + #{p.compressed_size}/#{p.uncompressed_size} bytes, " \
829
+ "#{p.num_values} values#{", #{p.num_nulls} nulls" if p.num_nulls}#{", #{p.num_rows} rows" if p.num_rows}" \
830
+ "#{" #{p.encoding}" if p.encoding}#{crc_text(p)}" \
831
+ "#{" [#{Inspector.display(st.min)} .. #{Inspector.display(st.max)}]" if st && !(st.min.nil? && st.max.nil?)}"
832
+ end
833
+ end
834
+ end
835
+ out.join("\n")
836
+ end
837
+
838
+ # @return [String] short description: name, rows, row group count and file size
839
+ def inspect
840
+ "#<#{self.class.name} #{@name || "(IO)"} rows=#{num_rows} row_groups=#{row_groups.size} size=#{@file_size}>"
841
+ end
842
+
843
+ # ---- used by the info objects ----
844
+
845
+ # Walks page headers from the chunk's first page. Returns [pages, error_message_or_nil].
846
+ # Mirrors the reader's tolerance: a chunk may extend past its declared total_compressed_size.
847
+ # @param chunk [ColumnChunkInfo] chunk whose pages to walk
848
+ # @return [Array(Array<PageInfo>, String), Array(Array<PageInfo>, nil)] the pages found, and why
849
+ # the walk stopped early (nil when every value was accounted for)
850
+ def walk_pages(chunk)
851
+ return [[], "column chunk stored in external file #{chunk.external_file}"] if chunk.external_file
852
+ pages = []
853
+ pos = chunk.start_offset
854
+ limit = footer_offset
855
+ total = chunk.num_values.to_i
856
+ seen = 0
857
+ declared_end = chunk.declared_end_offset
858
+ # Walk until all values are accounted for (reading past the declared end if a writer
859
+ # under-reported it), then pick up any trailing pages still inside the declared range
860
+ while pos < limit && (seen < total || pos < declared_end)
861
+ trailing = seen >= total
862
+ begin
863
+ header, header_size = read_page_header(pos, limit)
864
+ page = page_info(pages.size, header, pos, header_size, chunk.column)
865
+ rescue Thrift::Error, FormatError => e
866
+ break if trailing
867
+ return [pages, "corrupt page header at #{pos}: #{e.message}"]
868
+ end
869
+ if page.end_offset > limit
870
+ break if trailing
871
+ return [pages, "page #{pages.size} at #{pos} overruns the data section (#{page.end_offset} > #{limit})"]
872
+ end
873
+ pages << page
874
+ seen += page.num_values.to_i if page.data?
875
+ pos = page.end_offset
876
+ end
877
+ [pages, (seen < total) ? "found #{seen} of #{total} values in page headers" : nil]
878
+ end
879
+
880
+ # Reads and decodes a chunk's ColumnIndex, decoding the per-page min/max with the column type
881
+ # @param chunk [ColumnChunkInfo] chunk whose ColumnIndex to read
882
+ # @return [ColumnIndexInfo, nil] nil when the chunk has none or it fails to decode
883
+ def read_column_index(chunk)
884
+ offset, length = chunk.column_index_range
885
+ return nil unless offset && length.positive?
886
+ ci = Format::ColumnIndex.decode(read_at(offset, length)).first
887
+ col = chunk.column
888
+ mins = ci.min_values || []
889
+ maxes = ci.max_values || []
890
+ nulls = ci.null_pages || []
891
+ ColumnIndexInfo.new(
892
+ offset: offset, length: length,
893
+ null_pages: nulls,
894
+ min_values: mins.each_with_index.map { |v, i| nulls[i] ? nil : decode_value(v, col) },
895
+ max_values: maxes.each_with_index.map { |v, i| nulls[i] ? nil : decode_value(v, col) },
896
+ boundary_order: Format::BoundaryOrder::NAMES[ci.boundary_order]&.to_s || ci.boundary_order,
897
+ null_counts: ci.null_counts,
898
+ repetition_level_histograms: ci.repetition_level_histograms,
899
+ definition_level_histograms: ci.definition_level_histograms
900
+ )
901
+ rescue Thrift::Error
902
+ nil
903
+ end
904
+
905
+ # Reads and decodes a chunk's OffsetIndex
906
+ # @param chunk [ColumnChunkInfo] chunk whose OffsetIndex to read
907
+ # @return [OffsetIndexInfo, nil] nil when the chunk has none or it fails to decode
908
+ def read_offset_index(chunk)
909
+ offset, length = chunk.offset_index_range
910
+ return nil unless offset && length.positive?
911
+ oi = Format::OffsetIndex.decode(read_at(offset, length)).first
912
+ OffsetIndexInfo.new(
913
+ offset: offset, length: length,
914
+ page_locations: (oi.page_locations || []).map do |l|
915
+ PageLocation.new(offset: l.offset, compressed_page_size: l.compressed_page_size, first_row_index: l.first_row_index)
916
+ end,
917
+ unencoded_byte_array_data_bytes: oi.unencoded_byte_array_data_bytes
918
+ )
919
+ rescue Thrift::Error
920
+ nil
921
+ end
922
+
923
+ # Size in bytes of the bloom filter at +offset+ (its Thrift header plus the bitset)
924
+ # @param offset [Integer] file offset of the BloomFilterHeader
925
+ # @return [Integer, nil] nil when the header can't be decoded or has no num_bytes
926
+ def bloom_filter_size(offset)
927
+ buf = read_at(offset, 64)
928
+ header, size = BloomFilterHeader.decode(buf)
929
+ header.num_bytes ? size + header.num_bytes : nil
930
+ rescue Thrift::Error
931
+ nil
932
+ end
933
+
934
+ # :ok, :mismatch or :absent for one page (reads its body). Sets PageInfo#actual_crc.
935
+ # @param page [PageInfo] page to check
936
+ # @return [Symbol]
937
+ def page_checksum(page)
938
+ return :absent unless page.crc
939
+ page.actual_crc = page_crc(page)
940
+ (page.actual_crc == page.expected_crc) ? :ok : :mismatch
941
+ end
942
+
943
+ # CRC32 of a page's body as stored
944
+ # @param page [PageInfo] page whose compressed body to read
945
+ # @return [Integer] unsigned 32-bit CRC
946
+ def page_crc(page)
947
+ Zlib.crc32(read_at(page.body_offset, page.compressed_size))
948
+ end
949
+
950
+ # See ColumnChunkInfo#index_mismatches
951
+ # @param chunk [ColumnChunkInfo] chunk whose page headers to compare with its ColumnIndex
952
+ # @return [Array<Hash{Symbol => Object}>] one entry per disagreement; a single :page_count
953
+ # entry when the ColumnIndex and the data pages differ in number
954
+ def compare_page_index(chunk)
955
+ ci = chunk.column_index or return []
956
+ data = chunk.data_pages
957
+ if ci.null_pages.size != data.size
958
+ return [{page: nil, data_page: nil, field: :page_count, page_value: data.size, index_value: ci.null_pages.size}]
959
+ end
960
+ order = Inspector.sort_order(chunk.column)
961
+ out = []
962
+ data.each_with_index do |p, k|
963
+ report = ->(field, pv, iv) { out << {page: p.index, data_page: k, field: field, page_value: pv, index_value: iv} }
964
+ idx_nulls = ci.null_counts&.[](k)
965
+ report.call(:null_count, p.num_nulls, idx_nulls) if idx_nulls && p.num_nulls && idx_nulls != p.num_nulls
966
+ st = p.statistics
967
+ # Legacy min/max were computed with another ordering, so they can't be compared
968
+ next if st.nil? || st.caveat || (st.min.nil? && st.max.nil?)
969
+ if ci.null_pages[k]
970
+ report.call(:null_page, "has min/max", "null page")
971
+ next
972
+ end
973
+ imin = ci.min_values[k]
974
+ imax = ci.max_values[k]
975
+ report.call(:min, st.min, imin) if st.min_exact != false && narrower?(imin, st.min, order, :min)
976
+ report.call(:max, st.max, imax) if st.max_exact != false && narrower?(imax, st.max, order, :max)
977
+ end
978
+ out
979
+ end
980
+
981
+ # Decodes a Format::Statistics into Ruby values via the column's type converter
982
+ # Prefers min_value/max_value; falls back to the deprecated min/max, with a caveat when their
983
+ # ordering can't be trusted for the column's type.
984
+ # @param st [Format::Statistics, nil] statistics from column metadata or a page header
985
+ # @param column [Schema::Column] column the statistics describe
986
+ # @return [Stats, nil] nil when +st+ is nil
987
+ def decode_statistics(st, column)
988
+ return nil unless st
989
+ order = Inspector.sort_order(column)
990
+ if st.min_value || st.max_value
991
+ min_raw = st.min_value
992
+ max_raw = st.max_value
993
+ source = "min_value/max_value"
994
+ elsif st.min || st.max
995
+ min_raw = st.min
996
+ max_raw = st.max
997
+ source = "min/max (legacy)"
998
+ binary = column.type == T::BYTE_ARRAY || column.type == T::FIXED_LEN_BYTE_ARRAY
999
+ if order == :unsigned
1000
+ caveat = "legacy min/max were computed with signed comparison, which is wrong for unsigned-ordered " \
1001
+ "values (strings, binary, unsigned integers); most readers ignore them"
1002
+ elsif binary
1003
+ caveat = "legacy min/max of binary-backed values were compared bytewise; most readers ignore them"
1004
+ end
1005
+ end
1006
+ caveat ||= "sort order of #{T::NAMES[column.type]} is undefined; min/max may be meaningless" if order == :unknown && source
1007
+ Stats.new(
1008
+ min: decode_value(min_raw, column),
1009
+ max: decode_value(max_raw, column),
1010
+ null_count: st.null_count,
1011
+ distinct_count: st.distinct_count,
1012
+ min_exact: st.is_min_value_exact,
1013
+ max_exact: st.is_max_value_exact,
1014
+ source: source,
1015
+ caveat: caveat
1016
+ )
1017
+ end
1018
+
1019
+ # Decodes one PLAIN-encoded statistics value (no length prefix for byte arrays)
1020
+ # INT96 decodes to [nanoseconds, julian_day] before conversion. Values of the wrong width,
1021
+ # or that fail to convert, come back as hex (see Inspector.hex).
1022
+ # @param bytes [String, nil] encoded value
1023
+ # @param column [Schema::Column] column whose physical type and converter apply
1024
+ # @return [Object, nil] the converted value, a hex String when it could not be decoded, nil
1025
+ # for nil (or empty BOOLEAN) input
1026
+ def decode_value(bytes, column)
1027
+ return nil if bytes.nil?
1028
+ bytes = bytes.b
1029
+ raw = case column.type
1030
+ when T::BOOLEAN
1031
+ return nil if bytes.empty?
1032
+ bytes.getbyte(0) != 0
1033
+ when T::BYTE_ARRAY, T::FIXED_LEN_BYTE_ARRAY
1034
+ bytes
1035
+ when T::INT96
1036
+ return Inspector.hex(bytes) unless bytes.bytesize == 12
1037
+ bytes.unpack("Q<L<")
1038
+ else
1039
+ width = Encodings::Plain::FORMATS[column.type][1]
1040
+ return Inspector.hex(bytes) unless bytes.bytesize == width
1041
+ Encodings::Plain.decode(bytes, 0, 1, column.type).first.first
1042
+ end
1043
+ conv = column.converter
1044
+ conv ? conv.call(raw) : raw
1045
+ rescue
1046
+ Inspector.hex(bytes)
1047
+ end
1048
+
1049
+ # Decodes the ARROW:schema key/value that Arrow writers (pyarrow, arrow-rs, DuckDB...) store:
1050
+ # base64 of an Arrow IPC message whose header is a flatbuffer Schema (Arrow's Message.fbs and
1051
+ # Schema.fbs). Pure Ruby and read-only; type names follow pyarrow's (str(field.type)).
1052
+ #
1053
+ # Inspector::ArrowSchema.decode(value)
1054
+ # # => { endianness: "little", metadata: {...}, fields: [{ name: "a", type: "int32", nullable: true, ... }] }
1055
+ #
1056
+ # Fields carry :name, :type, :nullable and, when present, :children, :dictionary
1057
+ # ({ index_type:, ordered:, id: }), :extension (ARROW:extension:name) and :metadata.
1058
+ module ArrowSchema
1059
+ # Raised for malformed or unsupported ARROW:schema values
1060
+ class Error < StandardError; end
1061
+
1062
+ # Deepest field nesting accepted, against stack exhaustion on hostile input
1063
+ MAX_DEPTH = 64
1064
+ # Most fields (nested ones included) accepted in one schema
1065
+ MAX_FIELDS = 100_000
1066
+ # Arrow TimeUnit values (SECOND, MILLISECOND, MICROSECOND, NANOSECOND) as pyarrow abbreviates them
1067
+ TIME_UNITS = %w[s ms us ns].freeze
1068
+ # MessageHeader union tag of a Schema in Message.fbs
1069
+ MESSAGE_SCHEMA = 1
1070
+
1071
+ # A minimal flatbuffer reader: tables (through their vtables), scalars, strings, vectors of
1072
+ # scalars and tables, and unions. Every read is bounds-checked; malformed input raises Error.
1073
+ class FlatBuffer
1074
+ # @param bytes [String] the flatbuffer (copied as binary)
1075
+ def initialize(bytes)
1076
+ @b = bytes.b
1077
+ end
1078
+
1079
+ # @return [Table] the root table, whose offset is stored in the first 4 bytes
1080
+ def root = table_at(u32(0))
1081
+
1082
+ # @param pos [Integer] absolute position of a table
1083
+ # @return [Table]
1084
+ # @raise [Error] when its vtable is out of bounds or malformed
1085
+ def table_at(pos) = Table.new(self, pos)
1086
+
1087
+ # @param pos [Integer] absolute start of the read
1088
+ # @param len [Integer] bytes to read
1089
+ # @return [void]
1090
+ # @raise [Error] when the range is not inside the buffer
1091
+ def check(pos, len)
1092
+ return if pos >= 0 && len >= 0 && pos + len <= @b.bytesize
1093
+ raise Error, "flatbuffer read of #{len} bytes at #{pos} is out of bounds (#{@b.bytesize} bytes)"
1094
+ end
1095
+
1096
+ # @param pos [Integer] absolute start of the read
1097
+ # @param len [Integer] bytes to read
1098
+ # @param fmt [String] String#unpack1 directive for the bytes
1099
+ # @return [Integer] the unpacked scalar
1100
+ # @raise [Error] when the range is not inside the buffer
1101
+ def read(pos, len, fmt)
1102
+ check(pos, len)
1103
+ @b.byteslice(pos, len).unpack1(fmt)
1104
+ end
1105
+
1106
+ # @param pos [Integer] absolute position
1107
+ # @return [Integer] unsigned 8-bit value at +pos+
1108
+ def u8(pos) = read(pos, 1, "C")
1109
+
1110
+ # @param pos [Integer] absolute position
1111
+ # @return [Integer] unsigned little-endian 16-bit value at +pos+
1112
+ def u16(pos) = read(pos, 2, "S<")
1113
+
1114
+ # @param pos [Integer] absolute position
1115
+ # @return [Integer] signed little-endian 16-bit value at +pos+
1116
+ def i16(pos) = read(pos, 2, "s<")
1117
+
1118
+ # @param pos [Integer] absolute position
1119
+ # @return [Integer] unsigned little-endian 32-bit value at +pos+
1120
+ def u32(pos) = read(pos, 4, "L<")
1121
+
1122
+ # @param pos [Integer] absolute position
1123
+ # @return [Integer] signed little-endian 32-bit value at +pos+
1124
+ def i32(pos) = read(pos, 4, "l<")
1125
+
1126
+ # @param pos [Integer] absolute position
1127
+ # @return [Integer] signed little-endian 64-bit value at +pos+
1128
+ def i64(pos) = read(pos, 8, "q<")
1129
+
1130
+ # Offsets are relative to where they are stored
1131
+ # @param pos [Integer] absolute position of a uoffset
1132
+ # @return [Integer] absolute position it points to
1133
+ def deref(pos) = pos + u32(pos)
1134
+
1135
+ # @param pos [Integer] absolute position of the string's length prefix
1136
+ # @return [String] the string's bytes as UTF-8 (not validated)
1137
+ def string(pos)
1138
+ len = u32(pos)
1139
+ check(pos + 4, len)
1140
+ @b.byteslice(pos + 4, len).force_encoding(Encoding::UTF_8)
1141
+ end
1142
+
1143
+ # [start, length] of the vector at +pos+ with +size+-byte elements
1144
+ # @param pos [Integer] absolute position of the vector's length prefix
1145
+ # @param size [Integer] bytes per element
1146
+ # @return [Array(Integer, Integer)] start of the first element and the element count
1147
+ # @raise [Error] when the elements do not fit in the buffer
1148
+ def vector(pos, size)
1149
+ len = u32(pos)
1150
+ check(pos + 4, len * size)
1151
+ [pos + 4, len]
1152
+ end
1153
+ end
1154
+
1155
+ # One flatbuffer table; fields are addressed by their slot (declaration order in the .fbs)
1156
+ class Table
1157
+ # @param fb [FlatBuffer] buffer the table lives in
1158
+ # @param pos [Integer] absolute position of the table (where its vtable offset is stored)
1159
+ # @raise [Error] when the vtable is out of bounds or malformed
1160
+ def initialize(fb, pos)
1161
+ @fb = fb
1162
+ @pos = pos
1163
+ @vtable = pos - fb.i32(pos)
1164
+ @vtable_size = fb.u16(@vtable)
1165
+ raise Error, "bad flatbuffer vtable at #{@vtable}" if @vtable_size < 4 || @vtable_size.odd?
1166
+ fb.check(@vtable, @vtable_size)
1167
+ end
1168
+
1169
+ # Absolute position of a field's value, nil when absent
1170
+ # @param slot [Integer] field slot, counting from 0
1171
+ # @return [Integer, nil]
1172
+ def field(slot)
1173
+ o = 4 + (slot * 2)
1174
+ return nil if o + 2 > @vtable_size
1175
+ off = @fb.u16(@vtable + o)
1176
+ off.zero? ? nil : @pos + off
1177
+ end
1178
+
1179
+ # @param slot [Integer] field slot, counting from 0
1180
+ # @param default [Integer] returned when the field is absent (the .fbs default)
1181
+ # @return [Integer] the unsigned 8-bit field (also used for union type tags)
1182
+ def u8(slot, default = 0) = (p = field(slot)) ? @fb.u8(p) : default
1183
+
1184
+ # @param slot [Integer] field slot, counting from 0
1185
+ # @param default [Boolean] returned when the field is absent (the .fbs default)
1186
+ # @return [Boolean]
1187
+ def bool(slot, default = false) = (p = field(slot)) ? @fb.u8(p) != 0 : default
1188
+
1189
+ # @param slot [Integer] field slot, counting from 0
1190
+ # @param default [Integer] returned when the field is absent (the .fbs default)
1191
+ # @return [Integer] the signed 16-bit field (also used for enums such as TimeUnit)
1192
+ def i16(slot, default = 0) = (p = field(slot)) ? @fb.i16(p) : default
1193
+
1194
+ # @param slot [Integer] field slot, counting from 0
1195
+ # @param default [Integer] returned when the field is absent (the .fbs default)
1196
+ # @return [Integer] the signed 32-bit field
1197
+ def i32(slot, default = 0) = (p = field(slot)) ? @fb.i32(p) : default
1198
+
1199
+ # @param slot [Integer] field slot, counting from 0
1200
+ # @param default [Integer] returned when the field is absent (the .fbs default)
1201
+ # @return [Integer] the signed 64-bit field
1202
+ def i64(slot, default = 0) = (p = field(slot)) ? @fb.i64(p) : default
1203
+
1204
+ # @param slot [Integer] field slot, counting from 0
1205
+ # @return [String, nil] the string field, nil when absent
1206
+ def string(slot) = (p = field(slot)) && @fb.string(@fb.deref(p))
1207
+
1208
+ # @param slot [Integer] field slot, counting from 0 (also the value of a union)
1209
+ # @return [Table, nil] the sub-table, nil when absent
1210
+ def table(slot) = (p = field(slot)) && @fb.table_at(@fb.deref(p))
1211
+
1212
+ # @param slot [Integer] field slot, counting from 0
1213
+ # @return [Array<Table>] the vector of tables, empty when absent
1214
+ def tables(slot)
1215
+ p = field(slot) or return []
1216
+ start, len = @fb.vector(@fb.deref(p), 4)
1217
+ Array.new(len) { |i| @fb.table_at(@fb.deref(start + (4 * i))) }
1218
+ end
1219
+
1220
+ # @param slot [Integer] field slot, counting from 0
1221
+ # @return [Array<Integer>] the vector of signed 32-bit values, empty when absent
1222
+ def i32s(slot)
1223
+ p = field(slot) or return []
1224
+ start, len = @fb.vector(@fb.deref(p), 4)
1225
+ Array.new(len) { |i| @fb.i32(start + (4 * i)) }
1226
+ end
1227
+ end
1228
+
1229
+ module_function
1230
+
1231
+ # Decodes the base64 ARROW:schema value; raises ArrowSchema::Error when it can't
1232
+ # @param b64 [String] the key/value metadata value (base64 of an IPC Schema message)
1233
+ # @return [Hash{Symbol => Object}] { endianness:, fields:, metadata: }; +metadata+ is a
1234
+ # Hash{String => String}, left out when the schema has none
1235
+ # @raise [Error] when the value is empty, malformed or not a Schema message
1236
+ def decode(b64)
1237
+ bytes = b64.to_s.unpack1("m")
1238
+ raise Error, "empty value" if bytes.empty?
1239
+ fb = FlatBuffer.new(message_bytes(bytes))
1240
+ message = fb.root
1241
+ header_type = message.u8(1)
1242
+ raise Error, "IPC message holds a #{header_type} header, not a Schema" unless header_type == MESSAGE_SCHEMA
1243
+ schema = message.table(2) or raise Error, "IPC message has no Schema"
1244
+ count = [0]
1245
+ {
1246
+ endianness: schema.i16(0).zero? ? "little" : "big",
1247
+ fields: schema.tables(1).map { |f| field(f, 0, count) },
1248
+ metadata: key_values(schema.tables(2))
1249
+ }.compact
1250
+ rescue ArgumentError, TypeError, RangeError => e
1251
+ raise Error, e.message
1252
+ end
1253
+
1254
+ # The flatbuffer inside an encapsulated IPC message: [0xFFFFFFFF] int32 length, flatbuffer
1255
+ # (the continuation marker is missing in files from before Arrow 0.15)
1256
+ # @param bytes [String] the decoded (binary) ARROW:schema value
1257
+ # @return [String] the message's flatbuffer bytes
1258
+ # @raise [Error] when the length prefix is missing or does not fit
1259
+ def message_bytes(bytes)
1260
+ raise Error, "too short for an IPC message (#{bytes.bytesize} bytes)" if bytes.bytesize < 8
1261
+ len = bytes.unpack1("l<")
1262
+ start = 4
1263
+ if len == -1
1264
+ len = bytes.byteslice(4, 4).unpack1("l<")
1265
+ start = 8
1266
+ end
1267
+ raise Error, "IPC message length #{len} does not fit in #{bytes.bytesize} bytes" if len <= 0 || start + len > bytes.bytesize
1268
+ bytes.byteslice(start, len)
1269
+ end
1270
+
1271
+ # Describes one Arrow Field table and, recursively, its children
1272
+ # @param t [Table] the Field table
1273
+ # @param depth [Integer] nesting depth of the field, 0 at the top level
1274
+ # @param count [Array<Integer>] one-element counter of the fields seen so far, shared across
1275
+ # the recursion
1276
+ # @return [Hash{Symbol => Object}] see the module description for the keys
1277
+ # @raise [Error] past MAX_DEPTH or MAX_FIELDS, or on malformed input
1278
+ def field(t, depth, count)
1279
+ raise Error, "fields nested deeper than #{MAX_DEPTH} levels" if depth > MAX_DEPTH
1280
+ raise Error, "more than #{MAX_FIELDS} fields" if (count[0] += 1) > MAX_FIELDS
1281
+ children = t.tables(5).map { |c| field(c, depth + 1, count) }
1282
+ metadata = key_values(t.tables(6))
1283
+ type = type_name(t.u8(2), t.table(3), children)
1284
+ h = {name: t.string(0).to_s, type: type, nullable: t.bool(1)}
1285
+ if (d = t.table(4))
1286
+ index = d.table(1)
1287
+ index_type = index ? int_name(index) : "int32"
1288
+ ordered = d.bool(2)
1289
+ h[:dictionary] = {index_type: index_type, ordered: ordered, id: d.i64(0)}
1290
+ h[:type] = "dictionary<values=#{type}, indices=#{index_type}, ordered=#{ordered ? 1 : 0}>"
1291
+ end
1292
+ h[:children] = children unless children.empty?
1293
+ if metadata
1294
+ h[:extension] = metadata["ARROW:extension:name"] if metadata["ARROW:extension:name"]
1295
+ h[:metadata] = metadata
1296
+ end
1297
+ h
1298
+ end
1299
+
1300
+ # @param tables [Array<Table>] KeyValue tables (custom_metadata of a Schema or Field)
1301
+ # @return [Hash{String => String}, nil] nil when there are none
1302
+ def key_values(tables)
1303
+ return nil if tables.empty?
1304
+ tables.to_h { |kv| [kv.string(0).to_s, kv.string(1).to_s] }
1305
+ end
1306
+
1307
+ # A child as pyarrow prints it inside a nested type: "name: type" plus " not null"
1308
+ # @param c [Hash{Symbol => Object}] a field as returned by #field
1309
+ # @return [String]
1310
+ def child_text(c) = "#{c[:name]}: #{c[:type]}#{" not null" unless c[:nullable]}"
1311
+
1312
+ # @param t [Table] an Arrow Int table (bitWidth, is_signed)
1313
+ # @return [String] e.g. "int32" or "uint8"
1314
+ def int_name(t) = "#{"u" unless t.bool(1)}int#{t.i32(0)}"
1315
+
1316
+ # @param u [Integer] Arrow TimeUnit value
1317
+ # @return [String] "s", "ms", "us" or "ns" ("unitN" for unknown values)
1318
+ def unit(u) = TIME_UNITS[u] || "unit#{u}"
1319
+
1320
+ # Type names follow Arrow's DataType::ToString (what pyarrow prints)
1321
+ # @param kind [Integer] the Field's Type union tag (Schema.fbs)
1322
+ # @param t [Table, nil] the union's type table, nil when absent
1323
+ # @param children [Array<Hash{Symbol => Object}>] the already described child fields
1324
+ # @return [String] e.g. "timestamp[us, tz=UTC]" or "list<item: string>"
1325
+ # @raise [Error] for a Decimal type without its parameters table
1326
+ def type_name(kind, t, children)
1327
+ case kind
1328
+ when 1 then "null"
1329
+ when 2 then t ? int_name(t) : "int"
1330
+ when 3 then %w[halffloat float double][t ? t.i16(0) : 0] || "float?"
1331
+ when 4 then "binary"
1332
+ when 5 then "string"
1333
+ when 6 then "bool"
1334
+ when 7
1335
+ raise Error, "decimal type without parameters" unless t
1336
+ bits = t.i32(2, 128)
1337
+ "decimal#{bits}(#{t.i32(0)}, #{t.i32(1)})"
1338
+ when 8 then t&.i16(0, 1)&.zero? ? "date32[day]" : "date64[ms]"
1339
+ when 9
1340
+ bits = t ? t.i32(1, 32) : 32
1341
+ "time#{bits}[#{unit(t ? t.i16(0, 1) : 1)}]"
1342
+ when 10
1343
+ tz = t&.string(1)
1344
+ "timestamp[#{unit(t ? t.i16(0) : 0)}#{", tz=#{tz}" if tz}]"
1345
+ when 11 then %w[month_interval day_time_interval month_day_nano_interval][t ? t.i16(0) : 0] || "interval"
1346
+ when 12 then "list<#{children.map { |c| child_text(c) }.join(", ")}>"
1347
+ when 13 then "struct<#{children.map { |c| child_text(c) }.join(", ")}>"
1348
+ when 14
1349
+ mode = t&.i16(0)&.positive? ? "dense" : "sparse"
1350
+ ids = t ? t.i32s(1) : []
1351
+ members = children.each_with_index.map { |c, i| "#{child_text(c)}=#{ids[i] || i}" }
1352
+ "#{mode}_union<#{members.join(", ")}>"
1353
+ when 15 then "fixed_size_binary[#{t ? t.i32(0) : 0}]"
1354
+ when 16 then "fixed_size_list<#{children.map { |c| child_text(c) }.join(", ")}>[#{t ? t.i32(0) : 0}]"
1355
+ when 17 then map_name(t, children)
1356
+ when 18 then "duration[#{unit(t ? t.i16(0, 1) : 1)}]"
1357
+ when 19 then "large_binary"
1358
+ when 20 then "large_string"
1359
+ when 21 then "large_list<#{children.map { |c| child_text(c) }.join(", ")}>"
1360
+ when 22 then "run_end_encoded<#{children.map { |c| "#{(c[:name] == "values") ? "values" : "run_ends"}: #{c[:type]}" }.join(", ")}>"
1361
+ when 23 then "binary_view"
1362
+ when 24 then "string_view"
1363
+ when 25 then "list_view<#{children.map { |c| child_text(c) }.join(", ")}>"
1364
+ when 26 then "large_list_view<#{children.map { |c| child_text(c) }.join(", ")}>"
1365
+ else "unknown type #{kind}"
1366
+ end
1367
+ end
1368
+
1369
+ # map<key, value> with non-standard field names in parentheses, as Arrow prints it
1370
+ # @param t [Table, nil] the Map table (keysSorted), nil when absent
1371
+ # @param children [Array<Hash{Symbol => Object}>] the map's single entries struct field
1372
+ # @return [String]
1373
+ def map_name(t, children)
1374
+ entries = children.first
1375
+ kv = entries && entries[:children] || []
1376
+ named = ->(f, std) { f ? "#{f[:type]}#{" ('#{f[:name]}')" unless f[:name] == std}" : "?" }
1377
+ sorted = t&.bool(0) ? ", keys_sorted" : ""
1378
+ entries_name = (entries && entries[:name] != "entries") ? " ('#{entries[:name]}')" : ""
1379
+ "map<#{named.call(kv[0], "key")}, #{named.call(kv[1], "value")}#{sorted}#{entries_name}>"
1380
+ end
1381
+
1382
+ # "name: type" lines for a field and its children, indented, for text output
1383
+ # Children nested deeper than 8 levels are left out.
1384
+ # @param fields [Array<Hash{Symbol => Object}>] fields as returned by #decode under +:fields+
1385
+ # @param depth [Integer] indentation level of +fields+
1386
+ # @param out [Array<String>] accumulator the lines are appended to
1387
+ # @return [Array<String>] +out+
1388
+ def lines(fields, depth = 0, out = [])
1389
+ fields.each do |f|
1390
+ notes = []
1391
+ notes << "not null" unless f[:nullable]
1392
+ notes << "extension #{f[:extension]}" if f[:extension]
1393
+ meta = (f[:metadata] || {}).reject { |k, _| k.start_with?("ARROW:extension:") }
1394
+ notes << "metadata #{meta.map { |k, v| "#{k}=#{v.to_s[0, 60].inspect}" }.join(", ")}" unless meta.empty?
1395
+ out << "#{" " * depth}#{f[:name]}: #{f[:type]}#{" (#{notes.join("; ")})" unless notes.empty?}"
1396
+ lines(f[:children], depth + 1, out) if f[:children] && depth < 8
1397
+ end
1398
+ out
1399
+ end
1400
+ end
1401
+
1402
+ # ---- class helpers ----
1403
+
1404
+ # @param e [Integer] Parquet Encoding value
1405
+ # @return [String] its name, e.g. "RLE_DICTIONARY" (the number as a String when unknown)
1406
+ def self.encoding_name(e) = Format::Encoding::NAMES[e]&.to_s || e.to_s
1407
+
1408
+ # @param column [Schema::Column] leaf column
1409
+ # @return [String] physical type with its logical annotation, e.g. "BYTE_ARRAY STRING" or
1410
+ # "FIXED_LEN_BYTE_ARRAY(16) UUID"
1411
+ def self.type_name(column)
1412
+ node = column.node
1413
+ phys = T::NAMES[node.type].to_s
1414
+ phys += "(#{node.type_length})" if node.type == T::FIXED_LEN_BYTE_ARRAY && node.type_length
1415
+ logical = logical_type_name(node)
1416
+ logical ? "#{phys} #{logical}" : phys
1417
+ end
1418
+
1419
+ # The node's LogicalType, else its ConvertedType, as text
1420
+ # @param node [Schema::Node] schema node (leaf or group)
1421
+ # @return [String, nil] e.g. "INTEGER(8, unsigned)", "TIMESTAMP(MICROS, UTC)" or "DECIMAL(10, 2)";
1422
+ # nil when the node has no annotation
1423
+ def self.logical_type_name(node)
1424
+ if (kind = node.logical_type&.kind)
1425
+ name, payload = kind
1426
+ case name
1427
+ when :integer then "INTEGER(#{payload.bit_width}, #{payload.is_signed ? "signed" : "unsigned"})"
1428
+ when :decimal then "DECIMAL(#{payload.precision}, #{payload.scale})"
1429
+ when :timestamp, :time
1430
+ "#{name.upcase}(#{payload.unit&.to_sym&.upcase}, #{payload.is_adjusted_to_utc ? "UTC" : "local"})"
1431
+ when :unknown then "NULL"
1432
+ else name.to_s.upcase
1433
+ end
1434
+ elsif node.converted_type
1435
+ c = Format::ConvertedType::NAMES[node.converted_type].to_s
1436
+ (c == "DECIMAL") ? "DECIMAL(#{node.precision}, #{node.scale || 0})" : c
1437
+ end
1438
+ end
1439
+
1440
+ # :signed, :unsigned or :unknown, per the Parquet sort order rules for the column's type
1441
+ # @param column [Schema::Column] leaf column
1442
+ # @return [Symbol]
1443
+ def self.sort_order(column)
1444
+ kind, _a, signed = Types.logical_of(column.node)
1445
+ case kind
1446
+ when :integer then (signed == false) ? :unsigned : :signed
1447
+ when :decimal, :date, :time, :timestamp, :float16 then :signed
1448
+ when :string, :enum, :json, :bson, :uuid then :unsigned
1449
+ else
1450
+ case column.type
1451
+ when T::BYTE_ARRAY, T::FIXED_LEN_BYTE_ARRAY then :unsigned
1452
+ when T::INT96 then :unknown
1453
+ else :signed
1454
+ end
1455
+ end
1456
+ end
1457
+
1458
+ # Converts Ruby values (Time, BigDecimal, binary Strings, non-finite Floats...) to JSON-safe ones
1459
+ # Recurses into Hashes, Arrays and Structs; Hash keys other than Symbols become Strings.
1460
+ # @param v [Object] value to convert
1461
+ # @return [Hash, Array, String, Integer, Float, Boolean, nil]
1462
+ def self.jsonable(v)
1463
+ case v
1464
+ when Hash then v.each_with_object({}) { |(k, x), h| h[k.is_a?(Symbol) ? k : k.to_s] = jsonable(x) }
1465
+ when Array then v.map { |x| jsonable(x) }
1466
+ when Struct then v.respond_to?(:to_h) ? jsonable(v.to_h) : v.to_s
1467
+ when String then text(v)
1468
+ when Symbol then v.to_s
1469
+ when Float then v.finite? ? v : v.to_s
1470
+ when Time then v.utc.strftime(v.nsec.zero? ? "%Y-%m-%dT%H:%M:%SZ" : "%Y-%m-%dT%H:%M:%S.%NZ")
1471
+ when Date then v.iso8601
1472
+ when BigDecimal then v.to_s("F")
1473
+ when Integer, true, false, nil then v
1474
+ else v.to_s
1475
+ end
1476
+ end
1477
+
1478
+ # A String as readable text when it is valid UTF-8 without control characters, else as hex
1479
+ # Strings already tagged as valid UTF-8 are returned as they are, control characters and all.
1480
+ # @param s [String] string in any encoding
1481
+ # @return [String]
1482
+ def self.text(s)
1483
+ return s if s.encoding == Encoding::UTF_8 && s.valid_encoding?
1484
+ u = s.dup.force_encoding(Encoding::UTF_8)
1485
+ return u if u.valid_encoding? && !u.match?(/[\x00-\x08\x0e-\x1f\x7f]/)
1486
+ hex(s)
1487
+ end
1488
+
1489
+ # @param bytes [String] bytes to show
1490
+ # @return [String] "0x" and lowercase hex digits; only the first 64 bytes, followed by the
1491
+ # total size, for longer input
1492
+ def self.hex(bytes)
1493
+ b = bytes.b
1494
+ (b.bytesize > 64) ? "0x#{b.byteslice(0, 64).unpack1("H*")}… (#{b.bytesize} bytes)" : "0x#{b.unpack1("H*")}"
1495
+ end
1496
+
1497
+ # A short display form of a decoded value
1498
+ # Strings are quoted (binary ones go through Inspector.text first), nil shows as "null".
1499
+ # @param v [Object] decoded value
1500
+ # @param max [Integer] longest result, in characters; longer ones are cut and end in an ellipsis
1501
+ # @return [String]
1502
+ def self.display(v, max: 40)
1503
+ s = case v
1504
+ when String then (v.encoding == Encoding::BINARY) ? text(v) : v
1505
+ when nil then "null"
1506
+ else jsonable(v).to_s
1507
+ end
1508
+ s = s.inspect if v.is_a?(String)
1509
+ (s.size > max) ? "#{s[0, max - 1]}…" : s
1510
+ end
1511
+
1512
+ # @param n [Integer, nil] byte count
1513
+ # @return [String] e.g. "512 B", "1.50 KB" or "12.3 MB" (binary units); "?" for nil
1514
+ def self.human_bytes(n)
1515
+ return "?" unless n
1516
+ units = %w[B KB MB GB TB]
1517
+ f = n.to_f
1518
+ i = 0
1519
+ while f >= 1024 && i < units.size - 1
1520
+ f /= 1024
1521
+ i += 1
1522
+ end
1523
+ if i.zero?
1524
+ "#{n} B"
1525
+ else
1526
+ format("%.#{(f < 10) ? 2 : 1}f %s", f, units[i])
1527
+ end
1528
+ end
1529
+
1530
+ private
1531
+
1532
+ # @param uncompressed [Integer, nil] uncompressed byte count
1533
+ # @param compressed [Integer, nil] compressed byte count
1534
+ # @return [String] " (3.21x)" for #report, empty when the ratio is unknown
1535
+ def ratio_text(uncompressed, compressed)
1536
+ (compressed.to_i.positive? && uncompressed) ? format(" (%.2fx)", uncompressed.to_f / compressed) : ""
1537
+ end
1538
+
1539
+ # @param page [PageInfo] page whose CRC status to show
1540
+ # @return [String] the status for a #report page line: " crc ok", " CRC MISMATCH", " crc"
1541
+ # (has a CRC, not verified) or empty
1542
+ def crc_text(page)
1543
+ case page.checksum
1544
+ when :ok then " crc ok"
1545
+ when :mismatch then " CRC MISMATCH"
1546
+ else page.crc ? " crc" : ""
1547
+ end
1548
+ end
1549
+
1550
+ # @param m [Hash{Symbol => Object}] an entry of #index_mismatches
1551
+ # @return [String] one #report line describing it
1552
+ def index_mismatch_text(m)
1553
+ where = "row group #{m[:row_group]} #{m[:column]}"
1554
+ if m[:field] == :page_count
1555
+ "#{where}: #{m[:page_value]} data pages but #{m[:index_value]} column index entries"
1556
+ else
1557
+ "#{where} page #{m[:page]}: #{m[:field]} in page header #{Inspector.display(m[:page_value])}, " \
1558
+ "in column index #{Inspector.display(m[:index_value])}"
1559
+ end
1560
+ end
1561
+
1562
+ # Adds :arrow_type to schema nodes with a same-named Arrow field (top level, and struct members)
1563
+ # @param nodes [Array<Hash{Symbol => Object}>] #schema_tree nodes, updated in place
1564
+ # @param fields [Array<Hash{Symbol => Object}>] Arrow fields at the same level
1565
+ # @param depth [Integer] nesting depth; recursion stops at 32
1566
+ # @return [void]
1567
+ def annotate_arrow_types(nodes, fields, depth = 0)
1568
+ by_name = fields.to_h { |f| [f[:name], f] }
1569
+ nodes.each do |n|
1570
+ f = by_name[n[:name]] or next
1571
+ n[:arrow_type] = f[:type]
1572
+ if n[:children] && f[:children] && f[:type].start_with?("struct<") && depth < 32
1573
+ annotate_arrow_types(n[:children], f[:children], depth + 1)
1574
+ end
1575
+ end
1576
+ end
1577
+
1578
+ # Whether the index bound +idx+ excludes values the page's bound +page+ says are present
1579
+ # (index min above the page min, or index max below the page max). Truncated binary bounds
1580
+ # (one a prefix of the other) and values that can't be compared are never reported.
1581
+ # @param idx [Object, nil] decoded ColumnIndex bound
1582
+ # @param page [Object, nil] decoded page header bound
1583
+ # @param order [Symbol] :signed, :unsigned or :unknown (see Inspector.sort_order)
1584
+ # @param which [Symbol] :min or :max
1585
+ # @return [Boolean]
1586
+ def narrower?(idx, page, order, which)
1587
+ return false if idx.nil? || page.nil? || order == :unknown
1588
+ a, b = (which == :min) ? [page, idx] : [idx, page] # true when a < b
1589
+ if a.is_a?(String) && b.is_a?(String)
1590
+ a = a.b
1591
+ b = b.b
1592
+ return false if a.start_with?(b) || b.start_with?(a)
1593
+ return (order == :unsigned) ? a < b : false
1594
+ end
1595
+ return false if a.is_a?(Float) && a.nan? || b.is_a?(Float) && b.nan?
1596
+ a = a ? 1 : 0 if a == true || a == false
1597
+ b = b ? 1 : 0 if b == true || b == false
1598
+ (a <=> b) == -1
1599
+ rescue
1600
+ false
1601
+ end
1602
+
1603
+ # Overall min or max of per-chunk bounds, when they can be compared
1604
+ # @param values [Array<Object, nil>] decoded per-chunk minimums or maximums
1605
+ # @param which [Symbol] :min or :max
1606
+ # @return [Object, nil] nil when there are no values or they are of mixed or incomparable types
1607
+ def safe_extreme(values, which)
1608
+ vals = values.compact
1609
+ return nil if vals.empty?
1610
+ if vals.all? { |v| v == true || v == false }
1611
+ return (which == :min) ? vals.all? : vals.any?
1612
+ end
1613
+ return nil unless vals.map(&:class).uniq.size == 1
1614
+ (which == :min) ? vals.min : vals.max
1615
+ rescue ArgumentError, NoMethodError
1616
+ nil
1617
+ end
1618
+
1619
+ # Reads the file size, the footer length and magic, and decodes the FileMetaData into @metadata
1620
+ # @return [void]
1621
+ # @raise [FormatError] when the file is too small, lacks the magic bytes or has a corrupt footer
1622
+ # @raise [UnsupportedError] when the file is encrypted (PARE magic)
1623
+ def read_footer
1624
+ @io.seek(0, IO::SEEK_END)
1625
+ @file_size = @io.pos
1626
+ raise FormatError, "File too small to be Parquet (#{@file_size} bytes)" if @file_size < 12
1627
+ tail = read_at(@file_size - 8, 8)
1628
+ raise UnsupportedError, "Encrypted Parquet files are not supported" if tail.byteslice(4, 4) == "PARE"
1629
+ raise FormatError, "Missing PAR1 footer magic" unless tail.byteslice(4, 4) == MAGIC
1630
+ @footer_size = tail.unpack1("V")
1631
+ raise FormatError, "Footer length #{@footer_size} exceeds file size" if @footer_size + 12 > @file_size
1632
+ @metadata = Format::FileMetaData.decode(read_at(footer_offset, @footer_size)).first
1633
+ rescue Thrift::Error => e
1634
+ raise FormatError, "Corrupt file metadata: #{e.message}"
1635
+ end
1636
+
1637
+ # Reads +len+ bytes at +pos+ through a small read-ahead window, so walking many small pages
1638
+ # does not cost a syscall per header
1639
+ # @param pos [Integer] file offset
1640
+ # @param len [Integer] bytes wanted
1641
+ # @return [String] binary String; shorter than +len+ at the end of the file
1642
+ def read_at(pos, len)
1643
+ if @window && pos >= @window_pos && pos + len <= @window_pos + @window.bytesize
1644
+ return @window.byteslice(pos - @window_pos, len)
1645
+ end
1646
+ @io.seek(pos)
1647
+ if len >= WINDOW
1648
+ (@io.read(len) || "".b).b
1649
+ else
1650
+ @window_pos = pos
1651
+ @window = (@io.read(WINDOW) || "".b).b
1652
+ @window.byteslice(0, len)
1653
+ end
1654
+ end
1655
+
1656
+ # Decodes the page header at +pos+, reading more bytes when it is larger than the first guess
1657
+ # (page statistics of long strings can make headers big)
1658
+ # @param pos [Integer] file offset of the header
1659
+ # @param limit [Integer] offset the header must not extend past (the footer's start)
1660
+ # @return [Array(Format::PageHeader, Integer)] the header and its encoded size in bytes
1661
+ # @raise [FormatError] when there is no room for a header before +limit+
1662
+ # @raise [Thrift::Error] when it does not decode within +limit+ or 16 MiB
1663
+ def read_page_header(pos, limit)
1664
+ want = 256
1665
+ while true
1666
+ avail = [want, limit - pos].min
1667
+ raise FormatError, "no room for a page header" if avail <= 0
1668
+ buf = read_at(pos, avail)
1669
+ begin
1670
+ header, size = Format::PageHeader.decode(buf)
1671
+ return [header, size]
1672
+ rescue Thrift::Error
1673
+ raise if avail < want || want >= 16 * 1024 * 1024
1674
+ want *= 8
1675
+ end
1676
+ end
1677
+ end
1678
+
1679
+ # Builds a PageInfo from a decoded page header
1680
+ # @param index [Integer] position of the page in its chunk
1681
+ # @param h [Format::PageHeader] decoded header
1682
+ # @param pos [Integer] file offset of the header
1683
+ # @param header_size [Integer] encoded size of the header in bytes
1684
+ # @param column [Schema::Column] column the page belongs to (for statistics and row counts)
1685
+ # @return [PageInfo]
1686
+ # @raise [FormatError] when the header declares a negative compressed size
1687
+ def page_info(index, h, pos, header_size, column)
1688
+ info = PageInfo.new(
1689
+ index: index,
1690
+ type: Format::PageType::NAMES[h.type] || h.type,
1691
+ offset: pos,
1692
+ header_size: header_size,
1693
+ compressed_size: h.compressed_page_size.to_i,
1694
+ uncompressed_size: h.uncompressed_page_size.to_i,
1695
+ crc: h.crc
1696
+ )
1697
+ raise FormatError, "negative page size #{info.compressed_size}" if info.compressed_size.negative?
1698
+ if (d = h.data_page_header)
1699
+ info.num_values = d.num_values
1700
+ info.encoding = Inspector.encoding_name(d.encoding)
1701
+ info.definition_level_encoding = Inspector.encoding_name(d.definition_level_encoding) if d.definition_level_encoding
1702
+ info.repetition_level_encoding = Inspector.encoding_name(d.repetition_level_encoding) if d.repetition_level_encoding
1703
+ info.statistics = decode_statistics(d.statistics, column)
1704
+ info.num_nulls = info.statistics&.null_count
1705
+ # Without repetition every value is a row
1706
+ info.num_rows = d.num_values if column.max_repetition_level.zero?
1707
+ elsif (d = h.data_page_header_v2)
1708
+ info.num_values = d.num_values
1709
+ info.num_nulls = d.num_nulls
1710
+ info.num_rows = d.num_rows
1711
+ info.encoding = Inspector.encoding_name(d.encoding)
1712
+ info.definition_levels_byte_length = d.definition_levels_byte_length
1713
+ info.repetition_levels_byte_length = d.repetition_levels_byte_length
1714
+ info.is_compressed = d.is_compressed.nil? || d.is_compressed
1715
+ info.statistics = decode_statistics(d.statistics, column)
1716
+ elsif (d = h.dictionary_page_header)
1717
+ info.num_values = d.num_values
1718
+ info.encoding = Inspector.encoding_name(d.encoding)
1719
+ info.is_sorted = d.is_sorted
1720
+ end
1721
+ info
1722
+ end
1723
+
1724
+ # One #key_value_metadata entry
1725
+ # @param key [String] metadata key
1726
+ # @param value [String, nil] metadata value
1727
+ # @return [Hash{Symbol => Object}] see #key_value_metadata
1728
+ def describe_key_value(key, value)
1729
+ value = value.to_s
1730
+ h = {key: key, bytesize: value.bytesize}
1731
+ if key == "ARROW:schema"
1732
+ h[:format] = "arrow_schema"
1733
+ h[:value] = (value.size > 120) ? "#{value[0, 120]}…" : value
1734
+ if (arrow = arrow_schema)
1735
+ h[:summary] = "Arrow schema, #{arrow[:fields].size} field#{"s" unless arrow[:fields].size == 1}"
1736
+ h[:arrow_fields] = arrow[:fields].map { |f| f[:name] }
1737
+ meta = arrow[:metadata]&.transform_values { |v| (v.size > 4000) ? "#{v[0, 4000]}…" : v }
1738
+ h[:arrow_schema] = arrow.merge(metadata: meta).compact
1739
+ else
1740
+ h[:summary] = "Arrow IPC schema message, base64-encoded (#{value.bytesize} bytes)"
1741
+ h[:arrow_error] = arrow_schema_error
1742
+ names = arrow_field_names(value)
1743
+ h[:arrow_fields] = names if names
1744
+ end
1745
+ elsif value.lstrip.start_with?("{", "[") && value.bytesize < 4 * 1024 * 1024
1746
+ begin
1747
+ h[:json] = JSON.parse(value)
1748
+ h[:format] = "json"
1749
+ h[:summary] = (key == "pandas") ? pandas_summary(h[:json]) : "JSON"
1750
+ rescue JSON::ParserError
1751
+ h[:format] = "text"
1752
+ end
1753
+ h[:value] = (value.size > 4000) ? "#{value[0, 4000]}…" : value
1754
+ else
1755
+ t = Inspector.text(value.b)
1756
+ h[:format] = (t.start_with?("0x") && value.bytesize.positive?) ? "binary" : "text"
1757
+ h[:value] = (t.size > 4000) ? "#{t[0, 4000]}…" : t
1758
+ end
1759
+ h
1760
+ end
1761
+
1762
+ # @param json [Object] the parsed "pandas" metadata value
1763
+ # @return [String] e.g. "pandas 2.2.0 metadata, 3 columns"
1764
+ def pandas_summary(json)
1765
+ return "pandas metadata" unless json.is_a?(Hash)
1766
+ cols = json["columns"]&.size
1767
+ "pandas #{json["pandas_version"]} metadata, #{cols || "?"} columns"
1768
+ end
1769
+
1770
+ # Field names from an Arrow IPC schema message, found by scanning for its flatbuffer strings.
1771
+ # Best effort: only used to label the blob, nil when nothing sensible is found.
1772
+ # Only top-level Parquet column names that occur in the decoded bytes are reported.
1773
+ # @param b64 [String] the base64 ARROW:schema value
1774
+ # @return [Array<String>, nil]
1775
+ def arrow_field_names(b64)
1776
+ bytes = b64.unpack1("m")
1777
+ schema_cols = columns.map { |c| c.path.first }.uniq
1778
+ found = schema_cols.select { |n| bytes.include?(n.b) }
1779
+ found.empty? ? nil : found
1780
+ rescue ArgumentError
1781
+ nil
1782
+ end
1783
+ end
1784
+ end