herringbone 0.1.0 → 0.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,1388 @@
1
+ # frozen_string_literal: true
2
+
3
+ require "json"
4
+ require "zlib"
5
+
6
+ module Herringbone
7
+ # Examines a Parquet file using only its footer, page headers and page indexes. Values are
8
+ # never decompressed or decoded, so this works for files whose codecs are not installed and
9
+ # stays fast for big files (it seeks from page header to page header).
10
+ #
11
+ # File.open("data.parquet", "rb") do |io|
12
+ # inspector = Herringbone::Inspector.new(io)
13
+ # inspector.summary # => { file_size:, num_rows:, codecs:, ... }
14
+ # inspector.row_groups[0].column("name").pages
15
+ # inspector.to_h # everything, JSON-serializable
16
+ # puts inspector.report # readable text (bin/herringbone inspect)
17
+ # html = inspector.to_html # self-contained HTML page (see Visualizer)
18
+ # end
19
+ #
20
+ # Page headers and indexes are read lazily; #load_all reads them all up front, after which the
21
+ # inspector no longer needs the IO. The IO is never closed. After #verify_checksums, page CRC
22
+ # results are included in every output.
23
+ class Inspector
24
+ T = Format::Type
25
+ MAGIC = "PAR1"
26
+ WINDOW = 64 * 1024
27
+
28
+ # Just enough of parquet.thrift's BloomFilterHeader to learn the filter's size
29
+ class BloomFilterHeader < Thrift::Struct
30
+ field 1, :num_bytes, :i32
31
+ end
32
+
33
+ # Decoded min/max statistics (from a column chunk, a page header or a page index entry)
34
+ Stats = Struct.new(:min, :max, :null_count, :distinct_count, :min_exact, :max_exact, :source, :caveat,
35
+ keyword_init: true) do
36
+ def to_h = Inspector.jsonable(super.compact)
37
+ end
38
+
39
+ # One page header of a column chunk. +checksum+ is nil until the CRCs are verified
40
+ # (Inspector#verify_checksums), then :ok, :mismatch or :absent (the page has no CRC).
41
+ PageInfo = Struct.new(:index, :type, :offset, :header_size, :compressed_size, :uncompressed_size,
42
+ :num_values, :num_nulls, :num_rows, :first_row_index, :encoding, :definition_level_encoding,
43
+ :repetition_level_encoding, :definition_levels_byte_length, :repetition_levels_byte_length,
44
+ :is_compressed, :is_sorted, :statistics, :crc, :checksum, keyword_init: true) do
45
+ def total_size = header_size + compressed_size
46
+ def end_offset = offset + total_size
47
+ def body_offset = offset + header_size
48
+ def dictionary? = type == :DICTIONARY_PAGE
49
+ def data? = type == :DATA_PAGE || type == :DATA_PAGE_V2
50
+
51
+ # The CRC from the header as an unsigned 32-bit value (Thrift stores it as a signed i32)
52
+ def expected_crc = crc && (crc & 0xFFFF_FFFF)
53
+
54
+ def to_h
55
+ h = super
56
+ h[:statistics] = statistics&.to_h
57
+ h[:crc] = !crc.nil?
58
+ Inspector.jsonable(h.compact)
59
+ end
60
+ end
61
+
62
+ # ColumnIndex of one column chunk, with min/max decoded per page (nil for all-null pages)
63
+ ColumnIndexInfo = Struct.new(:offset, :length, :null_pages, :min_values, :max_values, :boundary_order,
64
+ :null_counts, :repetition_level_histograms, :definition_level_histograms, keyword_init: true) do
65
+ def to_h = Inspector.jsonable(super.compact)
66
+ end
67
+
68
+ PageLocation = Struct.new(:offset, :compressed_page_size, :first_row_index, keyword_init: true)
69
+
70
+ OffsetIndexInfo = Struct.new(:offset, :length, :page_locations, :unencoded_byte_array_data_bytes,
71
+ keyword_init: true) do
72
+ def to_h
73
+ h = super
74
+ h[:page_locations] = page_locations.map(&:to_h)
75
+ h.compact
76
+ end
77
+ end
78
+
79
+ # One column chunk of a row group
80
+ class ColumnChunkInfo
81
+ attr_reader :inspector, :row_group, :column, :chunk, :meta, :error
82
+
83
+ def initialize(inspector, row_group, column, chunk)
84
+ @inspector = inspector
85
+ @row_group = row_group
86
+ @column = column
87
+ @chunk = chunk
88
+ @meta = chunk.meta_data
89
+ end
90
+
91
+ def path = @column.dotted_path
92
+ def column_index_number = @column.index
93
+ def codec = Format::Codec::NAMES[@meta.codec] || @meta.codec
94
+ def encodings = (@meta.encodings || []).map { |e| Inspector.encoding_name(e) }
95
+ def num_values = @meta.num_values
96
+ def compressed_size = @meta.total_compressed_size
97
+ def uncompressed_size = @meta.total_uncompressed_size
98
+ def data_page_offset = @meta.data_page_offset
99
+ def external_file = @chunk.file_path
100
+
101
+ def compression_ratio
102
+ compressed_size.to_i.positive? ? uncompressed_size.to_f / compressed_size : nil
103
+ end
104
+
105
+ # Some writers store 0 when there is no dictionary page, and some store a data_page_offset
106
+ # of 0 for empty chunks (which would point at the magic bytes)
107
+ def dictionary_page_offset
108
+ d = @meta.dictionary_page_offset
109
+ d && d >= 4 && (d < @meta.data_page_offset.to_i || @meta.data_page_offset.to_i < 4) ? d : nil
110
+ end
111
+
112
+ # Where the chunk's first page starts
113
+ def start_offset = dictionary_page_offset || data_page_offset
114
+
115
+ # The end according to the metadata; some writers under-report it (see #end_offset)
116
+ def declared_end_offset = start_offset + compressed_size
117
+
118
+ # The end of the last page actually found (the declared end if the pages could not be walked)
119
+ def end_offset
120
+ last = pages.last
121
+ last ? [last.end_offset, declared_end_offset].max : declared_end_offset
122
+ end
123
+
124
+ def encoding_stats
125
+ (@meta.encoding_stats || []).map do |s|
126
+ { page_type: Format::PageType::NAMES[s.page_type] || s.page_type,
127
+ encoding: Inspector.encoding_name(s.encoding), count: s.count }
128
+ end
129
+ end
130
+
131
+ def statistics
132
+ return @statistics if defined?(@statistics)
133
+ @statistics = @inspector.decode_statistics(@meta.statistics, @column)
134
+ end
135
+
136
+ def size_statistics
137
+ s = @meta.size_statistics
138
+ s&.to_h
139
+ end
140
+
141
+ def key_value_metadata
142
+ (@meta.key_value_metadata || []).to_h { |kv| [kv.key, kv.value] }
143
+ end
144
+
145
+ def bloom_filter_offset = @meta.bloom_filter_offset
146
+
147
+ # Bytes taken by the bloom filter (header + bitset); read from its header when the footer has no length
148
+ def bloom_filter_length
149
+ return nil unless bloom_filter_offset
150
+ return @meta.bloom_filter_length if @meta.bloom_filter_length
151
+ return @bloom_filter_length if defined?(@bloom_filter_length)
152
+ @bloom_filter_length = @inspector.bloom_filter_size(bloom_filter_offset)
153
+ end
154
+
155
+ def column_index_range
156
+ o = @chunk.column_index_offset
157
+ o && @chunk.column_index_length ? [o, @chunk.column_index_length] : nil
158
+ end
159
+
160
+ def offset_index_range
161
+ o = @chunk.offset_index_offset
162
+ o && @chunk.offset_index_length ? [o, @chunk.offset_index_length] : nil
163
+ end
164
+
165
+ # All page headers, in file order. Errors while walking (corrupt or truncated headers) end
166
+ # the walk; #error then says what went wrong and #pages holds the pages found before it.
167
+ def pages
168
+ @pages ||= begin
169
+ list, @error = @inspector.walk_pages(self)
170
+ apply_offset_index(list)
171
+ list
172
+ end
173
+ end
174
+
175
+ def data_pages = pages.select(&:data?)
176
+ def dictionary_page = pages.find(&:dictionary?)
177
+
178
+ # Number of entries in the dictionary (from its page header)
179
+ def dictionary_size = dictionary_page&.num_values
180
+
181
+ def column_index
182
+ return @column_index if defined?(@column_index)
183
+ @column_index = @inspector.read_column_index(self)
184
+ end
185
+
186
+ def offset_index
187
+ return @offset_index if defined?(@offset_index)
188
+ @offset_index = @inspector.read_offset_index(self)
189
+ end
190
+
191
+ def null_count
192
+ return statistics.null_count if statistics&.null_count
193
+ counts = data_pages.map(&:num_nulls)
194
+ counts.all? ? counts.sum : nil
195
+ end
196
+
197
+ # Reads every page body (compressed bytes, as stored) and checks it against the CRC in its
198
+ # page header. Sets PageInfo#checksum on each page and returns the pages' statuses.
199
+ def verify_checksums
200
+ pages.map { |p| p.checksum = @inspector.page_checksum(p) }
201
+ end
202
+
203
+ # Page statistics (from page headers) that disagree with the ColumnIndex entry for the
204
+ # same page. Each: { page:, data_page:, field:, page_value:, index_value: } where +page+
205
+ # is the page's position in #pages, +data_page+ its position among the data pages (and
206
+ # in the ColumnIndex), +field+ one of :min, :max, :null_count, :null_page, :page_count.
207
+ # A ColumnIndex bound that is wider than the page's (e.g. a truncated string prefix) is
208
+ # allowed; one that is narrower, or a differing null count, is reported. Empty when the
209
+ # chunk has no ColumnIndex or its pages carry no statistics.
210
+ def index_mismatches
211
+ @index_mismatches ||= @inspector.compare_page_index(self)
212
+ end
213
+
214
+ def to_h
215
+ Inspector.jsonable({
216
+ path: path,
217
+ column: @column.index,
218
+ type: Inspector.type_name(@column),
219
+ codec: codec,
220
+ encodings: encodings,
221
+ encoding_stats: encoding_stats,
222
+ num_values: num_values,
223
+ null_count: null_count,
224
+ compressed_size: compressed_size,
225
+ uncompressed_size: uncompressed_size,
226
+ compression_ratio: compression_ratio&.round(4),
227
+ file_offset: @chunk.file_offset,
228
+ start_offset: start_offset,
229
+ end_offset: end_offset,
230
+ data_page_offset: data_page_offset,
231
+ dictionary_page_offset: dictionary_page_offset,
232
+ dictionary: dictionary_page && { offset: dictionary_page.offset, num_values: dictionary_page.num_values,
233
+ compressed_size: dictionary_page.compressed_size, uncompressed_size: dictionary_page.uncompressed_size,
234
+ is_sorted: dictionary_page.is_sorted },
235
+ statistics: statistics&.to_h,
236
+ size_statistics: size_statistics,
237
+ key_value_metadata: key_value_metadata.empty? ? nil : key_value_metadata,
238
+ bloom_filter: bloom_filter_offset && { offset: bloom_filter_offset, length: bloom_filter_length },
239
+ column_index_offset: column_index_range&.first,
240
+ column_index_length: column_index_range&.last,
241
+ offset_index_offset: offset_index_range&.first,
242
+ offset_index_length: offset_index_range&.last,
243
+ external_file: external_file,
244
+ num_pages: pages.size,
245
+ num_data_pages: data_pages.size,
246
+ index_mismatches: index_mismatches.empty? ? nil : index_mismatches,
247
+ error: error,
248
+ pages: pages.map(&:to_h),
249
+ column_index: column_index&.to_h,
250
+ offset_index: offset_index&.to_h
251
+ }.compact)
252
+ end
253
+
254
+ def inspect
255
+ "#<#{self.class.name} #{path} rg=#{@row_group.index} #{codec} #{compressed_size}/#{uncompressed_size} bytes>"
256
+ end
257
+
258
+ private
259
+
260
+ # Page row counts come from the offset index when there is one (v1 pages don't carry them)
261
+ def apply_offset_index(list)
262
+ oi = offset_index or return
263
+ by_offset = list.select(&:data?).to_h { |p| [p.offset, p] }
264
+ locs = oi.page_locations
265
+ locs.each_with_index do |loc, i|
266
+ page = by_offset[loc.offset] or next
267
+ page.first_row_index = loc.first_row_index
268
+ next_first = locs[i + 1]&.first_row_index || @row_group.num_rows
269
+ page.num_rows ||= next_first - loc.first_row_index
270
+ end
271
+ end
272
+ end
273
+
274
+ # One row group
275
+ class RowGroupInfo
276
+ attr_reader :index, :row_group, :columns
277
+
278
+ def initialize(inspector, index, row_group, first_row)
279
+ @index = index
280
+ @row_group = row_group
281
+ @first_row = first_row
282
+ leaves = inspector.schema.columns
283
+ @columns = (row_group.columns || []).each_with_index.map do |cc, i|
284
+ column = leaves[i] or raise FormatError, "Row group #{index} has more column chunks than the schema has columns"
285
+ ColumnChunkInfo.new(inspector, self, column, cc)
286
+ end
287
+ end
288
+
289
+ def num_rows = @row_group.num_rows
290
+ def first_row = @first_row
291
+ def total_byte_size = @row_group.total_byte_size
292
+ def compressed_size = @row_group.total_compressed_size || @columns.sum(&:compressed_size)
293
+ def uncompressed_size = @columns.sum { |c| c.uncompressed_size.to_i }
294
+ def start_offset = @columns.map(&:start_offset).min
295
+ def end_offset = @columns.map(&:end_offset).max
296
+ def column(path) = @columns.find { |c| c.path == path.to_s || c.column.path == Array(path) }
297
+
298
+ def sorting_columns
299
+ (@row_group.sorting_columns || []).map do |s|
300
+ { column: @columns[s.column_idx]&.path || s.column_idx, descending: s.descending, nulls_first: s.nulls_first }
301
+ end
302
+ end
303
+
304
+ def to_h
305
+ Inspector.jsonable({
306
+ index: index,
307
+ ordinal: @row_group.ordinal,
308
+ num_rows: num_rows,
309
+ first_row: first_row,
310
+ total_byte_size: total_byte_size,
311
+ compressed_size: compressed_size,
312
+ uncompressed_size: uncompressed_size,
313
+ file_offset: @row_group.file_offset,
314
+ start_offset: start_offset,
315
+ end_offset: end_offset,
316
+ sorting_columns: sorting_columns,
317
+ columns: @columns.map(&:to_h)
318
+ }.compact)
319
+ end
320
+
321
+ def inspect
322
+ "#<#{self.class.name} #{index} rows=#{num_rows} columns=#{@columns.size}>"
323
+ end
324
+ end
325
+
326
+ attr_reader :metadata, :schema, :file_size, :footer_size
327
+
328
+ # A label for the file (the basename of the IO's path, when it has one)
329
+ attr_reader :name
330
+
331
+ # +io+ is a random-access IO (responds to #seek and #read, e.g. File.open(path, "rb")). It is
332
+ # left open.
333
+ def initialize(io)
334
+ @io = io
335
+ unless @io.respond_to?(:read) && @io.respond_to?(:seek)
336
+ raise ArgumentError, "Herringbone::Inspector expects an IO that supports #seek and #read " \
337
+ "(e.g. File.open(path, \"rb\")), got #{io.class}"
338
+ end
339
+ @name = @io.respond_to?(:path) && @io.path ? File.basename(@io.path.to_s) : nil
340
+ read_footer
341
+ @schema = Schema.from_elements(@metadata.schema)
342
+ end
343
+
344
+ def num_rows = @metadata.num_rows
345
+ def created_by = @metadata.created_by
346
+ def version = @metadata.version
347
+ def columns = @schema.columns
348
+ def footer_offset = @file_size - 8 - @footer_size
349
+
350
+ def row_groups
351
+ @row_groups ||= begin
352
+ first = 0
353
+ (@metadata.row_groups || []).each_with_index.map do |rg, i|
354
+ info = RowGroupInfo.new(self, i, rg, first)
355
+ first += rg.num_rows.to_i
356
+ info
357
+ end
358
+ end
359
+ end
360
+
361
+ def column_chunks = row_groups.flat_map(&:columns)
362
+
363
+ # Walks every page header and page index now (e.g. before closing the file)
364
+ def load_all
365
+ column_chunks.each do |c|
366
+ c.pages
367
+ c.column_index
368
+ c.offset_index
369
+ c.bloom_filter_length
370
+ end
371
+ self
372
+ end
373
+
374
+ def page_index? = column_chunks.any? { |c| c.column_index_range || c.offset_index_range }
375
+ def bloom_filters? = column_chunks.any?(&:bloom_filter_offset)
376
+
377
+ def key_value_metadata
378
+ (@metadata.key_value_metadata || []).map { |kv| describe_key_value(kv.key, kv.value) }
379
+ end
380
+
381
+ def column_orders
382
+ orders = @metadata.column_orders or return nil
383
+ orders.map { |o| o.type_order ? "TYPE_DEFINED_ORDER" : "UNKNOWN" }
384
+ end
385
+
386
+ def summary
387
+ codecs = column_chunks.map(&:codec).uniq
388
+ {
389
+ name: @name,
390
+ file_size: @file_size,
391
+ footer_size: @footer_size,
392
+ footer_offset: footer_offset,
393
+ format_version: version,
394
+ created_by: created_by,
395
+ num_rows: num_rows,
396
+ num_row_groups: row_groups.size,
397
+ num_columns: columns.size,
398
+ codecs: codecs,
399
+ compressed_size: column_chunks.sum { |c| c.compressed_size.to_i },
400
+ uncompressed_size: column_chunks.sum { |c| c.uncompressed_size.to_i },
401
+ page_index: page_index?,
402
+ bloom_filters: bloom_filters?,
403
+ column_orders: column_orders
404
+ }.tap { |h| h[:checksums] = checksum_summary.except(:mismatches) if checksums_verified? }
405
+ end
406
+
407
+ # Reads every page body and checks it against the CRC32 in its page header (the CRC covers
408
+ # the page data as stored: compressed, and for v2 pages the levels plus the compressed
409
+ # values). Nothing is decompressed. Sets PageInfo#checksum on every page and returns
410
+ # #checksum_summary. Needs the IO, so call it before closing the file.
411
+ def verify_checksums
412
+ column_chunks.each(&:verify_checksums)
413
+ @checksums_verified = true
414
+ checksum_summary
415
+ end
416
+
417
+ def checksums_verified? = @checksums_verified == true
418
+
419
+ # After #verify_checksums: { ok:, mismatch:, absent:, mismatches: [{ row_group:, column:, page:, type:, offset:, crc:, actual: }] }
420
+ def checksum_summary
421
+ return nil unless checksums_verified?
422
+ all = column_chunks.flat_map { |c| c.pages.map { |p| [c, p] } }
423
+ tally = all.map { |_, p| p.checksum }.tally
424
+ {
425
+ ok: tally.fetch(:ok, 0), mismatch: tally.fetch(:mismatch, 0), absent: tally.fetch(:absent, 0),
426
+ mismatches: all.select { |_, p| p.checksum == :mismatch }.map do |c, p|
427
+ { row_group: c.row_group.index, column: c.path, page: p.index, type: p.type, offset: p.offset,
428
+ crc: p.expected_crc, actual: page_crc(p) }
429
+ end
430
+ }
431
+ end
432
+
433
+ # Every disagreement between page header statistics and the ColumnIndex, across the file,
434
+ # each with :row_group and :column added (see ColumnChunkInfo#index_mismatches)
435
+ def index_mismatches
436
+ column_chunks.flat_map do |c|
437
+ c.index_mismatches.map { |m| { row_group: c.row_group.index, column: c.path }.merge(m) }
438
+ end
439
+ end
440
+
441
+ # The decoded ARROW:schema key/value (see ArrowSchema.decode); nil when the file has none or
442
+ # it could not be decoded, and #arrow_schema_error then says why
443
+ def arrow_schema
444
+ return @arrow_schema if defined?(@arrow_schema)
445
+ @arrow_schema_error = nil
446
+ kv = (@metadata.key_value_metadata || []).find { |x| x.key == "ARROW:schema" }
447
+ @arrow_schema = kv && begin
448
+ ArrowSchema.decode(kv.value)
449
+ rescue StandardError => e
450
+ @arrow_schema_error = "could not decode ARROW:schema: #{e.message}"
451
+ nil
452
+ end
453
+ end
454
+
455
+ def arrow_schema_error
456
+ arrow_schema
457
+ @arrow_schema_error
458
+ end
459
+
460
+ # Schema tree: Hashes with name, repetition, types, levels (leaves) and children (groups)
461
+ def schema_tree
462
+ leaf_by_node = @schema.columns.to_h { |c| [c.node, c] }
463
+ build = lambda do |node|
464
+ h = { name: node.name, repetition: node.repetition }
465
+ if node.leaf?
466
+ col = leaf_by_node[node]
467
+ h[:physical_type] = T::NAMES[node.type]&.to_s
468
+ h[:type_length] = node.type_length
469
+ h[:column] = col.index
470
+ h[:path] = col.dotted_path
471
+ h[:max_definition_level] = col.max_definition_level
472
+ h[:max_repetition_level] = col.max_repetition_level
473
+ end
474
+ h[:logical_type] = Inspector.logical_type_name(node)
475
+ h[:converted_type] = Format::ConvertedType::NAMES[node.converted_type]&.to_s
476
+ h[:field_id] = node.field_id
477
+ h[:children] = node.children.map { |c| build.call(c) } if node.group?
478
+ h.compact
479
+ end
480
+ tree = @schema.root.children.map { |c| build.call(c) }
481
+ annotate_arrow_types(tree, arrow_schema[:fields]) if arrow_schema
482
+ tree
483
+ end
484
+
485
+ # Per leaf column: sums over all row groups, plus overall min/max where comparable
486
+ def column_totals
487
+ columns.map do |col|
488
+ chunks = row_groups.map { |rg| rg.columns[col.index] }.compact
489
+ compressed = chunks.sum { |c| c.compressed_size.to_i }
490
+ uncompressed = chunks.sum { |c| c.uncompressed_size.to_i }
491
+ nulls = chunks.map(&:null_count)
492
+ stats = chunks.map(&:statistics)
493
+ mins = stats.map { |s| s&.min }
494
+ maxes = stats.map { |s| s&.max }
495
+ {
496
+ column: col.index,
497
+ path: col.dotted_path,
498
+ type: Inspector.type_name(col),
499
+ codecs: chunks.map(&:codec).uniq,
500
+ encodings: chunks.flat_map(&:encodings).uniq,
501
+ num_values: chunks.sum { |c| c.num_values.to_i },
502
+ null_count: nulls.all? ? nulls.sum : nil,
503
+ compressed_size: compressed,
504
+ uncompressed_size: uncompressed,
505
+ compression_ratio: compressed.positive? ? (uncompressed.to_f / compressed).round(4) : nil,
506
+ num_pages: chunks.sum { |c| c.pages.size },
507
+ num_data_pages: chunks.sum { |c| c.data_pages.size },
508
+ dictionary_pages: chunks.count(&:dictionary_page),
509
+ dictionary_bytes: chunks.sum { |c| c.dictionary_page&.total_size.to_i },
510
+ min: mins.all? ? safe_extreme(mins, :min) : nil,
511
+ max: maxes.all? ? safe_extreme(maxes, :max) : nil
512
+ }.compact
513
+ end
514
+ end
515
+
516
+ # Byte ranges of the whole file, in offset order: magic, pages (or whole chunks when their pages
517
+ # could not be walked), bloom filters, page indexes, footer. Gaps right after a chunk at its
518
+ # file_offset are inline :column_metadata copies; other gaps are reported as :unknown. Each entry: { kind:, start:, length:, row_group:, column:, page: }
519
+ def layout
520
+ segs = [{ kind: :magic, start: 0, length: 4 }]
521
+ column_chunks.each do |c|
522
+ rg = c.row_group.index
523
+ col = c.column.index
524
+ if c.pages.empty?
525
+ segs << { kind: :chunk, start: c.start_offset, length: c.compressed_size, row_group: rg, column: col }
526
+ else
527
+ c.pages.each_with_index do |p, i|
528
+ segs << { kind: p.dictionary? ? :dictionary_page : :data_page, start: p.offset, length: p.total_size,
529
+ row_group: rg, column: col, page: i }
530
+ end
531
+ end
532
+ if c.bloom_filter_offset
533
+ segs << { kind: :bloom_filter, start: c.bloom_filter_offset, length: c.bloom_filter_length.to_i,
534
+ row_group: rg, column: col }
535
+ end
536
+ if (r = c.column_index_range)
537
+ segs << { kind: :column_index, start: r[0], length: r[1], row_group: rg, column: col }
538
+ end
539
+ if (r = c.offset_index_range)
540
+ segs << { kind: :offset_index, start: r[0], length: r[1], row_group: rg, column: col }
541
+ end
542
+ end
543
+ segs << { kind: :footer, start: footer_offset, length: @footer_size }
544
+ segs << { kind: :footer_length, start: @file_size - 8, length: 4 }
545
+ segs << { kind: :magic, start: @file_size - 4, length: 4 }
546
+ segs.sort_by! { |s| s[:start] }
547
+ # Some writers (old parquet-rs, parquet-mr for a while) store a copy of the ColumnMetaData
548
+ # right after the chunk, at ColumnChunk.file_offset
549
+ meta_copies = column_chunks.each_with_object({}) do |c, h|
550
+ fo = c.chunk.file_offset
551
+ h[fo] = c if fo && fo >= c.end_offset
552
+ end
553
+ out = []
554
+ pos = 0
555
+ segs.each do |s|
556
+ if s[:start] > pos
557
+ c = meta_copies[pos]
558
+ out << if c
559
+ { kind: :column_metadata, start: pos, length: s[:start] - pos, row_group: c.row_group.index, column: c.column.index }
560
+ else
561
+ { kind: :unknown, start: pos, length: s[:start] - pos }
562
+ end
563
+ end
564
+ out << s
565
+ pos = [pos, s[:start] + s[:length]].max
566
+ end
567
+ out
568
+ end
569
+
570
+ def to_h
571
+ Inspector.jsonable({
572
+ summary: summary,
573
+ key_value_metadata: key_value_metadata,
574
+ schema: schema_tree,
575
+ row_groups: row_groups.map(&:to_h),
576
+ column_totals: column_totals,
577
+ checksum_mismatches: checksum_summary&.fetch(:mismatches),
578
+ index_mismatches: index_mismatches
579
+ }.compact)
580
+ end
581
+
582
+ def to_json(*args) = to_h.to_json(*args)
583
+
584
+ # A self-contained HTML page showing the file's layout, see Visualizer
585
+ def to_html = Visualizer.new(self).to_html
586
+
587
+ # Readable text summary. With pages: true, lists every page header too.
588
+ def report(pages: false)
589
+ s = summary
590
+ out = []
591
+ out << "file: #{@name || "(IO)"}"
592
+ out << "size: #{Inspector.human_bytes(s[:file_size])} (#{s[:file_size]} bytes), footer #{s[:footer_size]} bytes at #{s[:footer_offset]}"
593
+ out << "rows: #{s[:num_rows]}, row groups: #{s[:num_row_groups]}, columns: #{s[:num_columns]}, format version #{s[:format_version]}"
594
+ out << "created by: #{s[:created_by]}"
595
+ out << "codecs: #{s[:codecs].join(", ")}; data #{Inspector.human_bytes(s[:compressed_size])} compressed, " \
596
+ "#{Inspector.human_bytes(s[:uncompressed_size])} uncompressed#{ratio_text(s[:uncompressed_size], s[:compressed_size])}"
597
+ out << "page index: #{s[:page_index] ? "yes" : "no"}, bloom filters: #{s[:bloom_filters] ? "yes" : "no"}"
598
+ if (cs = checksum_summary)
599
+ out << "page CRCs: #{cs[:ok]} ok, #{cs[:mismatch]} mismatched, #{cs[:absent]} without a CRC"
600
+ cs[:mismatches].each do |m|
601
+ out << " CRC MISMATCH: row group #{m[:row_group]} #{m[:column]} page #{m[:page]} (#{m[:type]} @#{m[:offset]}): " \
602
+ "header says #{format("%08x", m[:crc])}, data has #{format("%08x", m[:actual])}"
603
+ end
604
+ end
605
+ mismatches = index_mismatches
606
+ unless mismatches.empty?
607
+ out << "page statistics vs column index: #{mismatches.size} disagreement#{mismatches.size == 1 ? "" : "s"}"
608
+ mismatches.first(50).each { |m| out << " #{index_mismatch_text(m)}" }
609
+ out << " ..." if mismatches.size > 50
610
+ end
611
+ kvs = key_value_metadata
612
+ unless kvs.empty?
613
+ out << "key/value metadata:"
614
+ kvs.each do |kv|
615
+ out << " #{kv[:key]} (#{kv[:format]}, #{kv[:bytesize]} bytes): #{kv[:summary] || kv[:value].to_s[0, 80].inspect}"
616
+ out << " #{kv[:arrow_error]}" if kv[:arrow_error]
617
+ ArrowSchema.lines(kv[:arrow_schema][:fields]).each { |l| out << " #{l}" } if kv[:arrow_schema]
618
+ end
619
+ end
620
+ out << "schema:"
621
+ walk = lambda do |n, depth|
622
+ type = n[:children] ? "group" : [n[:physical_type], n[:type_length] && "(#{n[:type_length]})"].compact.join
623
+ ann = n[:logical_type] || n[:converted_type]
624
+ levels = n[:children] ? "" : " [def #{n[:max_definition_level]}, rep #{n[:max_repetition_level]}]"
625
+ arrow = n[:arrow_type] ? " arrow: #{n[:arrow_type]}" : ""
626
+ out << "#{" " * depth}#{n[:repetition]} #{type} #{n[:name]}#{ann ? " (#{ann})" : ""}#{levels}#{arrow}"
627
+ (n[:children] || []).each { |c| walk.call(c, depth + 1) }
628
+ end
629
+ schema_tree.each { |n| walk.call(n, 1) }
630
+ out << "columns:"
631
+ column_totals.each do |t|
632
+ range = t.key?(:min) ? " [#{Inspector.display(t[:min])} .. #{Inspector.display(t[:max])}]" : ""
633
+ out << " #{t[:path]}: #{t[:type]} #{t[:codecs].join(",")} #{t[:encodings].join(",")} " \
634
+ "#{Inspector.human_bytes(t[:compressed_size])}/#{Inspector.human_bytes(t[:uncompressed_size])}" \
635
+ "#{ratio_text(t[:uncompressed_size], t[:compressed_size])}, #{t[:num_values]} values" \
636
+ "#{t[:null_count] ? ", #{t[:null_count]} nulls" : ""}, #{t[:num_data_pages]} data pages#{range}"
637
+ end
638
+ row_groups.each do |rg|
639
+ sorting = rg.sorting_columns.map { |c| "#{c[:column]}#{c[:descending] ? " desc" : ""}" }
640
+ out << "row group #{rg.index}: #{rg.num_rows} rows, #{Inspector.human_bytes(rg.compressed_size)} " \
641
+ "at #{rg.start_offset}..#{rg.end_offset}#{sorting.empty? ? "" : ", sorted by #{sorting.join(", ")}"}"
642
+ rg.columns.each do |c|
643
+ st = c.statistics
644
+ range = st && (st.min || st.max) ? " [#{Inspector.display(st.min)} .. #{Inspector.display(st.max)}]" : ""
645
+ extras = []
646
+ extras << "dict #{c.dictionary_size} entries" if c.dictionary_page
647
+ extras << "column index" if c.column_index_range
648
+ extras << "offset index" if c.offset_index_range
649
+ extras << "bloom filter" if c.bloom_filter_offset
650
+ extras << "ERROR: #{c.error}" if c.error
651
+ out << " #{c.path}: #{c.codec} #{c.encodings.join(",")} " \
652
+ "#{c.compressed_size}/#{c.uncompressed_size} bytes#{ratio_text(c.uncompressed_size, c.compressed_size)}, " \
653
+ "#{c.num_values} values, #{c.pages.size} pages#{extras.empty? ? "" : ", #{extras.join(", ")}"}#{range}"
654
+ next unless pages
655
+ c.pages.each do |p|
656
+ st = p.statistics
657
+ out << " #{p.index}: #{p.type} @#{p.offset} header #{p.header_size} + #{p.compressed_size}/#{p.uncompressed_size} bytes, " \
658
+ "#{p.num_values} values#{p.num_nulls ? ", #{p.num_nulls} nulls" : ""}#{p.num_rows ? ", #{p.num_rows} rows" : ""}" \
659
+ "#{p.encoding ? " #{p.encoding}" : ""}#{crc_text(p)}" \
660
+ "#{st && (st.min || st.max) ? " [#{Inspector.display(st.min)} .. #{Inspector.display(st.max)}]" : ""}"
661
+ end
662
+ end
663
+ end
664
+ out.join("\n")
665
+ end
666
+
667
+ def inspect
668
+ "#<#{self.class.name} #{@name || "(IO)"} rows=#{num_rows} row_groups=#{row_groups.size} size=#{@file_size}>"
669
+ end
670
+
671
+ # ---- used by the info objects ----
672
+
673
+ # Walks page headers from the chunk's first page. Returns [pages, error_message_or_nil].
674
+ # Mirrors the reader's tolerance: a chunk may extend past its declared total_compressed_size.
675
+ def walk_pages(chunk)
676
+ return [[], "column chunk stored in external file #{chunk.external_file}"] if chunk.external_file
677
+ pages = []
678
+ pos = chunk.start_offset
679
+ limit = footer_offset
680
+ total = chunk.num_values.to_i
681
+ seen = 0
682
+ declared_end = chunk.declared_end_offset
683
+ # Walk until all values are accounted for (reading past the declared end if a writer
684
+ # under-reported it), then pick up any trailing pages still inside the declared range
685
+ while pos < limit && (seen < total || pos < declared_end)
686
+ trailing = seen >= total
687
+ begin
688
+ header, header_size = read_page_header(pos, limit)
689
+ page = page_info(pages.size, header, pos, header_size, chunk.column)
690
+ rescue Thrift::Error, FormatError => e
691
+ break if trailing
692
+ return [pages, "corrupt page header at #{pos}: #{e.message}"]
693
+ end
694
+ if page.end_offset > limit
695
+ break if trailing
696
+ return [pages, "page #{pages.size} at #{pos} overruns the data section (#{page.end_offset} > #{limit})"]
697
+ end
698
+ pages << page
699
+ seen += page.num_values.to_i if page.data?
700
+ pos = page.end_offset
701
+ end
702
+ [pages, seen < total ? "found #{seen} of #{total} values in page headers" : nil]
703
+ end
704
+
705
+ def read_column_index(chunk)
706
+ offset, length = chunk.column_index_range
707
+ return nil unless offset && length.positive?
708
+ ci = Format::ColumnIndex.decode(read_at(offset, length)).first
709
+ col = chunk.column
710
+ mins = ci.min_values || []
711
+ maxes = ci.max_values || []
712
+ nulls = ci.null_pages || []
713
+ ColumnIndexInfo.new(
714
+ offset: offset, length: length,
715
+ null_pages: nulls,
716
+ min_values: mins.each_with_index.map { |v, i| nulls[i] ? nil : decode_value(v, col) },
717
+ max_values: maxes.each_with_index.map { |v, i| nulls[i] ? nil : decode_value(v, col) },
718
+ boundary_order: Format::BoundaryOrder::NAMES[ci.boundary_order]&.to_s || ci.boundary_order,
719
+ null_counts: ci.null_counts,
720
+ repetition_level_histograms: ci.repetition_level_histograms,
721
+ definition_level_histograms: ci.definition_level_histograms
722
+ )
723
+ rescue Thrift::Error
724
+ nil
725
+ end
726
+
727
+ def read_offset_index(chunk)
728
+ offset, length = chunk.offset_index_range
729
+ return nil unless offset && length.positive?
730
+ oi = Format::OffsetIndex.decode(read_at(offset, length)).first
731
+ OffsetIndexInfo.new(
732
+ offset: offset, length: length,
733
+ page_locations: (oi.page_locations || []).map do |l|
734
+ PageLocation.new(offset: l.offset, compressed_page_size: l.compressed_page_size, first_row_index: l.first_row_index)
735
+ end,
736
+ unencoded_byte_array_data_bytes: oi.unencoded_byte_array_data_bytes
737
+ )
738
+ rescue Thrift::Error
739
+ nil
740
+ end
741
+
742
+ # Size in bytes of the bloom filter at +offset+ (its Thrift header plus the bitset)
743
+ def bloom_filter_size(offset)
744
+ buf = read_at(offset, 64)
745
+ header, size = BloomFilterHeader.decode(buf)
746
+ header.num_bytes ? size + header.num_bytes : nil
747
+ rescue Thrift::Error
748
+ nil
749
+ end
750
+
751
+ # :ok, :mismatch or :absent for one page (reads its body)
752
+ def page_checksum(page)
753
+ return :absent unless page.crc
754
+ page_crc(page) == page.expected_crc ? :ok : :mismatch
755
+ end
756
+
757
+ # CRC32 of a page's body as stored
758
+ def page_crc(page)
759
+ Zlib.crc32(read_at(page.body_offset, page.compressed_size))
760
+ end
761
+
762
+ # See ColumnChunkInfo#index_mismatches
763
+ def compare_page_index(chunk)
764
+ ci = chunk.column_index or return []
765
+ data = chunk.data_pages
766
+ if ci.null_pages.size != data.size
767
+ return [{ page: nil, data_page: nil, field: :page_count, page_value: data.size, index_value: ci.null_pages.size }]
768
+ end
769
+ order = Inspector.sort_order(chunk.column)
770
+ out = []
771
+ data.each_with_index do |p, k|
772
+ report = ->(field, pv, iv) { out << { page: p.index, data_page: k, field: field, page_value: pv, index_value: iv } }
773
+ idx_nulls = ci.null_counts&.[](k)
774
+ report.call(:null_count, p.num_nulls, idx_nulls) if idx_nulls && p.num_nulls && idx_nulls != p.num_nulls
775
+ st = p.statistics
776
+ # Legacy min/max were computed with another ordering, so they can't be compared
777
+ next if st.nil? || st.caveat || (st.min.nil? && st.max.nil?)
778
+ if ci.null_pages[k]
779
+ report.call(:null_page, "has min/max", "null page")
780
+ next
781
+ end
782
+ imin = ci.min_values[k]
783
+ imax = ci.max_values[k]
784
+ report.call(:min, st.min, imin) if st.min_exact != false && narrower?(imin, st.min, order, :min)
785
+ report.call(:max, st.max, imax) if st.max_exact != false && narrower?(imax, st.max, order, :max)
786
+ end
787
+ out
788
+ end
789
+
790
+ # Decodes a Format::Statistics into Ruby values via the column's type converter
791
+ def decode_statistics(st, column)
792
+ return nil unless st
793
+ order = Inspector.sort_order(column)
794
+ if st.min_value || st.max_value
795
+ min_raw = st.min_value
796
+ max_raw = st.max_value
797
+ source = "min_value/max_value"
798
+ elsif st.min || st.max
799
+ min_raw = st.min
800
+ max_raw = st.max
801
+ source = "min/max (legacy)"
802
+ binary = column.type == T::BYTE_ARRAY || column.type == T::FIXED_LEN_BYTE_ARRAY
803
+ if order == :unsigned
804
+ caveat = "legacy min/max were computed with signed comparison, which is wrong for unsigned-ordered " \
805
+ "values (strings, binary, unsigned integers); most readers ignore them"
806
+ elsif binary
807
+ caveat = "legacy min/max of binary-backed values were compared bytewise; most readers ignore them"
808
+ end
809
+ end
810
+ caveat ||= "sort order of #{T::NAMES[column.type]} is undefined; min/max may be meaningless" if order == :unknown && source
811
+ Stats.new(
812
+ min: decode_value(min_raw, column),
813
+ max: decode_value(max_raw, column),
814
+ null_count: st.null_count,
815
+ distinct_count: st.distinct_count,
816
+ min_exact: st.is_min_value_exact,
817
+ max_exact: st.is_max_value_exact,
818
+ source: source,
819
+ caveat: caveat
820
+ )
821
+ end
822
+
823
+ # Decodes one PLAIN-encoded statistics value (no length prefix for byte arrays)
824
+ def decode_value(bytes, column)
825
+ return nil if bytes.nil?
826
+ bytes = bytes.b
827
+ raw = case column.type
828
+ when T::BOOLEAN
829
+ return nil if bytes.empty?
830
+ bytes.getbyte(0) != 0
831
+ when T::BYTE_ARRAY, T::FIXED_LEN_BYTE_ARRAY
832
+ bytes
833
+ when T::INT96
834
+ return Inspector.hex(bytes) unless bytes.bytesize == 12
835
+ bytes.unpack("Q<L<")
836
+ else
837
+ width = Encodings::Plain::FORMATS[column.type][1]
838
+ return Inspector.hex(bytes) unless bytes.bytesize == width
839
+ Encodings::Plain.decode(bytes, 0, 1, column.type).first.first
840
+ end
841
+ conv = column.converter
842
+ conv ? conv.call(raw) : raw
843
+ rescue StandardError
844
+ Inspector.hex(bytes)
845
+ end
846
+
847
+ # Decodes the ARROW:schema key/value that Arrow writers (pyarrow, arrow-rs, DuckDB...) store:
848
+ # base64 of an Arrow IPC message whose header is a flatbuffer Schema (Arrow's Message.fbs and
849
+ # Schema.fbs). Pure Ruby and read-only; type names follow pyarrow's (str(field.type)).
850
+ #
851
+ # Inspector::ArrowSchema.decode(value)
852
+ # # => { endianness: "little", metadata: {...}, fields: [{ name: "a", type: "int32", nullable: true, ... }] }
853
+ #
854
+ # Fields carry :name, :type, :nullable and, when present, :children, :dictionary
855
+ # ({ index_type:, ordered:, id: }), :extension (ARROW:extension:name) and :metadata.
856
+ module ArrowSchema
857
+ class Error < StandardError; end
858
+
859
+ MAX_DEPTH = 64
860
+ MAX_FIELDS = 100_000
861
+ TIME_UNITS = %w[s ms us ns].freeze
862
+ MESSAGE_SCHEMA = 1
863
+
864
+ # A minimal flatbuffer reader: tables (through their vtables), scalars, strings, vectors of
865
+ # scalars and tables, and unions. Every read is bounds-checked; malformed input raises Error.
866
+ class FlatBuffer
867
+ def initialize(bytes)
868
+ @b = bytes.b
869
+ end
870
+
871
+ def root = table_at(u32(0))
872
+ def table_at(pos) = Table.new(self, pos)
873
+
874
+ def check(pos, len)
875
+ return if pos >= 0 && len >= 0 && pos + len <= @b.bytesize
876
+ raise Error, "flatbuffer read of #{len} bytes at #{pos} is out of bounds (#{@b.bytesize} bytes)"
877
+ end
878
+
879
+ def read(pos, len, fmt)
880
+ check(pos, len)
881
+ @b.byteslice(pos, len).unpack1(fmt)
882
+ end
883
+
884
+ def u8(pos) = read(pos, 1, "C")
885
+ def u16(pos) = read(pos, 2, "S<")
886
+ def i16(pos) = read(pos, 2, "s<")
887
+ def u32(pos) = read(pos, 4, "L<")
888
+ def i32(pos) = read(pos, 4, "l<")
889
+ def i64(pos) = read(pos, 8, "q<")
890
+
891
+ # Offsets are relative to where they are stored
892
+ def deref(pos) = pos + u32(pos)
893
+
894
+ def string(pos)
895
+ len = u32(pos)
896
+ check(pos + 4, len)
897
+ @b.byteslice(pos + 4, len).force_encoding(Encoding::UTF_8)
898
+ end
899
+
900
+ # [start, length] of the vector at +pos+ with +size+-byte elements
901
+ def vector(pos, size)
902
+ len = u32(pos)
903
+ check(pos + 4, len * size)
904
+ [pos + 4, len]
905
+ end
906
+ end
907
+
908
+ # One flatbuffer table; fields are addressed by their slot (declaration order in the .fbs)
909
+ class Table
910
+ def initialize(fb, pos)
911
+ @fb = fb
912
+ @pos = pos
913
+ @vtable = pos - fb.i32(pos)
914
+ @vtable_size = fb.u16(@vtable)
915
+ raise Error, "bad flatbuffer vtable at #{@vtable}" if @vtable_size < 4 || @vtable_size.odd?
916
+ fb.check(@vtable, @vtable_size)
917
+ end
918
+
919
+ # Absolute position of a field's value, nil when absent
920
+ def field(slot)
921
+ o = 4 + (slot * 2)
922
+ return nil if o + 2 > @vtable_size
923
+ off = @fb.u16(@vtable + o)
924
+ off.zero? ? nil : @pos + off
925
+ end
926
+
927
+ def u8(slot, default = 0) = (p = field(slot)) ? @fb.u8(p) : default
928
+ def bool(slot, default = false) = (p = field(slot)) ? @fb.u8(p) != 0 : default
929
+ def i16(slot, default = 0) = (p = field(slot)) ? @fb.i16(p) : default
930
+ def i32(slot, default = 0) = (p = field(slot)) ? @fb.i32(p) : default
931
+ def i64(slot, default = 0) = (p = field(slot)) ? @fb.i64(p) : default
932
+ def string(slot) = (p = field(slot)) && @fb.string(@fb.deref(p))
933
+ def table(slot) = (p = field(slot)) && @fb.table_at(@fb.deref(p))
934
+
935
+ def tables(slot)
936
+ p = field(slot) or return []
937
+ start, len = @fb.vector(@fb.deref(p), 4)
938
+ Array.new(len) { |i| @fb.table_at(@fb.deref(start + (4 * i))) }
939
+ end
940
+
941
+ def i32s(slot)
942
+ p = field(slot) or return []
943
+ start, len = @fb.vector(@fb.deref(p), 4)
944
+ Array.new(len) { |i| @fb.i32(start + (4 * i)) }
945
+ end
946
+ end
947
+
948
+ module_function
949
+
950
+ # Decodes the base64 ARROW:schema value; raises ArrowSchema::Error when it can't
951
+ def decode(b64)
952
+ bytes = b64.to_s.unpack1("m")
953
+ raise Error, "empty value" if bytes.empty?
954
+ fb = FlatBuffer.new(message_bytes(bytes))
955
+ message = fb.root
956
+ header_type = message.u8(1)
957
+ raise Error, "IPC message holds a #{header_type} header, not a Schema" unless header_type == MESSAGE_SCHEMA
958
+ schema = message.table(2) or raise Error, "IPC message has no Schema"
959
+ count = [0]
960
+ {
961
+ endianness: schema.i16(0).zero? ? "little" : "big",
962
+ fields: schema.tables(1).map { |f| field(f, 0, count) },
963
+ metadata: key_values(schema.tables(2))
964
+ }.compact
965
+ rescue ArgumentError, TypeError, RangeError => e
966
+ raise Error, e.message
967
+ end
968
+
969
+ # The flatbuffer inside an encapsulated IPC message: [0xFFFFFFFF] int32 length, flatbuffer
970
+ # (the continuation marker is missing in files from before Arrow 0.15)
971
+ def message_bytes(bytes)
972
+ raise Error, "too short for an IPC message (#{bytes.bytesize} bytes)" if bytes.bytesize < 8
973
+ len = bytes.unpack1("l<")
974
+ start = 4
975
+ if len == -1
976
+ len = bytes.byteslice(4, 4).unpack1("l<")
977
+ start = 8
978
+ end
979
+ raise Error, "IPC message length #{len} does not fit in #{bytes.bytesize} bytes" if len <= 0 || start + len > bytes.bytesize
980
+ bytes.byteslice(start, len)
981
+ end
982
+
983
+ def field(t, depth, count)
984
+ raise Error, "fields nested deeper than #{MAX_DEPTH} levels" if depth > MAX_DEPTH
985
+ raise Error, "more than #{MAX_FIELDS} fields" if (count[0] += 1) > MAX_FIELDS
986
+ children = t.tables(5).map { |c| field(c, depth + 1, count) }
987
+ metadata = key_values(t.tables(6))
988
+ type = type_name(t.u8(2), t.table(3), children)
989
+ h = { name: t.string(0).to_s, type: type, nullable: t.bool(1) }
990
+ if (d = t.table(4))
991
+ index = d.table(1)
992
+ index_type = index ? int_name(index) : "int32"
993
+ ordered = d.bool(2)
994
+ h[:dictionary] = { index_type: index_type, ordered: ordered, id: d.i64(0) }
995
+ h[:type] = "dictionary<values=#{type}, indices=#{index_type}, ordered=#{ordered ? 1 : 0}>"
996
+ end
997
+ h[:children] = children unless children.empty?
998
+ if metadata
999
+ h[:extension] = metadata["ARROW:extension:name"] if metadata["ARROW:extension:name"]
1000
+ h[:metadata] = metadata
1001
+ end
1002
+ h
1003
+ end
1004
+
1005
+ def key_values(tables)
1006
+ return nil if tables.empty?
1007
+ tables.to_h { |kv| [kv.string(0).to_s, kv.string(1).to_s] }
1008
+ end
1009
+
1010
+ # A child as pyarrow prints it inside a nested type: "name: type" plus " not null"
1011
+ def child_text(c) = "#{c[:name]}: #{c[:type]}#{c[:nullable] ? "" : " not null"}"
1012
+
1013
+ def int_name(t) = "#{t.bool(1) ? "" : "u"}int#{t.i32(0)}"
1014
+
1015
+ def unit(u) = TIME_UNITS[u] || "unit#{u}"
1016
+
1017
+ # Type names follow Arrow's DataType::ToString (what pyarrow prints)
1018
+ def type_name(kind, t, children)
1019
+ case kind
1020
+ when 1 then "null"
1021
+ when 2 then t ? int_name(t) : "int"
1022
+ when 3 then %w[halffloat float double][t ? t.i16(0) : 0] || "float?"
1023
+ when 4 then "binary"
1024
+ when 5 then "string"
1025
+ when 6 then "bool"
1026
+ when 7
1027
+ raise Error, "decimal type without parameters" unless t
1028
+ bits = t.i32(2, 128)
1029
+ "decimal#{bits}(#{t.i32(0)}, #{t.i32(1)})"
1030
+ when 8 then t&.i16(0, 1)&.zero? ? "date32[day]" : "date64[ms]"
1031
+ when 9
1032
+ bits = t ? t.i32(1, 32) : 32
1033
+ "time#{bits}[#{unit(t ? t.i16(0, 1) : 1)}]"
1034
+ when 10
1035
+ tz = t&.string(1)
1036
+ "timestamp[#{unit(t ? t.i16(0) : 0)}#{tz ? ", tz=#{tz}" : ""}]"
1037
+ when 11 then %w[month_interval day_time_interval month_day_nano_interval][t ? t.i16(0) : 0] || "interval"
1038
+ when 12 then "list<#{children.map { |c| child_text(c) }.join(", ")}>"
1039
+ when 13 then "struct<#{children.map { |c| child_text(c) }.join(", ")}>"
1040
+ when 14
1041
+ mode = t&.i16(0)&.positive? ? "dense" : "sparse"
1042
+ ids = t ? t.i32s(1) : []
1043
+ members = children.each_with_index.map { |c, i| "#{child_text(c)}=#{ids[i] || i}" }
1044
+ "#{mode}_union<#{members.join(", ")}>"
1045
+ when 15 then "fixed_size_binary[#{t ? t.i32(0) : 0}]"
1046
+ when 16 then "fixed_size_list<#{children.map { |c| child_text(c) }.join(", ")}>[#{t ? t.i32(0) : 0}]"
1047
+ when 17 then map_name(t, children)
1048
+ when 18 then "duration[#{unit(t ? t.i16(0, 1) : 1)}]"
1049
+ when 19 then "large_binary"
1050
+ when 20 then "large_string"
1051
+ when 21 then "large_list<#{children.map { |c| child_text(c) }.join(", ")}>"
1052
+ when 22 then "run_end_encoded<#{children.map { |c| "#{c[:name] == "values" ? "values" : "run_ends"}: #{c[:type]}" }.join(", ")}>"
1053
+ when 23 then "binary_view"
1054
+ when 24 then "string_view"
1055
+ when 25 then "list_view<#{children.map { |c| child_text(c) }.join(", ")}>"
1056
+ when 26 then "large_list_view<#{children.map { |c| child_text(c) }.join(", ")}>"
1057
+ else "unknown type #{kind}"
1058
+ end
1059
+ end
1060
+
1061
+ # map<key, value> with non-standard field names in parentheses, as Arrow prints it
1062
+ def map_name(t, children)
1063
+ entries = children.first
1064
+ kv = entries && entries[:children] || []
1065
+ named = ->(f, std) { f ? "#{f[:type]}#{f[:name] == std ? "" : " ('#{f[:name]}')"}" : "?" }
1066
+ sorted = t&.bool(0) ? ", keys_sorted" : ""
1067
+ entries_name = entries && entries[:name] != "entries" ? " ('#{entries[:name]}')" : ""
1068
+ "map<#{named.call(kv[0], "key")}, #{named.call(kv[1], "value")}#{sorted}#{entries_name}>"
1069
+ end
1070
+
1071
+ # "name: type" lines for a field and its children, indented, for text output
1072
+ def lines(fields, depth = 0, out = [])
1073
+ fields.each do |f|
1074
+ notes = []
1075
+ notes << "not null" unless f[:nullable]
1076
+ notes << "extension #{f[:extension]}" if f[:extension]
1077
+ meta = (f[:metadata] || {}).reject { |k, _| k.start_with?("ARROW:extension:") }
1078
+ notes << "metadata #{meta.map { |k, v| "#{k}=#{v.to_s[0, 60].inspect}" }.join(", ")}" unless meta.empty?
1079
+ out << "#{" " * depth}#{f[:name]}: #{f[:type]}#{notes.empty? ? "" : " (#{notes.join("; ")})"}"
1080
+ lines(f[:children], depth + 1, out) if f[:children] && depth < 8
1081
+ end
1082
+ out
1083
+ end
1084
+ end
1085
+
1086
+ # ---- class helpers ----
1087
+
1088
+ def self.encoding_name(e) = Format::Encoding::NAMES[e]&.to_s || e.to_s
1089
+
1090
+ def self.type_name(column)
1091
+ node = column.node
1092
+ phys = T::NAMES[node.type].to_s
1093
+ phys += "(#{node.type_length})" if node.type == T::FIXED_LEN_BYTE_ARRAY && node.type_length
1094
+ logical = logical_type_name(node)
1095
+ logical ? "#{phys} #{logical}" : phys
1096
+ end
1097
+
1098
+ def self.logical_type_name(node)
1099
+ if (kind = node.logical_type&.kind)
1100
+ name, payload = kind
1101
+ case name
1102
+ when :integer then "INTEGER(#{payload.bit_width}, #{payload.is_signed ? "signed" : "unsigned"})"
1103
+ when :decimal then "DECIMAL(#{payload.precision}, #{payload.scale})"
1104
+ when :timestamp, :time
1105
+ "#{name.upcase}(#{payload.unit&.to_sym&.upcase}, #{payload.is_adjusted_to_utc ? "UTC" : "local"})"
1106
+ when :unknown then "NULL"
1107
+ else name.to_s.upcase
1108
+ end
1109
+ elsif node.converted_type
1110
+ c = Format::ConvertedType::NAMES[node.converted_type].to_s
1111
+ c == "DECIMAL" ? "DECIMAL(#{node.precision}, #{node.scale || 0})" : c
1112
+ end
1113
+ end
1114
+
1115
+ # :signed, :unsigned or :unknown, per the Parquet sort order rules for the column's type
1116
+ def self.sort_order(column)
1117
+ kind, _a, signed = Types.logical_of(column.node)
1118
+ case kind
1119
+ when :integer then signed == false ? :unsigned : :signed
1120
+ when :decimal, :date, :time, :timestamp, :float16 then :signed
1121
+ when :string, :enum, :json, :bson, :uuid then :unsigned
1122
+ else
1123
+ case column.type
1124
+ when T::BYTE_ARRAY, T::FIXED_LEN_BYTE_ARRAY then :unsigned
1125
+ when T::INT96 then :unknown
1126
+ else :signed
1127
+ end
1128
+ end
1129
+ end
1130
+
1131
+ # Converts Ruby values (Time, BigDecimal, binary Strings, non-finite Floats...) to JSON-safe ones
1132
+ def self.jsonable(v)
1133
+ case v
1134
+ when Hash then v.each_with_object({}) { |(k, x), h| h[k.is_a?(Symbol) ? k : k.to_s] = jsonable(x) }
1135
+ when Array then v.map { |x| jsonable(x) }
1136
+ when Struct then v.respond_to?(:to_h) ? jsonable(v.to_h) : v.to_s
1137
+ when String then text(v)
1138
+ when Symbol then v.to_s
1139
+ when Float then v.finite? ? v : v.to_s
1140
+ when Time then v.utc.strftime(v.nsec.zero? ? "%Y-%m-%dT%H:%M:%SZ" : "%Y-%m-%dT%H:%M:%S.%NZ")
1141
+ when Date then v.iso8601
1142
+ when BigDecimal then v.to_s("F")
1143
+ when Integer, true, false, nil then v
1144
+ else v.to_s
1145
+ end
1146
+ end
1147
+
1148
+ # A String as readable text when it is valid UTF-8 without control characters, else as hex
1149
+ def self.text(s)
1150
+ return s if s.encoding == Encoding::UTF_8 && s.valid_encoding?
1151
+ u = s.dup.force_encoding(Encoding::UTF_8)
1152
+ return u if u.valid_encoding? && !u.match?(/[\x00-\x08\x0e-\x1f\x7f]/)
1153
+ hex(s)
1154
+ end
1155
+
1156
+ def self.hex(bytes)
1157
+ b = bytes.b
1158
+ b.bytesize > 64 ? "0x#{b.byteslice(0, 64).unpack1("H*")}… (#{b.bytesize} bytes)" : "0x#{b.unpack1("H*")}"
1159
+ end
1160
+
1161
+ # A short display form of a decoded value
1162
+ def self.display(v, max: 40)
1163
+ s = case v
1164
+ when String then v.encoding == Encoding::BINARY ? text(v) : v
1165
+ when nil then "null"
1166
+ else jsonable(v).to_s
1167
+ end
1168
+ s = s.inspect if v.is_a?(String)
1169
+ s.size > max ? "#{s[0, max - 1]}…" : s
1170
+ end
1171
+
1172
+ def self.human_bytes(n)
1173
+ return "?" unless n
1174
+ units = %w[B KB MB GB TB]
1175
+ f = n.to_f
1176
+ i = 0
1177
+ while f >= 1024 && i < units.size - 1
1178
+ f /= 1024
1179
+ i += 1
1180
+ end
1181
+ i.zero? ? "#{n} B" : format("%.#{f < 10 ? 2 : 1}f %s", f, units[i])
1182
+ end
1183
+
1184
+ private
1185
+
1186
+ def ratio_text(uncompressed, compressed)
1187
+ compressed.to_i.positive? && uncompressed ? format(" (%.2fx)", uncompressed.to_f / compressed) : ""
1188
+ end
1189
+
1190
+ def crc_text(page)
1191
+ case page.checksum
1192
+ when :ok then " crc ok"
1193
+ when :mismatch then " CRC MISMATCH"
1194
+ else page.crc ? " crc" : ""
1195
+ end
1196
+ end
1197
+
1198
+ def index_mismatch_text(m)
1199
+ where = "row group #{m[:row_group]} #{m[:column]}"
1200
+ if m[:field] == :page_count
1201
+ "#{where}: #{m[:page_value]} data pages but #{m[:index_value]} column index entries"
1202
+ else
1203
+ "#{where} page #{m[:page]}: #{m[:field]} in page header #{Inspector.display(m[:page_value])}, " \
1204
+ "in column index #{Inspector.display(m[:index_value])}"
1205
+ end
1206
+ end
1207
+
1208
+ # Adds :arrow_type to schema nodes with a same-named Arrow field (top level, and struct members)
1209
+ def annotate_arrow_types(nodes, fields, depth = 0)
1210
+ by_name = fields.to_h { |f| [f[:name], f] }
1211
+ nodes.each do |n|
1212
+ f = by_name[n[:name]] or next
1213
+ n[:arrow_type] = f[:type]
1214
+ if n[:children] && f[:children] && f[:type].start_with?("struct<") && depth < 32
1215
+ annotate_arrow_types(n[:children], f[:children], depth + 1)
1216
+ end
1217
+ end
1218
+ end
1219
+
1220
+ # Whether the index bound +idx+ excludes values the page's bound +page+ says are present
1221
+ # (index min above the page min, or index max below the page max). Truncated binary bounds
1222
+ # (one a prefix of the other) and values that can't be compared are never reported.
1223
+ def narrower?(idx, page, order, which)
1224
+ return false if idx.nil? || page.nil? || order == :unknown
1225
+ a, b = which == :min ? [page, idx] : [idx, page] # true when a < b
1226
+ if a.is_a?(String) && b.is_a?(String)
1227
+ a = a.b
1228
+ b = b.b
1229
+ return false if a.start_with?(b) || b.start_with?(a)
1230
+ return order == :unsigned ? a < b : false
1231
+ end
1232
+ return false if a.is_a?(Float) && a.nan? || b.is_a?(Float) && b.nan?
1233
+ a = a ? 1 : 0 if a == true || a == false
1234
+ b = b ? 1 : 0 if b == true || b == false
1235
+ (a <=> b) == -1
1236
+ rescue StandardError
1237
+ false
1238
+ end
1239
+
1240
+ def safe_extreme(values, which)
1241
+ vals = values.compact
1242
+ return nil if vals.empty?
1243
+ if vals.all? { |v| v == true || v == false }
1244
+ return which == :min ? vals.all? : vals.any?
1245
+ end
1246
+ return nil unless vals.map(&:class).uniq.size == 1
1247
+ which == :min ? vals.min : vals.max
1248
+ rescue ArgumentError, NoMethodError
1249
+ nil
1250
+ end
1251
+
1252
+ def read_footer
1253
+ @io.seek(0, IO::SEEK_END)
1254
+ @file_size = @io.pos
1255
+ raise FormatError, "File too small to be Parquet (#{@file_size} bytes)" if @file_size < 12
1256
+ tail = read_at(@file_size - 8, 8)
1257
+ raise UnsupportedError, "Encrypted Parquet files are not supported" if tail.byteslice(4, 4) == "PARE"
1258
+ raise FormatError, "Missing PAR1 footer magic" unless tail.byteslice(4, 4) == MAGIC
1259
+ @footer_size = tail.unpack1("V")
1260
+ raise FormatError, "Footer length #{@footer_size} exceeds file size" if @footer_size + 12 > @file_size
1261
+ @metadata = Format::FileMetaData.decode(read_at(footer_offset, @footer_size)).first
1262
+ rescue Thrift::Error => e
1263
+ raise FormatError, "Corrupt file metadata: #{e.message}"
1264
+ end
1265
+
1266
+ # Reads +len+ bytes at +pos+ through a small read-ahead window, so walking many small pages
1267
+ # does not cost a syscall per header
1268
+ def read_at(pos, len)
1269
+ if @window && pos >= @window_pos && pos + len <= @window_pos + @window.bytesize
1270
+ return @window.byteslice(pos - @window_pos, len)
1271
+ end
1272
+ @io.seek(pos)
1273
+ if len >= WINDOW
1274
+ (@io.read(len) || "".b).b
1275
+ else
1276
+ @window_pos = pos
1277
+ @window = (@io.read(WINDOW) || "".b).b
1278
+ @window.byteslice(0, len)
1279
+ end
1280
+ end
1281
+
1282
+ # Decodes the page header at +pos+, reading more bytes when it is larger than the first guess
1283
+ # (page statistics of long strings can make headers big)
1284
+ def read_page_header(pos, limit)
1285
+ want = 256
1286
+ while true
1287
+ avail = [want, limit - pos].min
1288
+ raise FormatError, "no room for a page header" if avail <= 0
1289
+ buf = read_at(pos, avail)
1290
+ begin
1291
+ header, size = Format::PageHeader.decode(buf)
1292
+ return [header, size]
1293
+ rescue Thrift::Error
1294
+ raise if avail < want || want >= 16 * 1024 * 1024
1295
+ want *= 8
1296
+ end
1297
+ end
1298
+ end
1299
+
1300
+ def page_info(index, h, pos, header_size, column)
1301
+ info = PageInfo.new(
1302
+ index: index,
1303
+ type: Format::PageType::NAMES[h.type] || h.type,
1304
+ offset: pos,
1305
+ header_size: header_size,
1306
+ compressed_size: h.compressed_page_size.to_i,
1307
+ uncompressed_size: h.uncompressed_page_size.to_i,
1308
+ crc: h.crc
1309
+ )
1310
+ raise FormatError, "negative page size #{info.compressed_size}" if info.compressed_size.negative?
1311
+ if (d = h.data_page_header)
1312
+ info.num_values = d.num_values
1313
+ info.encoding = Inspector.encoding_name(d.encoding)
1314
+ info.definition_level_encoding = Inspector.encoding_name(d.definition_level_encoding) if d.definition_level_encoding
1315
+ info.repetition_level_encoding = Inspector.encoding_name(d.repetition_level_encoding) if d.repetition_level_encoding
1316
+ info.statistics = decode_statistics(d.statistics, column)
1317
+ info.num_nulls = info.statistics&.null_count
1318
+ # Without repetition every value is a row
1319
+ info.num_rows = d.num_values if column.max_repetition_level.zero?
1320
+ elsif (d = h.data_page_header_v2)
1321
+ info.num_values = d.num_values
1322
+ info.num_nulls = d.num_nulls
1323
+ info.num_rows = d.num_rows
1324
+ info.encoding = Inspector.encoding_name(d.encoding)
1325
+ info.definition_levels_byte_length = d.definition_levels_byte_length
1326
+ info.repetition_levels_byte_length = d.repetition_levels_byte_length
1327
+ info.is_compressed = d.is_compressed.nil? ? true : d.is_compressed
1328
+ info.statistics = decode_statistics(d.statistics, column)
1329
+ elsif (d = h.dictionary_page_header)
1330
+ info.num_values = d.num_values
1331
+ info.encoding = Inspector.encoding_name(d.encoding)
1332
+ info.is_sorted = d.is_sorted
1333
+ end
1334
+ info
1335
+ end
1336
+
1337
+ def describe_key_value(key, value)
1338
+ value = value.to_s
1339
+ h = { key: key, bytesize: value.bytesize }
1340
+ if key == "ARROW:schema"
1341
+ h[:format] = "arrow_schema"
1342
+ h[:value] = value.size > 120 ? "#{value[0, 120]}…" : value
1343
+ if (arrow = arrow_schema)
1344
+ h[:summary] = "Arrow schema, #{arrow[:fields].size} field#{arrow[:fields].size == 1 ? "" : "s"}"
1345
+ h[:arrow_fields] = arrow[:fields].map { |f| f[:name] }
1346
+ meta = arrow[:metadata]&.transform_values { |v| v.size > 4000 ? "#{v[0, 4000]}…" : v }
1347
+ h[:arrow_schema] = arrow.merge(metadata: meta).compact
1348
+ else
1349
+ h[:summary] = "Arrow IPC schema message, base64-encoded (#{value.bytesize} bytes)"
1350
+ h[:arrow_error] = arrow_schema_error
1351
+ names = arrow_field_names(value)
1352
+ h[:arrow_fields] = names if names
1353
+ end
1354
+ elsif value.lstrip.start_with?("{", "[") && value.bytesize < 4 * 1024 * 1024
1355
+ begin
1356
+ h[:json] = JSON.parse(value)
1357
+ h[:format] = "json"
1358
+ h[:summary] = key == "pandas" ? pandas_summary(h[:json]) : "JSON"
1359
+ rescue JSON::ParserError
1360
+ h[:format] = "text"
1361
+ end
1362
+ h[:value] = value.size > 4000 ? "#{value[0, 4000]}…" : value
1363
+ else
1364
+ t = Inspector.text(value.b)
1365
+ h[:format] = t.start_with?("0x") && value.bytesize.positive? ? "binary" : "text"
1366
+ h[:value] = t.size > 4000 ? "#{t[0, 4000]}…" : t
1367
+ end
1368
+ h
1369
+ end
1370
+
1371
+ def pandas_summary(json)
1372
+ return "pandas metadata" unless json.is_a?(Hash)
1373
+ cols = json["columns"]&.size
1374
+ "pandas #{json["pandas_version"]} metadata, #{cols || "?"} columns"
1375
+ end
1376
+
1377
+ # Field names from an Arrow IPC schema message, found by scanning for its flatbuffer strings.
1378
+ # Best effort: only used to label the blob, nil when nothing sensible is found.
1379
+ def arrow_field_names(b64)
1380
+ bytes = b64.unpack1("m")
1381
+ schema_cols = columns.map { |c| c.path.first }.uniq
1382
+ found = schema_cols.select { |n| bytes.include?(n.b) }
1383
+ found.empty? ? nil : found
1384
+ rescue ArgumentError
1385
+ nil
1386
+ end
1387
+ end
1388
+ end