herringbone 0.1.0 → 0.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/README.md +198 -74
- data/bin/herringbone +38 -26
- data/lib/herringbone/active_record.rb +48 -19
- data/lib/herringbone/bloom_filter.rb +370 -0
- data/lib/herringbone/byte_values.rb +29 -0
- data/lib/herringbone/codecs/lz4.rb +66 -3
- data/lib/herringbone/codecs/snappy.rb +132 -3
- data/lib/herringbone/compression.rb +121 -16
- data/lib/herringbone/encodings/delta.rb +76 -19
- data/lib/herringbone/encodings/plain.rb +29 -7
- data/lib/herringbone/encodings/rle.rb +55 -3
- data/lib/herringbone/format.rb +203 -0
- data/lib/herringbone/inspector.rb +1784 -0
- data/lib/herringbone/io_buffer_support.rb +3 -1
- data/lib/herringbone/reader/column_chunk_reader.rb +388 -0
- data/lib/herringbone/reader/column_cursor.rb +225 -0
- data/lib/herringbone/reader/numo.rb +365 -0
- data/lib/herringbone/reader/page_stream.rb +615 -0
- data/lib/herringbone/reader/scan.rb +350 -0
- data/lib/herringbone/reader.rb +585 -272
- data/lib/herringbone/schema.rb +326 -90
- data/lib/herringbone/thrift.rb +155 -4
- data/lib/herringbone/types.rb +177 -42
- data/lib/herringbone/version.rb +2 -1
- data/lib/herringbone/visualizer.rb +1136 -0
- data/lib/herringbone/writer.rb +550 -105
- data/lib/herringbone/xxhash.rb +425 -0
- data/lib/herringbone.rb +64 -9
- metadata +14 -33
|
@@ -0,0 +1,1784 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require "json"
|
|
4
|
+
require "zlib"
|
|
5
|
+
|
|
6
|
+
module Herringbone
|
|
7
|
+
# Examines a Parquet file using only its footer, page headers and page indexes. Values are
|
|
8
|
+
# never decompressed or decoded, so this works for files whose codecs are not installed and
|
|
9
|
+
# stays fast for big files (it seeks from page header to page header).
|
|
10
|
+
#
|
|
11
|
+
# File.open("data.parquet", "rb") do |io|
|
|
12
|
+
# inspector = Herringbone::Inspector.new(io)
|
|
13
|
+
# inspector.summary # => { file_size:, num_rows:, codecs:, ... }
|
|
14
|
+
# inspector.row_groups[0].column("name").pages
|
|
15
|
+
# inspector.to_h # everything, JSON-serializable
|
|
16
|
+
# puts inspector.report # readable text (bin/herringbone inspect)
|
|
17
|
+
# html = inspector.to_html # self-contained HTML page (see Visualizer)
|
|
18
|
+
# end
|
|
19
|
+
#
|
|
20
|
+
# Page headers and indexes are read lazily; #load_all reads them all up front, after which the
|
|
21
|
+
# inspector no longer needs the IO. The IO is never closed. After #verify_checksums, page CRC
|
|
22
|
+
# results are included in every output.
|
|
23
|
+
class Inspector
|
|
24
|
+
# Shorthand for the Parquet physical type constants
|
|
25
|
+
T = Format::Type
|
|
26
|
+
# Magic bytes at the start and end of every (unencrypted) Parquet file
|
|
27
|
+
MAGIC = "PAR1"
|
|
28
|
+
# Size of the read-ahead window used by #read_at, in bytes
|
|
29
|
+
WINDOW = 64 * 1024
|
|
30
|
+
|
|
31
|
+
# Just enough of parquet.thrift's BloomFilterHeader to learn the filter's size
|
|
32
|
+
class BloomFilterHeader < Thrift::Struct
|
|
33
|
+
field 1, :num_bytes, :i32
|
|
34
|
+
end
|
|
35
|
+
|
|
36
|
+
# Decoded min/max statistics (from a column chunk, a page header or a page index entry).
|
|
37
|
+
# +min+/+max+ are converted with the column's type converter (hex Strings when they could
|
|
38
|
+
# not be decoded); +min_exact+/+max_exact+ mirror is_min_value_exact/is_max_value_exact;
|
|
39
|
+
# +source+ says which Thrift fields they came from and +caveat+ why they may be unreliable.
|
|
40
|
+
Stats = Struct.new(:min, :max, :null_count, :distinct_count, :min_exact, :max_exact, :source, :caveat,
|
|
41
|
+
keyword_init: true) do
|
|
42
|
+
# @return [Hash{Symbol => Object}] the set members, JSON-safe (see Inspector.jsonable)
|
|
43
|
+
def to_h = Inspector.jsonable(super.compact)
|
|
44
|
+
end
|
|
45
|
+
|
|
46
|
+
# One page header of a column chunk. +checksum+ is nil until the CRCs are verified
|
|
47
|
+
# (Inspector#verify_checksums), then :ok, :mismatch or :absent (the page has no CRC);
|
|
48
|
+
# +actual_crc+ is then the CRC32 of the page's stored bytes (nil when it has no CRC).
|
|
49
|
+
# +index+ is the page's position in the chunk, +offset+ where its header starts; +type+ is
|
|
50
|
+
# a PageType name such as :DATA_PAGE (the raw Integer when unknown). +first_row_index+ and,
|
|
51
|
+
# for v1 pages of repeated columns, +num_rows+ are only known from the OffsetIndex.
|
|
52
|
+
PageInfo = Struct.new(:index, :type, :offset, :header_size, :compressed_size, :uncompressed_size,
|
|
53
|
+
:num_values, :num_nulls, :num_rows, :first_row_index, :encoding, :definition_level_encoding,
|
|
54
|
+
:repetition_level_encoding, :definition_levels_byte_length, :repetition_levels_byte_length,
|
|
55
|
+
:is_compressed, :is_sorted, :statistics, :crc, :checksum, :actual_crc, keyword_init: true) do
|
|
56
|
+
# @return [Integer] bytes taken by the page in the file: header plus compressed body
|
|
57
|
+
def total_size = header_size + compressed_size
|
|
58
|
+
|
|
59
|
+
# @return [Integer] file offset just past the page's body
|
|
60
|
+
def end_offset = offset + total_size
|
|
61
|
+
|
|
62
|
+
# @return [Integer] file offset where the page's body (after the header) starts
|
|
63
|
+
def body_offset = offset + header_size
|
|
64
|
+
|
|
65
|
+
# @return [Boolean] whether this is a dictionary page
|
|
66
|
+
def dictionary? = type == :DICTIONARY_PAGE
|
|
67
|
+
|
|
68
|
+
# @return [Boolean] whether this is a v1 or v2 data page
|
|
69
|
+
def data? = type == :DATA_PAGE || type == :DATA_PAGE_V2
|
|
70
|
+
|
|
71
|
+
# The CRC from the header as an unsigned 32-bit value (Thrift stores it as a signed i32)
|
|
72
|
+
# @return [Integer, nil] nil when the header has no CRC
|
|
73
|
+
def expected_crc = crc && (crc & 0xFFFF_FFFF)
|
|
74
|
+
|
|
75
|
+
# @return [Hash{Symbol => Object}] the set members, JSON-safe; +:crc+ becomes a Boolean
|
|
76
|
+
# saying whether the header has a CRC, +:actual_crc+ is left out
|
|
77
|
+
def to_h
|
|
78
|
+
h = super
|
|
79
|
+
h.delete(:actual_crc)
|
|
80
|
+
h[:statistics] = statistics&.to_h
|
|
81
|
+
h[:crc] = !crc.nil?
|
|
82
|
+
Inspector.jsonable(h.compact)
|
|
83
|
+
end
|
|
84
|
+
end
|
|
85
|
+
|
|
86
|
+
# ColumnIndex of one column chunk, with min/max decoded per page (nil for all-null pages)
|
|
87
|
+
ColumnIndexInfo = Struct.new(:offset, :length, :null_pages, :min_values, :max_values, :boundary_order,
|
|
88
|
+
:null_counts, :repetition_level_histograms, :definition_level_histograms, keyword_init: true) do
|
|
89
|
+
# @return [Hash{Symbol => Object}] the set members, JSON-safe (see Inspector.jsonable)
|
|
90
|
+
def to_h = Inspector.jsonable(super.compact)
|
|
91
|
+
end
|
|
92
|
+
|
|
93
|
+
# One OffsetIndex entry: where a data page starts, its size (header included) and the index
|
|
94
|
+
# of its first row within the row group
|
|
95
|
+
PageLocation = Struct.new(:offset, :compressed_page_size, :first_row_index, keyword_init: true)
|
|
96
|
+
|
|
97
|
+
# OffsetIndex of one column chunk: its own +offset+/+length+ in the file, the PageLocations
|
|
98
|
+
# of its data pages and the optional per-page unencoded BYTE_ARRAY sizes
|
|
99
|
+
OffsetIndexInfo = Struct.new(:offset, :length, :page_locations, :unencoded_byte_array_data_bytes,
|
|
100
|
+
keyword_init: true) do
|
|
101
|
+
# @return [Hash{Symbol => Object}] the set members, with +:page_locations+ as Hashes
|
|
102
|
+
def to_h
|
|
103
|
+
h = super
|
|
104
|
+
h[:page_locations] = page_locations.map(&:to_h)
|
|
105
|
+
h.compact
|
|
106
|
+
end
|
|
107
|
+
end
|
|
108
|
+
|
|
109
|
+
# One column chunk of a row group
|
|
110
|
+
class ColumnChunkInfo
|
|
111
|
+
attr_reader :inspector, :row_group, :column, :chunk, :meta, :error
|
|
112
|
+
|
|
113
|
+
# @param inspector [Inspector] owner, used to read page headers and indexes lazily
|
|
114
|
+
# @param row_group [RowGroupInfo] row group the chunk belongs to
|
|
115
|
+
# @param column [Schema::Column] leaf column the chunk stores
|
|
116
|
+
# @param chunk [Format::ColumnChunk] chunk as decoded from the footer
|
|
117
|
+
def initialize(inspector, row_group, column, chunk)
|
|
118
|
+
@inspector = inspector
|
|
119
|
+
@row_group = row_group
|
|
120
|
+
@column = column
|
|
121
|
+
@chunk = chunk
|
|
122
|
+
@meta = chunk.meta_data
|
|
123
|
+
end
|
|
124
|
+
|
|
125
|
+
# @return [String] dotted path of the column, e.g. +"address.city"+
|
|
126
|
+
def path = @column.dotted_path
|
|
127
|
+
|
|
128
|
+
# @return [Integer] position of the column among the schema's leaf columns
|
|
129
|
+
def column_index_number = @column.index
|
|
130
|
+
|
|
131
|
+
# @return [Symbol, Integer] codec name such as :SNAPPY (the raw Integer when unknown)
|
|
132
|
+
def codec = Format::Codec::NAMES[@meta.codec] || @meta.codec
|
|
133
|
+
|
|
134
|
+
# @return [Array<String>] encoding names listed in the column metadata
|
|
135
|
+
def encodings = (@meta.encodings || []).map { |e| Inspector.encoding_name(e) }
|
|
136
|
+
|
|
137
|
+
# @return [Integer] values in the chunk, nulls and repeated entries included
|
|
138
|
+
def num_values = @meta.num_values
|
|
139
|
+
|
|
140
|
+
# @return [Integer] total_compressed_size from the column metadata (page headers included)
|
|
141
|
+
def compressed_size = @meta.total_compressed_size
|
|
142
|
+
|
|
143
|
+
# @return [Integer] total_uncompressed_size from the column metadata (page headers included)
|
|
144
|
+
def uncompressed_size = @meta.total_uncompressed_size
|
|
145
|
+
|
|
146
|
+
# @return [Integer] offset of the first data page, as declared in the column metadata
|
|
147
|
+
def data_page_offset = @meta.data_page_offset
|
|
148
|
+
|
|
149
|
+
# @return [String, nil] path of the file holding the chunk when it is not stored in this one
|
|
150
|
+
def external_file = @chunk.file_path
|
|
151
|
+
|
|
152
|
+
# @return [Float, nil] uncompressed size divided by compressed size; nil when nothing is compressed
|
|
153
|
+
def compression_ratio
|
|
154
|
+
compressed_size.to_i.positive? ? uncompressed_size.to_f / compressed_size : nil
|
|
155
|
+
end
|
|
156
|
+
|
|
157
|
+
# Some writers store 0 when there is no dictionary page, and some store a data_page_offset
|
|
158
|
+
# of 0 for empty chunks (which would point at the magic bytes)
|
|
159
|
+
# @return [Integer, nil] the declared dictionary page offset, nil when absent or implausible
|
|
160
|
+
def dictionary_page_offset
|
|
161
|
+
d = @meta.dictionary_page_offset
|
|
162
|
+
(d && d >= 4 && (d < @meta.data_page_offset.to_i || @meta.data_page_offset.to_i < 4)) ? d : nil
|
|
163
|
+
end
|
|
164
|
+
|
|
165
|
+
# Where the chunk's first page starts
|
|
166
|
+
# @return [Integer] file offset
|
|
167
|
+
def start_offset = dictionary_page_offset || data_page_offset
|
|
168
|
+
|
|
169
|
+
# The end according to the metadata; some writers under-report it (see #end_offset)
|
|
170
|
+
# @return [Integer] file offset just past the chunk
|
|
171
|
+
def declared_end_offset = start_offset + compressed_size
|
|
172
|
+
|
|
173
|
+
# The end of the last page actually found (the declared end if the pages could not be walked)
|
|
174
|
+
# @return [Integer] file offset just past the chunk, never before #declared_end_offset
|
|
175
|
+
def end_offset
|
|
176
|
+
last = pages.last
|
|
177
|
+
last ? [last.end_offset, declared_end_offset].max : declared_end_offset
|
|
178
|
+
end
|
|
179
|
+
|
|
180
|
+
# Page counts per page type and encoding, from the column metadata's encoding_stats
|
|
181
|
+
# @return [Array<Hash{Symbol => Object}>] each { page_type:, encoding:, count: }; empty when
|
|
182
|
+
# the writer stored none
|
|
183
|
+
def encoding_stats
|
|
184
|
+
(@meta.encoding_stats || []).map do |s|
|
|
185
|
+
{page_type: Format::PageType::NAMES[s.page_type] || s.page_type,
|
|
186
|
+
encoding: Inspector.encoding_name(s.encoding), count: s.count}
|
|
187
|
+
end
|
|
188
|
+
end
|
|
189
|
+
|
|
190
|
+
# Chunk-level statistics from the column metadata, decoded (memoized)
|
|
191
|
+
# @return [Stats, nil] nil when the chunk has no statistics
|
|
192
|
+
def statistics
|
|
193
|
+
return @statistics if defined?(@statistics)
|
|
194
|
+
@statistics = @inspector.decode_statistics(@meta.statistics, @column)
|
|
195
|
+
end
|
|
196
|
+
|
|
197
|
+
# @return [Hash{Symbol => Object}, nil] the SizeStatistics (unencoded byte array sizes and
|
|
198
|
+
# level histograms) as a Hash, nil when absent
|
|
199
|
+
def size_statistics
|
|
200
|
+
s = @meta.size_statistics
|
|
201
|
+
s&.to_h
|
|
202
|
+
end
|
|
203
|
+
|
|
204
|
+
# @return [Hash{String => String}] the chunk's own key/value metadata (rarely used by writers)
|
|
205
|
+
def key_value_metadata
|
|
206
|
+
(@meta.key_value_metadata || []).to_h { |kv| [kv.key, kv.value] }
|
|
207
|
+
end
|
|
208
|
+
|
|
209
|
+
# @return [Integer, nil] file offset of the chunk's bloom filter, nil when it has none
|
|
210
|
+
def bloom_filter_offset = @meta.bloom_filter_offset
|
|
211
|
+
|
|
212
|
+
# Bytes taken by the bloom filter (header + bitset); read from its header when the footer has no length
|
|
213
|
+
# @return [Integer, nil] nil when there is no bloom filter or its header can't be decoded
|
|
214
|
+
def bloom_filter_length
|
|
215
|
+
return nil unless bloom_filter_offset
|
|
216
|
+
return @meta.bloom_filter_length if @meta.bloom_filter_length
|
|
217
|
+
return @bloom_filter_length if defined?(@bloom_filter_length)
|
|
218
|
+
@bloom_filter_length = @inspector.bloom_filter_size(bloom_filter_offset)
|
|
219
|
+
end
|
|
220
|
+
|
|
221
|
+
# @return [Array(Integer, Integer), nil] [offset, length] of the chunk's ColumnIndex, nil when
|
|
222
|
+
# it has none
|
|
223
|
+
def column_index_range
|
|
224
|
+
o = @chunk.column_index_offset
|
|
225
|
+
(o && @chunk.column_index_length) ? [o, @chunk.column_index_length] : nil
|
|
226
|
+
end
|
|
227
|
+
|
|
228
|
+
# @return [Array(Integer, Integer), nil] [offset, length] of the chunk's OffsetIndex, nil when
|
|
229
|
+
# it has none
|
|
230
|
+
def offset_index_range
|
|
231
|
+
o = @chunk.offset_index_offset
|
|
232
|
+
(o && @chunk.offset_index_length) ? [o, @chunk.offset_index_length] : nil
|
|
233
|
+
end
|
|
234
|
+
|
|
235
|
+
# All page headers, in file order. Errors while walking (corrupt or truncated headers) end
|
|
236
|
+
# the walk; #error then says what went wrong and #pages holds the pages found before it.
|
|
237
|
+
# @return [Array<PageInfo>]
|
|
238
|
+
def pages
|
|
239
|
+
@pages ||= begin
|
|
240
|
+
list, @error = @inspector.walk_pages(self)
|
|
241
|
+
apply_offset_index(list)
|
|
242
|
+
list
|
|
243
|
+
end
|
|
244
|
+
end
|
|
245
|
+
|
|
246
|
+
# @return [Array<PageInfo>] the v1 and v2 data pages, in file order
|
|
247
|
+
def data_pages = pages.select(&:data?)
|
|
248
|
+
|
|
249
|
+
# @return [PageInfo, nil] the dictionary page, nil when the chunk has none
|
|
250
|
+
def dictionary_page = pages.find(&:dictionary?)
|
|
251
|
+
|
|
252
|
+
# Number of entries in the dictionary (from its page header)
|
|
253
|
+
# @return [Integer, nil] nil when the chunk has no dictionary page
|
|
254
|
+
def dictionary_size = dictionary_page&.num_values
|
|
255
|
+
|
|
256
|
+
# The chunk's ColumnIndex, read and decoded on first use
|
|
257
|
+
# @return [ColumnIndexInfo, nil] nil when there is none or it is corrupt
|
|
258
|
+
def column_index
|
|
259
|
+
return @column_index if defined?(@column_index)
|
|
260
|
+
@column_index = @inspector.read_column_index(self)
|
|
261
|
+
end
|
|
262
|
+
|
|
263
|
+
# The chunk's OffsetIndex, read and decoded on first use
|
|
264
|
+
# @return [OffsetIndexInfo, nil] nil when there is none or it is corrupt
|
|
265
|
+
def offset_index
|
|
266
|
+
return @offset_index if defined?(@offset_index)
|
|
267
|
+
@offset_index = @inspector.read_offset_index(self)
|
|
268
|
+
end
|
|
269
|
+
|
|
270
|
+
# Null count from the chunk statistics, else the sum over the data page headers
|
|
271
|
+
# @return [Integer, nil] nil when neither the statistics nor every data page has it
|
|
272
|
+
def null_count
|
|
273
|
+
return statistics.null_count if statistics&.null_count
|
|
274
|
+
counts = data_pages.map(&:num_nulls)
|
|
275
|
+
counts.all? ? counts.sum : nil
|
|
276
|
+
end
|
|
277
|
+
|
|
278
|
+
# Reads every page body (compressed bytes, as stored) and checks it against the CRC in its
|
|
279
|
+
# page header. Sets PageInfo#checksum on each page and returns the pages' statuses.
|
|
280
|
+
# @return [Array<Symbol>] :ok, :mismatch or :absent per page, in #pages order
|
|
281
|
+
def verify_checksums
|
|
282
|
+
pages.map { |p| p.checksum = @inspector.page_checksum(p) }
|
|
283
|
+
end
|
|
284
|
+
|
|
285
|
+
# Page statistics (from page headers) that disagree with the ColumnIndex entry for the
|
|
286
|
+
# same page. Each: { page:, data_page:, field:, page_value:, index_value: } where +page+
|
|
287
|
+
# is the page's position in #pages, +data_page+ its position among the data pages (and
|
|
288
|
+
# in the ColumnIndex), +field+ one of :min, :max, :null_count, :null_page, :page_count.
|
|
289
|
+
# A ColumnIndex bound that is wider than the page's (e.g. a truncated string prefix) is
|
|
290
|
+
# allowed; one that is narrower, or a differing null count, is reported. Empty when the
|
|
291
|
+
# chunk has no ColumnIndex or its pages carry no statistics.
|
|
292
|
+
# @return [Array<Hash{Symbol => Object}>]
|
|
293
|
+
def index_mismatches
|
|
294
|
+
@index_mismatches ||= @inspector.compare_page_index(self)
|
|
295
|
+
end
|
|
296
|
+
|
|
297
|
+
# Everything known about the chunk, including its pages and page indexes (reads them if
|
|
298
|
+
# they were not read yet)
|
|
299
|
+
# @return [Hash{Symbol => Object}] JSON-safe; keys without a value are left out
|
|
300
|
+
def to_h
|
|
301
|
+
Inspector.jsonable({
|
|
302
|
+
path: path,
|
|
303
|
+
column: @column.index,
|
|
304
|
+
type: Inspector.type_name(@column),
|
|
305
|
+
codec: codec,
|
|
306
|
+
encodings: encodings,
|
|
307
|
+
encoding_stats: encoding_stats,
|
|
308
|
+
num_values: num_values,
|
|
309
|
+
null_count: null_count,
|
|
310
|
+
compressed_size: compressed_size,
|
|
311
|
+
uncompressed_size: uncompressed_size,
|
|
312
|
+
compression_ratio: compression_ratio&.round(4),
|
|
313
|
+
file_offset: @chunk.file_offset,
|
|
314
|
+
start_offset: start_offset,
|
|
315
|
+
end_offset: end_offset,
|
|
316
|
+
data_page_offset: data_page_offset,
|
|
317
|
+
dictionary_page_offset: dictionary_page_offset,
|
|
318
|
+
dictionary: dictionary_page && {offset: dictionary_page.offset, num_values: dictionary_page.num_values,
|
|
319
|
+
compressed_size: dictionary_page.compressed_size, uncompressed_size: dictionary_page.uncompressed_size,
|
|
320
|
+
is_sorted: dictionary_page.is_sorted},
|
|
321
|
+
statistics: statistics&.to_h,
|
|
322
|
+
size_statistics: size_statistics,
|
|
323
|
+
key_value_metadata: key_value_metadata.empty? ? nil : key_value_metadata,
|
|
324
|
+
bloom_filter: bloom_filter_offset && {offset: bloom_filter_offset, length: bloom_filter_length},
|
|
325
|
+
column_index_offset: column_index_range&.first,
|
|
326
|
+
column_index_length: column_index_range&.last,
|
|
327
|
+
offset_index_offset: offset_index_range&.first,
|
|
328
|
+
offset_index_length: offset_index_range&.last,
|
|
329
|
+
external_file: external_file,
|
|
330
|
+
num_pages: pages.size,
|
|
331
|
+
num_data_pages: data_pages.size,
|
|
332
|
+
index_mismatches: index_mismatches.empty? ? nil : index_mismatches,
|
|
333
|
+
error: error,
|
|
334
|
+
pages: pages.map(&:to_h),
|
|
335
|
+
column_index: column_index&.to_h,
|
|
336
|
+
offset_index: offset_index&.to_h
|
|
337
|
+
}.compact)
|
|
338
|
+
end
|
|
339
|
+
|
|
340
|
+
# @return [String] short description: path, row group, codec and sizes
|
|
341
|
+
def inspect
|
|
342
|
+
"#<#{self.class.name} #{path} rg=#{@row_group.index} #{codec} #{compressed_size}/#{uncompressed_size} bytes>"
|
|
343
|
+
end
|
|
344
|
+
|
|
345
|
+
private
|
|
346
|
+
|
|
347
|
+
# Page row counts come from the offset index when there is one (v1 pages don't carry them)
|
|
348
|
+
# Sets +first_row_index+ on the data pages the index points at, and +num_rows+ where the
|
|
349
|
+
# page header did not provide it.
|
|
350
|
+
# @param list [Array<PageInfo>] pages of this chunk, updated in place
|
|
351
|
+
# @return [void]
|
|
352
|
+
def apply_offset_index(list)
|
|
353
|
+
oi = offset_index or return
|
|
354
|
+
by_offset = list.select(&:data?).to_h { |p| [p.offset, p] }
|
|
355
|
+
locs = oi.page_locations
|
|
356
|
+
locs.each_with_index do |loc, i|
|
|
357
|
+
page = by_offset[loc.offset] or next
|
|
358
|
+
page.first_row_index = loc.first_row_index
|
|
359
|
+
next_first = locs[i + 1]&.first_row_index || @row_group.num_rows
|
|
360
|
+
page.num_rows ||= next_first - loc.first_row_index
|
|
361
|
+
end
|
|
362
|
+
end
|
|
363
|
+
end
|
|
364
|
+
|
|
365
|
+
# One row group
|
|
366
|
+
class RowGroupInfo
|
|
367
|
+
attr_reader :index, :row_group, :columns
|
|
368
|
+
|
|
369
|
+
# @param inspector [Inspector] owner, passed on to the column chunks
|
|
370
|
+
# @param index [Integer] position of the row group in the file
|
|
371
|
+
# @param row_group [Format::RowGroup] row group as decoded from the footer
|
|
372
|
+
# @param first_row [Integer] file-wide index of the row group's first row
|
|
373
|
+
# @raise [FormatError] when the row group has more column chunks than the schema has leaves
|
|
374
|
+
def initialize(inspector, index, row_group, first_row)
|
|
375
|
+
@index = index
|
|
376
|
+
@row_group = row_group
|
|
377
|
+
@first_row = first_row
|
|
378
|
+
leaves = inspector.schema.columns
|
|
379
|
+
@columns = (row_group.columns || []).each_with_index.map do |cc, i|
|
|
380
|
+
column = leaves[i] or raise FormatError, "Row group #{index} has more column chunks than the schema has columns"
|
|
381
|
+
ColumnChunkInfo.new(inspector, self, column, cc)
|
|
382
|
+
end
|
|
383
|
+
end
|
|
384
|
+
|
|
385
|
+
# @return [Integer] rows in the row group
|
|
386
|
+
def num_rows = @row_group.num_rows
|
|
387
|
+
# @return [Integer] file-wide index of the row group's first row
|
|
388
|
+
attr_reader :first_row
|
|
389
|
+
|
|
390
|
+
# @return [Integer] total_byte_size from the footer (uncompressed size of all column data)
|
|
391
|
+
def total_byte_size = @row_group.total_byte_size
|
|
392
|
+
|
|
393
|
+
# @return [Integer] total_compressed_size from the footer, else the sum over the chunks
|
|
394
|
+
def compressed_size = @row_group.total_compressed_size || @columns.sum(&:compressed_size)
|
|
395
|
+
|
|
396
|
+
# @return [Integer] sum of the chunks' total_uncompressed_size
|
|
397
|
+
def uncompressed_size = @columns.sum { |c| c.uncompressed_size.to_i }
|
|
398
|
+
|
|
399
|
+
# @return [Integer, nil] lowest start offset of the row group's chunks; nil without chunks
|
|
400
|
+
def start_offset = @columns.map(&:start_offset).min
|
|
401
|
+
|
|
402
|
+
# @return [Integer, nil] highest end offset of the row group's chunks; nil without chunks
|
|
403
|
+
def end_offset = @columns.map(&:end_offset).max
|
|
404
|
+
|
|
405
|
+
# @param path [String, Array<String>] dotted path (+"a.b"+) or path segments (+["a", "b"]+)
|
|
406
|
+
# @return [ColumnChunkInfo, nil] the chunk of that column, nil when there is none
|
|
407
|
+
def column(path) = @columns.find { |c| c.path == path.to_s || c.column.path == Array(path) }
|
|
408
|
+
|
|
409
|
+
# The row group's declared sort order
|
|
410
|
+
# @return [Array<Hash{Symbol => Object}>] each { column:, descending:, nulls_first: }, where
|
|
411
|
+
# +column+ is the column's path (its index when out of range)
|
|
412
|
+
def sorting_columns
|
|
413
|
+
(@row_group.sorting_columns || []).map do |s|
|
|
414
|
+
{column: @columns[s.column_idx]&.path || s.column_idx, descending: s.descending, nulls_first: s.nulls_first}
|
|
415
|
+
end
|
|
416
|
+
end
|
|
417
|
+
|
|
418
|
+
# Everything known about the row group, with each column chunk's #to_h
|
|
419
|
+
# @return [Hash{Symbol => Object}] JSON-safe; keys without a value are left out
|
|
420
|
+
def to_h
|
|
421
|
+
Inspector.jsonable({
|
|
422
|
+
index: index,
|
|
423
|
+
ordinal: @row_group.ordinal,
|
|
424
|
+
num_rows: num_rows,
|
|
425
|
+
first_row: first_row,
|
|
426
|
+
total_byte_size: total_byte_size,
|
|
427
|
+
compressed_size: compressed_size,
|
|
428
|
+
uncompressed_size: uncompressed_size,
|
|
429
|
+
file_offset: @row_group.file_offset,
|
|
430
|
+
start_offset: start_offset,
|
|
431
|
+
end_offset: end_offset,
|
|
432
|
+
sorting_columns: sorting_columns,
|
|
433
|
+
columns: @columns.map(&:to_h)
|
|
434
|
+
}.compact)
|
|
435
|
+
end
|
|
436
|
+
|
|
437
|
+
# @return [String] short description: index, row count and number of columns
|
|
438
|
+
def inspect
|
|
439
|
+
"#<#{self.class.name} #{index} rows=#{num_rows} columns=#{@columns.size}>"
|
|
440
|
+
end
|
|
441
|
+
end
|
|
442
|
+
|
|
443
|
+
attr_reader :metadata, :schema, :file_size, :footer_size
|
|
444
|
+
|
|
445
|
+
# A label for the file (the basename of the IO's path, when it has one)
|
|
446
|
+
attr_reader :name
|
|
447
|
+
|
|
448
|
+
# +io+ is a random-access IO (responds to #seek and #read, e.g. File.open(path, "rb")). It is
|
|
449
|
+
# left open.
|
|
450
|
+
# @param io [IO] random-access IO positioned anywhere; only the footer is read here
|
|
451
|
+
# @raise [ArgumentError] when +io+ does not support #seek and #read
|
|
452
|
+
# @raise [FormatError] when the file is too small, lacks the magic bytes or has a corrupt footer
|
|
453
|
+
# @raise [UnsupportedError] when the file is encrypted (PARE magic)
|
|
454
|
+
def initialize(io)
|
|
455
|
+
@io = io
|
|
456
|
+
unless @io.respond_to?(:read) && @io.respond_to?(:seek)
|
|
457
|
+
raise ArgumentError, "Herringbone::Inspector expects an IO that supports #seek and #read " \
|
|
458
|
+
"(e.g. File.open(path, \"rb\")), got #{io.class}"
|
|
459
|
+
end
|
|
460
|
+
@name = (@io.respond_to?(:path) && @io.path) ? File.basename(@io.path.to_s) : nil
|
|
461
|
+
read_footer
|
|
462
|
+
@schema = Schema.from_elements(@metadata.schema)
|
|
463
|
+
end
|
|
464
|
+
|
|
465
|
+
# @return [Integer] row count declared in the footer
|
|
466
|
+
def num_rows = @metadata.num_rows
|
|
467
|
+
|
|
468
|
+
# @return [String, nil] the writer's created_by string, e.g. +"parquet-cpp-arrow version 15.0.0"+
|
|
469
|
+
def created_by = @metadata.created_by
|
|
470
|
+
|
|
471
|
+
# @return [Integer] format version from the footer (1 or 2; says little about the features used)
|
|
472
|
+
def version = @metadata.version
|
|
473
|
+
|
|
474
|
+
# @return [Array<Schema::Column>] the schema's leaf columns
|
|
475
|
+
def columns = @schema.columns
|
|
476
|
+
|
|
477
|
+
# @return [Integer] file offset where the Thrift-encoded FileMetaData starts
|
|
478
|
+
def footer_offset = @file_size - 8 - @footer_size
|
|
479
|
+
|
|
480
|
+
# @return [Array<RowGroupInfo>] the row groups, in footer order (built on first use)
|
|
481
|
+
def row_groups
|
|
482
|
+
@row_groups ||= begin
|
|
483
|
+
first = 0
|
|
484
|
+
(@metadata.row_groups || []).each_with_index.map do |rg, i|
|
|
485
|
+
info = RowGroupInfo.new(self, i, rg, first)
|
|
486
|
+
first += rg.num_rows.to_i
|
|
487
|
+
info
|
|
488
|
+
end
|
|
489
|
+
end
|
|
490
|
+
end
|
|
491
|
+
|
|
492
|
+
# @return [Array<ColumnChunkInfo>] the column chunks of all row groups, row group by row group
|
|
493
|
+
def column_chunks = row_groups.flat_map(&:columns)
|
|
494
|
+
|
|
495
|
+
# Walks every page header and page index now (e.g. before closing the file)
|
|
496
|
+
# @return [Inspector] self
|
|
497
|
+
def load_all
|
|
498
|
+
column_chunks.each do |c|
|
|
499
|
+
c.pages
|
|
500
|
+
c.column_index
|
|
501
|
+
c.offset_index
|
|
502
|
+
c.bloom_filter_length
|
|
503
|
+
end
|
|
504
|
+
self
|
|
505
|
+
end
|
|
506
|
+
|
|
507
|
+
# @return [Boolean] whether any column chunk has a ColumnIndex or an OffsetIndex
|
|
508
|
+
def page_index? = column_chunks.any? { |c| c.column_index_range || c.offset_index_range }
|
|
509
|
+
|
|
510
|
+
# @return [Boolean] whether any column chunk has a bloom filter
|
|
511
|
+
def bloom_filters? = column_chunks.any?(&:bloom_filter_offset)
|
|
512
|
+
|
|
513
|
+
# The file-level key/value metadata, each entry described: ARROW:schema is decoded, JSON
|
|
514
|
+
# values are parsed (pandas metadata summarized), binary values are shown as hex
|
|
515
|
+
# @return [Array<Hash{Symbol => Object}>] each with :key, :bytesize, :format ("arrow_schema",
|
|
516
|
+
# "json", "text" or "binary"), :value (truncated) and, depending on the format, :summary,
|
|
517
|
+
# :json, :arrow_schema, :arrow_fields, :arrow_error
|
|
518
|
+
def key_value_metadata
|
|
519
|
+
(@metadata.key_value_metadata || []).map { |kv| describe_key_value(kv.key, kv.value) }
|
|
520
|
+
end
|
|
521
|
+
|
|
522
|
+
# @return [Array<String>, nil] "TYPE_DEFINED_ORDER" or "UNKNOWN" per leaf column; nil when the
|
|
523
|
+
# footer has no column_orders
|
|
524
|
+
def column_orders
|
|
525
|
+
orders = @metadata.column_orders or return nil
|
|
526
|
+
orders.map { |o| o.type_order ? "TYPE_DEFINED_ORDER" : "UNKNOWN" }
|
|
527
|
+
end
|
|
528
|
+
|
|
529
|
+
# File-wide facts: sizes, row and column counts, codecs, whether page indexes and bloom
|
|
530
|
+
# filters are present, and (after #verify_checksums) the CRC tallies under :checksums.
|
|
531
|
+
# Walks no page headers.
|
|
532
|
+
# @return [Hash{Symbol => Object}]
|
|
533
|
+
def summary
|
|
534
|
+
codecs = column_chunks.map(&:codec).uniq
|
|
535
|
+
{
|
|
536
|
+
name: @name,
|
|
537
|
+
file_size: @file_size,
|
|
538
|
+
footer_size: @footer_size,
|
|
539
|
+
footer_offset: footer_offset,
|
|
540
|
+
format_version: version,
|
|
541
|
+
created_by: created_by,
|
|
542
|
+
num_rows: num_rows,
|
|
543
|
+
num_row_groups: row_groups.size,
|
|
544
|
+
num_columns: columns.size,
|
|
545
|
+
codecs: codecs,
|
|
546
|
+
compressed_size: column_chunks.sum { |c| c.compressed_size.to_i },
|
|
547
|
+
uncompressed_size: column_chunks.sum { |c| c.uncompressed_size.to_i },
|
|
548
|
+
page_index: page_index?,
|
|
549
|
+
bloom_filters: bloom_filters?,
|
|
550
|
+
column_orders: column_orders
|
|
551
|
+
}.tap { |h| h[:checksums] = checksum_summary.except(:mismatches) if checksums_verified? }
|
|
552
|
+
end
|
|
553
|
+
|
|
554
|
+
# Reads every page body and checks it against the CRC32 in its page header (the CRC covers
|
|
555
|
+
# the page data as stored: compressed, and for v2 pages the levels plus the compressed
|
|
556
|
+
# values). Nothing is decompressed. Sets PageInfo#checksum on every page and returns
|
|
557
|
+
# #checksum_summary. Needs the IO, so call it before closing the file.
|
|
558
|
+
# @return [Hash{Symbol => Object}] see #checksum_summary
|
|
559
|
+
def verify_checksums
|
|
560
|
+
column_chunks.each(&:verify_checksums)
|
|
561
|
+
@checksums_verified = true
|
|
562
|
+
checksum_summary
|
|
563
|
+
end
|
|
564
|
+
|
|
565
|
+
# @return [Boolean] whether #verify_checksums has run
|
|
566
|
+
def checksums_verified? = @checksums_verified == true
|
|
567
|
+
|
|
568
|
+
# After #verify_checksums: { ok:, mismatch:, absent:, mismatches: [{ row_group:, column:, page:, type:, offset:, crc:, actual: }] }
|
|
569
|
+
# The counts are pages per status; in each mismatch +crc+ is the CRC from the page header and
|
|
570
|
+
# +actual+ the CRC32 of the stored bytes (kept from the verification, so the IO is not needed).
|
|
571
|
+
# @return [Hash{Symbol => Object}, nil] nil before #verify_checksums
|
|
572
|
+
def checksum_summary
|
|
573
|
+
return nil unless checksums_verified?
|
|
574
|
+
all = column_chunks.flat_map { |c| c.pages.map { |p| [c, p] } }
|
|
575
|
+
tally = all.map { |_, p| p.checksum }.tally
|
|
576
|
+
{
|
|
577
|
+
ok: tally.fetch(:ok, 0), mismatch: tally.fetch(:mismatch, 0), absent: tally.fetch(:absent, 0),
|
|
578
|
+
mismatches: all.select { |_, p| p.checksum == :mismatch }.map do |c, p|
|
|
579
|
+
{row_group: c.row_group.index, column: c.path, page: p.index, type: p.type, offset: p.offset,
|
|
580
|
+
crc: p.expected_crc, actual: p.actual_crc}
|
|
581
|
+
end
|
|
582
|
+
}
|
|
583
|
+
end
|
|
584
|
+
|
|
585
|
+
# Every disagreement between page header statistics and the ColumnIndex, across the file,
|
|
586
|
+
# each with :row_group and :column added (see ColumnChunkInfo#index_mismatches)
|
|
587
|
+
# @return [Array<Hash{Symbol => Object}>]
|
|
588
|
+
def index_mismatches
|
|
589
|
+
column_chunks.flat_map do |c|
|
|
590
|
+
c.index_mismatches.map { |m| {row_group: c.row_group.index, column: c.path}.merge(m) }
|
|
591
|
+
end
|
|
592
|
+
end
|
|
593
|
+
|
|
594
|
+
# The decoded ARROW:schema key/value (see ArrowSchema.decode); nil when the file has none or
|
|
595
|
+
# it could not be decoded, and #arrow_schema_error then says why
|
|
596
|
+
# @return [Hash{Symbol => Object}, nil] see ArrowSchema.decode
|
|
597
|
+
def arrow_schema
|
|
598
|
+
return @arrow_schema if defined?(@arrow_schema)
|
|
599
|
+
@arrow_schema_error = nil
|
|
600
|
+
kv = (@metadata.key_value_metadata || []).find { |x| x.key == "ARROW:schema" }
|
|
601
|
+
@arrow_schema = kv && begin
|
|
602
|
+
ArrowSchema.decode(kv.value)
|
|
603
|
+
rescue => e
|
|
604
|
+
@arrow_schema_error = "could not decode ARROW:schema: #{e.message}"
|
|
605
|
+
nil
|
|
606
|
+
end
|
|
607
|
+
end
|
|
608
|
+
|
|
609
|
+
# @return [String, nil] why the ARROW:schema value could not be decoded; nil when it decoded
|
|
610
|
+
# fine or the file has none
|
|
611
|
+
def arrow_schema_error
|
|
612
|
+
arrow_schema
|
|
613
|
+
@arrow_schema_error
|
|
614
|
+
end
|
|
615
|
+
|
|
616
|
+
# Schema tree: Hashes with name, repetition, types, levels (leaves) and children (groups)
|
|
617
|
+
# Nodes that match a field of the ARROW:schema also get its type as :arrow_type.
|
|
618
|
+
# @return [Array<Hash{Symbol => Object}>] the root's children
|
|
619
|
+
def schema_tree
|
|
620
|
+
leaf_by_node = @schema.columns.to_h { |c| [c.node, c] }
|
|
621
|
+
build = lambda do |node|
|
|
622
|
+
h = {name: node.name, repetition: node.repetition}
|
|
623
|
+
if node.leaf?
|
|
624
|
+
col = leaf_by_node[node]
|
|
625
|
+
h[:physical_type] = T::NAMES[node.type]&.to_s
|
|
626
|
+
h[:type_length] = node.type_length
|
|
627
|
+
h[:column] = col.index
|
|
628
|
+
h[:path] = col.dotted_path
|
|
629
|
+
h[:max_definition_level] = col.max_definition_level
|
|
630
|
+
h[:max_repetition_level] = col.max_repetition_level
|
|
631
|
+
end
|
|
632
|
+
h[:logical_type] = Inspector.logical_type_name(node)
|
|
633
|
+
h[:converted_type] = Format::ConvertedType::NAMES[node.converted_type]&.to_s
|
|
634
|
+
h[:field_id] = node.field_id
|
|
635
|
+
h[:children] = node.children.map { |c| build.call(c) } if node.group?
|
|
636
|
+
h.compact
|
|
637
|
+
end
|
|
638
|
+
tree = @schema.root.children.map { |c| build.call(c) }
|
|
639
|
+
annotate_arrow_types(tree, arrow_schema[:fields]) if arrow_schema
|
|
640
|
+
tree
|
|
641
|
+
end
|
|
642
|
+
|
|
643
|
+
# Per leaf column: sums over all row groups, plus overall min/max where comparable
|
|
644
|
+
# Walks every page header (for the page counts).
|
|
645
|
+
# @return [Array<Hash{Symbol => Object}>] one per leaf column, in schema order
|
|
646
|
+
def column_totals
|
|
647
|
+
columns.map do |col|
|
|
648
|
+
chunks = row_groups.map { |rg| rg.columns[col.index] }.compact
|
|
649
|
+
compressed = chunks.sum { |c| c.compressed_size.to_i }
|
|
650
|
+
uncompressed = chunks.sum { |c| c.uncompressed_size.to_i }
|
|
651
|
+
nulls = chunks.map(&:null_count)
|
|
652
|
+
stats = chunks.map(&:statistics)
|
|
653
|
+
mins = stats.map { |s| s&.min }
|
|
654
|
+
maxes = stats.map { |s| s&.max }
|
|
655
|
+
{
|
|
656
|
+
column: col.index,
|
|
657
|
+
path: col.dotted_path,
|
|
658
|
+
type: Inspector.type_name(col),
|
|
659
|
+
codecs: chunks.map(&:codec).uniq,
|
|
660
|
+
encodings: chunks.flat_map(&:encodings).uniq,
|
|
661
|
+
num_values: chunks.sum { |c| c.num_values.to_i },
|
|
662
|
+
null_count: nulls.all? ? nulls.sum : nil,
|
|
663
|
+
compressed_size: compressed,
|
|
664
|
+
uncompressed_size: uncompressed,
|
|
665
|
+
compression_ratio: compressed.positive? ? (uncompressed.to_f / compressed).round(4) : nil,
|
|
666
|
+
num_pages: chunks.sum { |c| c.pages.size },
|
|
667
|
+
num_data_pages: chunks.sum { |c| c.data_pages.size },
|
|
668
|
+
dictionary_pages: chunks.count(&:dictionary_page),
|
|
669
|
+
dictionary_bytes: chunks.sum { |c| c.dictionary_page&.total_size.to_i },
|
|
670
|
+
min: mins.include?(nil) ? nil : safe_extreme(mins, :min),
|
|
671
|
+
max: maxes.include?(nil) ? nil : safe_extreme(maxes, :max)
|
|
672
|
+
}.compact
|
|
673
|
+
end
|
|
674
|
+
end
|
|
675
|
+
|
|
676
|
+
# Byte ranges of the whole file, in offset order: magic, pages (or whole chunks when their pages
|
|
677
|
+
# could not be walked), bloom filters, page indexes, footer. Gaps right after a chunk at its
|
|
678
|
+
# file_offset are inline :column_metadata copies; other gaps are reported as :unknown. Each entry: { kind:, start:, length:, row_group:, column:, page: }
|
|
679
|
+
# +row_group+, +column+ and +page+ are indexes, present only where they apply. Segments can
|
|
680
|
+
# overlap when the file is damaged; gaps are only computed past the furthest end seen so far.
|
|
681
|
+
# @return [Array<Hash{Symbol => Object}>]
|
|
682
|
+
def layout
|
|
683
|
+
segs = [{kind: :magic, start: 0, length: 4}]
|
|
684
|
+
column_chunks.each do |c|
|
|
685
|
+
rg = c.row_group.index
|
|
686
|
+
col = c.column.index
|
|
687
|
+
if c.pages.empty?
|
|
688
|
+
segs << {kind: :chunk, start: c.start_offset, length: c.compressed_size, row_group: rg, column: col}
|
|
689
|
+
else
|
|
690
|
+
c.pages.each_with_index do |p, i|
|
|
691
|
+
segs << {kind: p.dictionary? ? :dictionary_page : :data_page, start: p.offset, length: p.total_size,
|
|
692
|
+
row_group: rg, column: col, page: i}
|
|
693
|
+
end
|
|
694
|
+
end
|
|
695
|
+
if c.bloom_filter_offset
|
|
696
|
+
segs << {kind: :bloom_filter, start: c.bloom_filter_offset, length: c.bloom_filter_length.to_i,
|
|
697
|
+
row_group: rg, column: col}
|
|
698
|
+
end
|
|
699
|
+
if (r = c.column_index_range)
|
|
700
|
+
segs << {kind: :column_index, start: r[0], length: r[1], row_group: rg, column: col}
|
|
701
|
+
end
|
|
702
|
+
if (r = c.offset_index_range)
|
|
703
|
+
segs << {kind: :offset_index, start: r[0], length: r[1], row_group: rg, column: col}
|
|
704
|
+
end
|
|
705
|
+
end
|
|
706
|
+
segs << {kind: :footer, start: footer_offset, length: @footer_size}
|
|
707
|
+
segs << {kind: :footer_length, start: @file_size - 8, length: 4}
|
|
708
|
+
segs << {kind: :magic, start: @file_size - 4, length: 4}
|
|
709
|
+
segs.sort_by! { |s| s[:start] }
|
|
710
|
+
# Some writers (old parquet-rs, parquet-mr for a while) store a copy of the ColumnMetaData
|
|
711
|
+
# right after the chunk, at ColumnChunk.file_offset
|
|
712
|
+
meta_copies = column_chunks.each_with_object({}) do |c, h|
|
|
713
|
+
fo = c.chunk.file_offset
|
|
714
|
+
h[fo] = c if fo && fo >= c.end_offset
|
|
715
|
+
end
|
|
716
|
+
out = []
|
|
717
|
+
pos = 0
|
|
718
|
+
segs.each do |s|
|
|
719
|
+
if s[:start] > pos
|
|
720
|
+
c = meta_copies[pos]
|
|
721
|
+
out << if c
|
|
722
|
+
{kind: :column_metadata, start: pos, length: s[:start] - pos, row_group: c.row_group.index, column: c.column.index}
|
|
723
|
+
else
|
|
724
|
+
{kind: :unknown, start: pos, length: s[:start] - pos}
|
|
725
|
+
end
|
|
726
|
+
end
|
|
727
|
+
out << s
|
|
728
|
+
pos = [pos, s[:start] + s[:length]].max
|
|
729
|
+
end
|
|
730
|
+
out
|
|
731
|
+
end
|
|
732
|
+
|
|
733
|
+
# Everything: summary, key/value metadata, schema tree, row groups with their chunks and
|
|
734
|
+
# pages, column totals, and any checksum and page index mismatches. Walks every page header.
|
|
735
|
+
# @return [Hash{Symbol => Object}] JSON-safe; keys without a value are left out
|
|
736
|
+
def to_h
|
|
737
|
+
Inspector.jsonable({
|
|
738
|
+
summary: summary,
|
|
739
|
+
key_value_metadata: key_value_metadata,
|
|
740
|
+
schema: schema_tree,
|
|
741
|
+
row_groups: row_groups.map(&:to_h),
|
|
742
|
+
column_totals: column_totals,
|
|
743
|
+
checksum_mismatches: checksum_summary&.fetch(:mismatches),
|
|
744
|
+
index_mismatches: index_mismatches
|
|
745
|
+
}.compact)
|
|
746
|
+
end
|
|
747
|
+
|
|
748
|
+
# @param args [Array] passed on to Hash#to_json (e.g. a JSON::State)
|
|
749
|
+
# @return [String] #to_h as JSON
|
|
750
|
+
def to_json(*args) = to_h.to_json(*args)
|
|
751
|
+
|
|
752
|
+
# A self-contained HTML page showing the file's layout, see Visualizer
|
|
753
|
+
# @return [String] HTML document
|
|
754
|
+
def to_html = Visualizer.new(self).to_html
|
|
755
|
+
|
|
756
|
+
# Readable text summary. With pages: true, lists every page header too.
|
|
757
|
+
# @param pages [Boolean] whether to add a line per page header under each column chunk
|
|
758
|
+
# @return [String] multi-line text, as printed by +bin/herringbone inspect+
|
|
759
|
+
def report(pages: false)
|
|
760
|
+
s = summary
|
|
761
|
+
out = []
|
|
762
|
+
out << "file: #{@name || "(IO)"}"
|
|
763
|
+
out << "size: #{Inspector.human_bytes(s[:file_size])} (#{s[:file_size]} bytes), footer #{s[:footer_size]} bytes at #{s[:footer_offset]}"
|
|
764
|
+
out << "rows: #{s[:num_rows]}, row groups: #{s[:num_row_groups]}, columns: #{s[:num_columns]}, format version #{s[:format_version]}"
|
|
765
|
+
out << "created by: #{s[:created_by]}"
|
|
766
|
+
out << "codecs: #{s[:codecs].join(", ")}; data #{Inspector.human_bytes(s[:compressed_size])} compressed, " \
|
|
767
|
+
"#{Inspector.human_bytes(s[:uncompressed_size])} uncompressed#{ratio_text(s[:uncompressed_size], s[:compressed_size])}"
|
|
768
|
+
out << "page index: #{s[:page_index] ? "yes" : "no"}, bloom filters: #{s[:bloom_filters] ? "yes" : "no"}"
|
|
769
|
+
if (cs = checksum_summary)
|
|
770
|
+
out << "page CRCs: #{cs[:ok]} ok, #{cs[:mismatch]} mismatched, #{cs[:absent]} without a CRC"
|
|
771
|
+
cs[:mismatches].each do |m|
|
|
772
|
+
out << " CRC MISMATCH: row group #{m[:row_group]} #{m[:column]} page #{m[:page]} (#{m[:type]} @#{m[:offset]}): " \
|
|
773
|
+
"header says #{format("%08x", m[:crc])}, data has #{format("%08x", m[:actual])}"
|
|
774
|
+
end
|
|
775
|
+
end
|
|
776
|
+
mismatches = index_mismatches
|
|
777
|
+
unless mismatches.empty?
|
|
778
|
+
out << "page statistics vs column index: #{mismatches.size} disagreement#{"s" unless mismatches.size == 1}"
|
|
779
|
+
mismatches.first(50).each { |m| out << " #{index_mismatch_text(m)}" }
|
|
780
|
+
out << " ..." if mismatches.size > 50
|
|
781
|
+
end
|
|
782
|
+
kvs = key_value_metadata
|
|
783
|
+
unless kvs.empty?
|
|
784
|
+
out << "key/value metadata:"
|
|
785
|
+
kvs.each do |kv|
|
|
786
|
+
out << " #{kv[:key]} (#{kv[:format]}, #{kv[:bytesize]} bytes): #{kv[:summary] || kv[:value].to_s[0, 80].inspect}"
|
|
787
|
+
out << " #{kv[:arrow_error]}" if kv[:arrow_error]
|
|
788
|
+
ArrowSchema.lines(kv[:arrow_schema][:fields]).each { |l| out << " #{l}" } if kv[:arrow_schema]
|
|
789
|
+
end
|
|
790
|
+
end
|
|
791
|
+
out << "schema:"
|
|
792
|
+
walk = lambda do |n, depth|
|
|
793
|
+
type = n[:children] ? "group" : [n[:physical_type], n[:type_length] && "(#{n[:type_length]})"].compact.join
|
|
794
|
+
ann = n[:logical_type] || n[:converted_type]
|
|
795
|
+
levels = n[:children] ? "" : " [def #{n[:max_definition_level]}, rep #{n[:max_repetition_level]}]"
|
|
796
|
+
arrow = n[:arrow_type] ? " arrow: #{n[:arrow_type]}" : ""
|
|
797
|
+
out << "#{" " * depth}#{n[:repetition]} #{type} #{n[:name]}#{" (#{ann})" if ann}#{levels}#{arrow}"
|
|
798
|
+
(n[:children] || []).each { |c| walk.call(c, depth + 1) }
|
|
799
|
+
end
|
|
800
|
+
schema_tree.each { |n| walk.call(n, 1) }
|
|
801
|
+
out << "columns:"
|
|
802
|
+
column_totals.each do |t|
|
|
803
|
+
range = t.key?(:min) ? " [#{Inspector.display(t[:min])} .. #{Inspector.display(t[:max])}]" : ""
|
|
804
|
+
out << " #{t[:path]}: #{t[:type]} #{t[:codecs].join(",")} #{t[:encodings].join(",")} " \
|
|
805
|
+
"#{Inspector.human_bytes(t[:compressed_size])}/#{Inspector.human_bytes(t[:uncompressed_size])}" \
|
|
806
|
+
"#{ratio_text(t[:uncompressed_size], t[:compressed_size])}, #{t[:num_values]} values" \
|
|
807
|
+
"#{", #{t[:null_count]} nulls" if t[:null_count]}, #{t[:num_data_pages]} data pages#{range}"
|
|
808
|
+
end
|
|
809
|
+
row_groups.each do |rg|
|
|
810
|
+
sorting = rg.sorting_columns.map { |c| "#{c[:column]}#{" desc" if c[:descending]}" }
|
|
811
|
+
out << "row group #{rg.index}: #{rg.num_rows} rows, #{Inspector.human_bytes(rg.compressed_size)} " \
|
|
812
|
+
"at #{rg.start_offset}..#{rg.end_offset}#{", sorted by #{sorting.join(", ")}" unless sorting.empty?}"
|
|
813
|
+
rg.columns.each do |c|
|
|
814
|
+
st = c.statistics
|
|
815
|
+
range = (st && !(st.min.nil? && st.max.nil?)) ? " [#{Inspector.display(st.min)} .. #{Inspector.display(st.max)}]" : ""
|
|
816
|
+
extras = []
|
|
817
|
+
extras << "dict #{c.dictionary_size} entries" if c.dictionary_page
|
|
818
|
+
extras << "column index" if c.column_index_range
|
|
819
|
+
extras << "offset index" if c.offset_index_range
|
|
820
|
+
extras << "bloom filter" if c.bloom_filter_offset
|
|
821
|
+
extras << "ERROR: #{c.error}" if c.error
|
|
822
|
+
out << " #{c.path}: #{c.codec} #{c.encodings.join(",")} " \
|
|
823
|
+
"#{c.compressed_size}/#{c.uncompressed_size} bytes#{ratio_text(c.uncompressed_size, c.compressed_size)}, " \
|
|
824
|
+
"#{c.num_values} values, #{c.pages.size} pages#{", #{extras.join(", ")}" unless extras.empty?}#{range}"
|
|
825
|
+
next unless pages
|
|
826
|
+
c.pages.each do |p|
|
|
827
|
+
st = p.statistics
|
|
828
|
+
out << " #{p.index}: #{p.type} @#{p.offset} header #{p.header_size} + #{p.compressed_size}/#{p.uncompressed_size} bytes, " \
|
|
829
|
+
"#{p.num_values} values#{", #{p.num_nulls} nulls" if p.num_nulls}#{", #{p.num_rows} rows" if p.num_rows}" \
|
|
830
|
+
"#{" #{p.encoding}" if p.encoding}#{crc_text(p)}" \
|
|
831
|
+
"#{" [#{Inspector.display(st.min)} .. #{Inspector.display(st.max)}]" if st && !(st.min.nil? && st.max.nil?)}"
|
|
832
|
+
end
|
|
833
|
+
end
|
|
834
|
+
end
|
|
835
|
+
out.join("\n")
|
|
836
|
+
end
|
|
837
|
+
|
|
838
|
+
# @return [String] short description: name, rows, row group count and file size
|
|
839
|
+
def inspect
|
|
840
|
+
"#<#{self.class.name} #{@name || "(IO)"} rows=#{num_rows} row_groups=#{row_groups.size} size=#{@file_size}>"
|
|
841
|
+
end
|
|
842
|
+
|
|
843
|
+
# ---- used by the info objects ----
|
|
844
|
+
|
|
845
|
+
# Walks page headers from the chunk's first page. Returns [pages, error_message_or_nil].
|
|
846
|
+
# Mirrors the reader's tolerance: a chunk may extend past its declared total_compressed_size.
|
|
847
|
+
# @param chunk [ColumnChunkInfo] chunk whose pages to walk
|
|
848
|
+
# @return [Array(Array<PageInfo>, String), Array(Array<PageInfo>, nil)] the pages found, and why
|
|
849
|
+
# the walk stopped early (nil when every value was accounted for)
|
|
850
|
+
def walk_pages(chunk)
|
|
851
|
+
return [[], "column chunk stored in external file #{chunk.external_file}"] if chunk.external_file
|
|
852
|
+
pages = []
|
|
853
|
+
pos = chunk.start_offset
|
|
854
|
+
limit = footer_offset
|
|
855
|
+
total = chunk.num_values.to_i
|
|
856
|
+
seen = 0
|
|
857
|
+
declared_end = chunk.declared_end_offset
|
|
858
|
+
# Walk until all values are accounted for (reading past the declared end if a writer
|
|
859
|
+
# under-reported it), then pick up any trailing pages still inside the declared range
|
|
860
|
+
while pos < limit && (seen < total || pos < declared_end)
|
|
861
|
+
trailing = seen >= total
|
|
862
|
+
begin
|
|
863
|
+
header, header_size = read_page_header(pos, limit)
|
|
864
|
+
page = page_info(pages.size, header, pos, header_size, chunk.column)
|
|
865
|
+
rescue Thrift::Error, FormatError => e
|
|
866
|
+
break if trailing
|
|
867
|
+
return [pages, "corrupt page header at #{pos}: #{e.message}"]
|
|
868
|
+
end
|
|
869
|
+
if page.end_offset > limit
|
|
870
|
+
break if trailing
|
|
871
|
+
return [pages, "page #{pages.size} at #{pos} overruns the data section (#{page.end_offset} > #{limit})"]
|
|
872
|
+
end
|
|
873
|
+
pages << page
|
|
874
|
+
seen += page.num_values.to_i if page.data?
|
|
875
|
+
pos = page.end_offset
|
|
876
|
+
end
|
|
877
|
+
[pages, (seen < total) ? "found #{seen} of #{total} values in page headers" : nil]
|
|
878
|
+
end
|
|
879
|
+
|
|
880
|
+
# Reads and decodes a chunk's ColumnIndex, decoding the per-page min/max with the column type
|
|
881
|
+
# @param chunk [ColumnChunkInfo] chunk whose ColumnIndex to read
|
|
882
|
+
# @return [ColumnIndexInfo, nil] nil when the chunk has none or it fails to decode
|
|
883
|
+
def read_column_index(chunk)
|
|
884
|
+
offset, length = chunk.column_index_range
|
|
885
|
+
return nil unless offset && length.positive?
|
|
886
|
+
ci = Format::ColumnIndex.decode(read_at(offset, length)).first
|
|
887
|
+
col = chunk.column
|
|
888
|
+
mins = ci.min_values || []
|
|
889
|
+
maxes = ci.max_values || []
|
|
890
|
+
nulls = ci.null_pages || []
|
|
891
|
+
ColumnIndexInfo.new(
|
|
892
|
+
offset: offset, length: length,
|
|
893
|
+
null_pages: nulls,
|
|
894
|
+
min_values: mins.each_with_index.map { |v, i| nulls[i] ? nil : decode_value(v, col) },
|
|
895
|
+
max_values: maxes.each_with_index.map { |v, i| nulls[i] ? nil : decode_value(v, col) },
|
|
896
|
+
boundary_order: Format::BoundaryOrder::NAMES[ci.boundary_order]&.to_s || ci.boundary_order,
|
|
897
|
+
null_counts: ci.null_counts,
|
|
898
|
+
repetition_level_histograms: ci.repetition_level_histograms,
|
|
899
|
+
definition_level_histograms: ci.definition_level_histograms
|
|
900
|
+
)
|
|
901
|
+
rescue Thrift::Error
|
|
902
|
+
nil
|
|
903
|
+
end
|
|
904
|
+
|
|
905
|
+
# Reads and decodes a chunk's OffsetIndex
|
|
906
|
+
# @param chunk [ColumnChunkInfo] chunk whose OffsetIndex to read
|
|
907
|
+
# @return [OffsetIndexInfo, nil] nil when the chunk has none or it fails to decode
|
|
908
|
+
def read_offset_index(chunk)
|
|
909
|
+
offset, length = chunk.offset_index_range
|
|
910
|
+
return nil unless offset && length.positive?
|
|
911
|
+
oi = Format::OffsetIndex.decode(read_at(offset, length)).first
|
|
912
|
+
OffsetIndexInfo.new(
|
|
913
|
+
offset: offset, length: length,
|
|
914
|
+
page_locations: (oi.page_locations || []).map do |l|
|
|
915
|
+
PageLocation.new(offset: l.offset, compressed_page_size: l.compressed_page_size, first_row_index: l.first_row_index)
|
|
916
|
+
end,
|
|
917
|
+
unencoded_byte_array_data_bytes: oi.unencoded_byte_array_data_bytes
|
|
918
|
+
)
|
|
919
|
+
rescue Thrift::Error
|
|
920
|
+
nil
|
|
921
|
+
end
|
|
922
|
+
|
|
923
|
+
# Size in bytes of the bloom filter at +offset+ (its Thrift header plus the bitset)
|
|
924
|
+
# @param offset [Integer] file offset of the BloomFilterHeader
|
|
925
|
+
# @return [Integer, nil] nil when the header can't be decoded or has no num_bytes
|
|
926
|
+
def bloom_filter_size(offset)
|
|
927
|
+
buf = read_at(offset, 64)
|
|
928
|
+
header, size = BloomFilterHeader.decode(buf)
|
|
929
|
+
header.num_bytes ? size + header.num_bytes : nil
|
|
930
|
+
rescue Thrift::Error
|
|
931
|
+
nil
|
|
932
|
+
end
|
|
933
|
+
|
|
934
|
+
# :ok, :mismatch or :absent for one page (reads its body). Sets PageInfo#actual_crc.
|
|
935
|
+
# @param page [PageInfo] page to check
|
|
936
|
+
# @return [Symbol]
|
|
937
|
+
def page_checksum(page)
|
|
938
|
+
return :absent unless page.crc
|
|
939
|
+
page.actual_crc = page_crc(page)
|
|
940
|
+
(page.actual_crc == page.expected_crc) ? :ok : :mismatch
|
|
941
|
+
end
|
|
942
|
+
|
|
943
|
+
# CRC32 of a page's body as stored
|
|
944
|
+
# @param page [PageInfo] page whose compressed body to read
|
|
945
|
+
# @return [Integer] unsigned 32-bit CRC
|
|
946
|
+
def page_crc(page)
|
|
947
|
+
Zlib.crc32(read_at(page.body_offset, page.compressed_size))
|
|
948
|
+
end
|
|
949
|
+
|
|
950
|
+
# See ColumnChunkInfo#index_mismatches
|
|
951
|
+
# @param chunk [ColumnChunkInfo] chunk whose page headers to compare with its ColumnIndex
|
|
952
|
+
# @return [Array<Hash{Symbol => Object}>] one entry per disagreement; a single :page_count
|
|
953
|
+
# entry when the ColumnIndex and the data pages differ in number
|
|
954
|
+
def compare_page_index(chunk)
|
|
955
|
+
ci = chunk.column_index or return []
|
|
956
|
+
data = chunk.data_pages
|
|
957
|
+
if ci.null_pages.size != data.size
|
|
958
|
+
return [{page: nil, data_page: nil, field: :page_count, page_value: data.size, index_value: ci.null_pages.size}]
|
|
959
|
+
end
|
|
960
|
+
order = Inspector.sort_order(chunk.column)
|
|
961
|
+
out = []
|
|
962
|
+
data.each_with_index do |p, k|
|
|
963
|
+
report = ->(field, pv, iv) { out << {page: p.index, data_page: k, field: field, page_value: pv, index_value: iv} }
|
|
964
|
+
idx_nulls = ci.null_counts&.[](k)
|
|
965
|
+
report.call(:null_count, p.num_nulls, idx_nulls) if idx_nulls && p.num_nulls && idx_nulls != p.num_nulls
|
|
966
|
+
st = p.statistics
|
|
967
|
+
# Legacy min/max were computed with another ordering, so they can't be compared
|
|
968
|
+
next if st.nil? || st.caveat || (st.min.nil? && st.max.nil?)
|
|
969
|
+
if ci.null_pages[k]
|
|
970
|
+
report.call(:null_page, "has min/max", "null page")
|
|
971
|
+
next
|
|
972
|
+
end
|
|
973
|
+
imin = ci.min_values[k]
|
|
974
|
+
imax = ci.max_values[k]
|
|
975
|
+
report.call(:min, st.min, imin) if st.min_exact != false && narrower?(imin, st.min, order, :min)
|
|
976
|
+
report.call(:max, st.max, imax) if st.max_exact != false && narrower?(imax, st.max, order, :max)
|
|
977
|
+
end
|
|
978
|
+
out
|
|
979
|
+
end
|
|
980
|
+
|
|
981
|
+
# Decodes a Format::Statistics into Ruby values via the column's type converter
|
|
982
|
+
# Prefers min_value/max_value; falls back to the deprecated min/max, with a caveat when their
|
|
983
|
+
# ordering can't be trusted for the column's type.
|
|
984
|
+
# @param st [Format::Statistics, nil] statistics from column metadata or a page header
|
|
985
|
+
# @param column [Schema::Column] column the statistics describe
|
|
986
|
+
# @return [Stats, nil] nil when +st+ is nil
|
|
987
|
+
def decode_statistics(st, column)
|
|
988
|
+
return nil unless st
|
|
989
|
+
order = Inspector.sort_order(column)
|
|
990
|
+
if st.min_value || st.max_value
|
|
991
|
+
min_raw = st.min_value
|
|
992
|
+
max_raw = st.max_value
|
|
993
|
+
source = "min_value/max_value"
|
|
994
|
+
elsif st.min || st.max
|
|
995
|
+
min_raw = st.min
|
|
996
|
+
max_raw = st.max
|
|
997
|
+
source = "min/max (legacy)"
|
|
998
|
+
binary = column.type == T::BYTE_ARRAY || column.type == T::FIXED_LEN_BYTE_ARRAY
|
|
999
|
+
if order == :unsigned
|
|
1000
|
+
caveat = "legacy min/max were computed with signed comparison, which is wrong for unsigned-ordered " \
|
|
1001
|
+
"values (strings, binary, unsigned integers); most readers ignore them"
|
|
1002
|
+
elsif binary
|
|
1003
|
+
caveat = "legacy min/max of binary-backed values were compared bytewise; most readers ignore them"
|
|
1004
|
+
end
|
|
1005
|
+
end
|
|
1006
|
+
caveat ||= "sort order of #{T::NAMES[column.type]} is undefined; min/max may be meaningless" if order == :unknown && source
|
|
1007
|
+
Stats.new(
|
|
1008
|
+
min: decode_value(min_raw, column),
|
|
1009
|
+
max: decode_value(max_raw, column),
|
|
1010
|
+
null_count: st.null_count,
|
|
1011
|
+
distinct_count: st.distinct_count,
|
|
1012
|
+
min_exact: st.is_min_value_exact,
|
|
1013
|
+
max_exact: st.is_max_value_exact,
|
|
1014
|
+
source: source,
|
|
1015
|
+
caveat: caveat
|
|
1016
|
+
)
|
|
1017
|
+
end
|
|
1018
|
+
|
|
1019
|
+
# Decodes one PLAIN-encoded statistics value (no length prefix for byte arrays)
|
|
1020
|
+
# INT96 decodes to [nanoseconds, julian_day] before conversion. Values of the wrong width,
|
|
1021
|
+
# or that fail to convert, come back as hex (see Inspector.hex).
|
|
1022
|
+
# @param bytes [String, nil] encoded value
|
|
1023
|
+
# @param column [Schema::Column] column whose physical type and converter apply
|
|
1024
|
+
# @return [Object, nil] the converted value, a hex String when it could not be decoded, nil
|
|
1025
|
+
# for nil (or empty BOOLEAN) input
|
|
1026
|
+
def decode_value(bytes, column)
|
|
1027
|
+
return nil if bytes.nil?
|
|
1028
|
+
bytes = bytes.b
|
|
1029
|
+
raw = case column.type
|
|
1030
|
+
when T::BOOLEAN
|
|
1031
|
+
return nil if bytes.empty?
|
|
1032
|
+
bytes.getbyte(0) != 0
|
|
1033
|
+
when T::BYTE_ARRAY, T::FIXED_LEN_BYTE_ARRAY
|
|
1034
|
+
bytes
|
|
1035
|
+
when T::INT96
|
|
1036
|
+
return Inspector.hex(bytes) unless bytes.bytesize == 12
|
|
1037
|
+
bytes.unpack("Q<L<")
|
|
1038
|
+
else
|
|
1039
|
+
width = Encodings::Plain::FORMATS[column.type][1]
|
|
1040
|
+
return Inspector.hex(bytes) unless bytes.bytesize == width
|
|
1041
|
+
Encodings::Plain.decode(bytes, 0, 1, column.type).first.first
|
|
1042
|
+
end
|
|
1043
|
+
conv = column.converter
|
|
1044
|
+
conv ? conv.call(raw) : raw
|
|
1045
|
+
rescue
|
|
1046
|
+
Inspector.hex(bytes)
|
|
1047
|
+
end
|
|
1048
|
+
|
|
1049
|
+
# Decodes the ARROW:schema key/value that Arrow writers (pyarrow, arrow-rs, DuckDB...) store:
|
|
1050
|
+
# base64 of an Arrow IPC message whose header is a flatbuffer Schema (Arrow's Message.fbs and
|
|
1051
|
+
# Schema.fbs). Pure Ruby and read-only; type names follow pyarrow's (str(field.type)).
|
|
1052
|
+
#
|
|
1053
|
+
# Inspector::ArrowSchema.decode(value)
|
|
1054
|
+
# # => { endianness: "little", metadata: {...}, fields: [{ name: "a", type: "int32", nullable: true, ... }] }
|
|
1055
|
+
#
|
|
1056
|
+
# Fields carry :name, :type, :nullable and, when present, :children, :dictionary
|
|
1057
|
+
# ({ index_type:, ordered:, id: }), :extension (ARROW:extension:name) and :metadata.
|
|
1058
|
+
module ArrowSchema
|
|
1059
|
+
# Raised for malformed or unsupported ARROW:schema values
|
|
1060
|
+
class Error < StandardError; end
|
|
1061
|
+
|
|
1062
|
+
# Deepest field nesting accepted, against stack exhaustion on hostile input
|
|
1063
|
+
MAX_DEPTH = 64
|
|
1064
|
+
# Most fields (nested ones included) accepted in one schema
|
|
1065
|
+
MAX_FIELDS = 100_000
|
|
1066
|
+
# Arrow TimeUnit values (SECOND, MILLISECOND, MICROSECOND, NANOSECOND) as pyarrow abbreviates them
|
|
1067
|
+
TIME_UNITS = %w[s ms us ns].freeze
|
|
1068
|
+
# MessageHeader union tag of a Schema in Message.fbs
|
|
1069
|
+
MESSAGE_SCHEMA = 1
|
|
1070
|
+
|
|
1071
|
+
# A minimal flatbuffer reader: tables (through their vtables), scalars, strings, vectors of
|
|
1072
|
+
# scalars and tables, and unions. Every read is bounds-checked; malformed input raises Error.
|
|
1073
|
+
class FlatBuffer
|
|
1074
|
+
# @param bytes [String] the flatbuffer (copied as binary)
|
|
1075
|
+
def initialize(bytes)
|
|
1076
|
+
@b = bytes.b
|
|
1077
|
+
end
|
|
1078
|
+
|
|
1079
|
+
# @return [Table] the root table, whose offset is stored in the first 4 bytes
|
|
1080
|
+
def root = table_at(u32(0))
|
|
1081
|
+
|
|
1082
|
+
# @param pos [Integer] absolute position of a table
|
|
1083
|
+
# @return [Table]
|
|
1084
|
+
# @raise [Error] when its vtable is out of bounds or malformed
|
|
1085
|
+
def table_at(pos) = Table.new(self, pos)
|
|
1086
|
+
|
|
1087
|
+
# @param pos [Integer] absolute start of the read
|
|
1088
|
+
# @param len [Integer] bytes to read
|
|
1089
|
+
# @return [void]
|
|
1090
|
+
# @raise [Error] when the range is not inside the buffer
|
|
1091
|
+
def check(pos, len)
|
|
1092
|
+
return if pos >= 0 && len >= 0 && pos + len <= @b.bytesize
|
|
1093
|
+
raise Error, "flatbuffer read of #{len} bytes at #{pos} is out of bounds (#{@b.bytesize} bytes)"
|
|
1094
|
+
end
|
|
1095
|
+
|
|
1096
|
+
# @param pos [Integer] absolute start of the read
|
|
1097
|
+
# @param len [Integer] bytes to read
|
|
1098
|
+
# @param fmt [String] String#unpack1 directive for the bytes
|
|
1099
|
+
# @return [Integer] the unpacked scalar
|
|
1100
|
+
# @raise [Error] when the range is not inside the buffer
|
|
1101
|
+
def read(pos, len, fmt)
|
|
1102
|
+
check(pos, len)
|
|
1103
|
+
@b.byteslice(pos, len).unpack1(fmt)
|
|
1104
|
+
end
|
|
1105
|
+
|
|
1106
|
+
# @param pos [Integer] absolute position
|
|
1107
|
+
# @return [Integer] unsigned 8-bit value at +pos+
|
|
1108
|
+
def u8(pos) = read(pos, 1, "C")
|
|
1109
|
+
|
|
1110
|
+
# @param pos [Integer] absolute position
|
|
1111
|
+
# @return [Integer] unsigned little-endian 16-bit value at +pos+
|
|
1112
|
+
def u16(pos) = read(pos, 2, "S<")
|
|
1113
|
+
|
|
1114
|
+
# @param pos [Integer] absolute position
|
|
1115
|
+
# @return [Integer] signed little-endian 16-bit value at +pos+
|
|
1116
|
+
def i16(pos) = read(pos, 2, "s<")
|
|
1117
|
+
|
|
1118
|
+
# @param pos [Integer] absolute position
|
|
1119
|
+
# @return [Integer] unsigned little-endian 32-bit value at +pos+
|
|
1120
|
+
def u32(pos) = read(pos, 4, "L<")
|
|
1121
|
+
|
|
1122
|
+
# @param pos [Integer] absolute position
|
|
1123
|
+
# @return [Integer] signed little-endian 32-bit value at +pos+
|
|
1124
|
+
def i32(pos) = read(pos, 4, "l<")
|
|
1125
|
+
|
|
1126
|
+
# @param pos [Integer] absolute position
|
|
1127
|
+
# @return [Integer] signed little-endian 64-bit value at +pos+
|
|
1128
|
+
def i64(pos) = read(pos, 8, "q<")
|
|
1129
|
+
|
|
1130
|
+
# Offsets are relative to where they are stored
|
|
1131
|
+
# @param pos [Integer] absolute position of a uoffset
|
|
1132
|
+
# @return [Integer] absolute position it points to
|
|
1133
|
+
def deref(pos) = pos + u32(pos)
|
|
1134
|
+
|
|
1135
|
+
# @param pos [Integer] absolute position of the string's length prefix
|
|
1136
|
+
# @return [String] the string's bytes as UTF-8 (not validated)
|
|
1137
|
+
def string(pos)
|
|
1138
|
+
len = u32(pos)
|
|
1139
|
+
check(pos + 4, len)
|
|
1140
|
+
@b.byteslice(pos + 4, len).force_encoding(Encoding::UTF_8)
|
|
1141
|
+
end
|
|
1142
|
+
|
|
1143
|
+
# [start, length] of the vector at +pos+ with +size+-byte elements
|
|
1144
|
+
# @param pos [Integer] absolute position of the vector's length prefix
|
|
1145
|
+
# @param size [Integer] bytes per element
|
|
1146
|
+
# @return [Array(Integer, Integer)] start of the first element and the element count
|
|
1147
|
+
# @raise [Error] when the elements do not fit in the buffer
|
|
1148
|
+
def vector(pos, size)
|
|
1149
|
+
len = u32(pos)
|
|
1150
|
+
check(pos + 4, len * size)
|
|
1151
|
+
[pos + 4, len]
|
|
1152
|
+
end
|
|
1153
|
+
end
|
|
1154
|
+
|
|
1155
|
+
# One flatbuffer table; fields are addressed by their slot (declaration order in the .fbs)
|
|
1156
|
+
class Table
|
|
1157
|
+
# @param fb [FlatBuffer] buffer the table lives in
|
|
1158
|
+
# @param pos [Integer] absolute position of the table (where its vtable offset is stored)
|
|
1159
|
+
# @raise [Error] when the vtable is out of bounds or malformed
|
|
1160
|
+
def initialize(fb, pos)
|
|
1161
|
+
@fb = fb
|
|
1162
|
+
@pos = pos
|
|
1163
|
+
@vtable = pos - fb.i32(pos)
|
|
1164
|
+
@vtable_size = fb.u16(@vtable)
|
|
1165
|
+
raise Error, "bad flatbuffer vtable at #{@vtable}" if @vtable_size < 4 || @vtable_size.odd?
|
|
1166
|
+
fb.check(@vtable, @vtable_size)
|
|
1167
|
+
end
|
|
1168
|
+
|
|
1169
|
+
# Absolute position of a field's value, nil when absent
|
|
1170
|
+
# @param slot [Integer] field slot, counting from 0
|
|
1171
|
+
# @return [Integer, nil]
|
|
1172
|
+
def field(slot)
|
|
1173
|
+
o = 4 + (slot * 2)
|
|
1174
|
+
return nil if o + 2 > @vtable_size
|
|
1175
|
+
off = @fb.u16(@vtable + o)
|
|
1176
|
+
off.zero? ? nil : @pos + off
|
|
1177
|
+
end
|
|
1178
|
+
|
|
1179
|
+
# @param slot [Integer] field slot, counting from 0
|
|
1180
|
+
# @param default [Integer] returned when the field is absent (the .fbs default)
|
|
1181
|
+
# @return [Integer] the unsigned 8-bit field (also used for union type tags)
|
|
1182
|
+
def u8(slot, default = 0) = (p = field(slot)) ? @fb.u8(p) : default
|
|
1183
|
+
|
|
1184
|
+
# @param slot [Integer] field slot, counting from 0
|
|
1185
|
+
# @param default [Boolean] returned when the field is absent (the .fbs default)
|
|
1186
|
+
# @return [Boolean]
|
|
1187
|
+
def bool(slot, default = false) = (p = field(slot)) ? @fb.u8(p) != 0 : default
|
|
1188
|
+
|
|
1189
|
+
# @param slot [Integer] field slot, counting from 0
|
|
1190
|
+
# @param default [Integer] returned when the field is absent (the .fbs default)
|
|
1191
|
+
# @return [Integer] the signed 16-bit field (also used for enums such as TimeUnit)
|
|
1192
|
+
def i16(slot, default = 0) = (p = field(slot)) ? @fb.i16(p) : default
|
|
1193
|
+
|
|
1194
|
+
# @param slot [Integer] field slot, counting from 0
|
|
1195
|
+
# @param default [Integer] returned when the field is absent (the .fbs default)
|
|
1196
|
+
# @return [Integer] the signed 32-bit field
|
|
1197
|
+
def i32(slot, default = 0) = (p = field(slot)) ? @fb.i32(p) : default
|
|
1198
|
+
|
|
1199
|
+
# @param slot [Integer] field slot, counting from 0
|
|
1200
|
+
# @param default [Integer] returned when the field is absent (the .fbs default)
|
|
1201
|
+
# @return [Integer] the signed 64-bit field
|
|
1202
|
+
def i64(slot, default = 0) = (p = field(slot)) ? @fb.i64(p) : default
|
|
1203
|
+
|
|
1204
|
+
# @param slot [Integer] field slot, counting from 0
|
|
1205
|
+
# @return [String, nil] the string field, nil when absent
|
|
1206
|
+
def string(slot) = (p = field(slot)) && @fb.string(@fb.deref(p))
|
|
1207
|
+
|
|
1208
|
+
# @param slot [Integer] field slot, counting from 0 (also the value of a union)
|
|
1209
|
+
# @return [Table, nil] the sub-table, nil when absent
|
|
1210
|
+
def table(slot) = (p = field(slot)) && @fb.table_at(@fb.deref(p))
|
|
1211
|
+
|
|
1212
|
+
# @param slot [Integer] field slot, counting from 0
|
|
1213
|
+
# @return [Array<Table>] the vector of tables, empty when absent
|
|
1214
|
+
def tables(slot)
|
|
1215
|
+
p = field(slot) or return []
|
|
1216
|
+
start, len = @fb.vector(@fb.deref(p), 4)
|
|
1217
|
+
Array.new(len) { |i| @fb.table_at(@fb.deref(start + (4 * i))) }
|
|
1218
|
+
end
|
|
1219
|
+
|
|
1220
|
+
# @param slot [Integer] field slot, counting from 0
|
|
1221
|
+
# @return [Array<Integer>] the vector of signed 32-bit values, empty when absent
|
|
1222
|
+
def i32s(slot)
|
|
1223
|
+
p = field(slot) or return []
|
|
1224
|
+
start, len = @fb.vector(@fb.deref(p), 4)
|
|
1225
|
+
Array.new(len) { |i| @fb.i32(start + (4 * i)) }
|
|
1226
|
+
end
|
|
1227
|
+
end
|
|
1228
|
+
|
|
1229
|
+
module_function
|
|
1230
|
+
|
|
1231
|
+
# Decodes the base64 ARROW:schema value; raises ArrowSchema::Error when it can't
|
|
1232
|
+
# @param b64 [String] the key/value metadata value (base64 of an IPC Schema message)
|
|
1233
|
+
# @return [Hash{Symbol => Object}] { endianness:, fields:, metadata: }; +metadata+ is a
|
|
1234
|
+
# Hash{String => String}, left out when the schema has none
|
|
1235
|
+
# @raise [Error] when the value is empty, malformed or not a Schema message
|
|
1236
|
+
def decode(b64)
|
|
1237
|
+
bytes = b64.to_s.unpack1("m")
|
|
1238
|
+
raise Error, "empty value" if bytes.empty?
|
|
1239
|
+
fb = FlatBuffer.new(message_bytes(bytes))
|
|
1240
|
+
message = fb.root
|
|
1241
|
+
header_type = message.u8(1)
|
|
1242
|
+
raise Error, "IPC message holds a #{header_type} header, not a Schema" unless header_type == MESSAGE_SCHEMA
|
|
1243
|
+
schema = message.table(2) or raise Error, "IPC message has no Schema"
|
|
1244
|
+
count = [0]
|
|
1245
|
+
{
|
|
1246
|
+
endianness: schema.i16(0).zero? ? "little" : "big",
|
|
1247
|
+
fields: schema.tables(1).map { |f| field(f, 0, count) },
|
|
1248
|
+
metadata: key_values(schema.tables(2))
|
|
1249
|
+
}.compact
|
|
1250
|
+
rescue ArgumentError, TypeError, RangeError => e
|
|
1251
|
+
raise Error, e.message
|
|
1252
|
+
end
|
|
1253
|
+
|
|
1254
|
+
# The flatbuffer inside an encapsulated IPC message: [0xFFFFFFFF] int32 length, flatbuffer
|
|
1255
|
+
# (the continuation marker is missing in files from before Arrow 0.15)
|
|
1256
|
+
# @param bytes [String] the decoded (binary) ARROW:schema value
|
|
1257
|
+
# @return [String] the message's flatbuffer bytes
|
|
1258
|
+
# @raise [Error] when the length prefix is missing or does not fit
|
|
1259
|
+
def message_bytes(bytes)
|
|
1260
|
+
raise Error, "too short for an IPC message (#{bytes.bytesize} bytes)" if bytes.bytesize < 8
|
|
1261
|
+
len = bytes.unpack1("l<")
|
|
1262
|
+
start = 4
|
|
1263
|
+
if len == -1
|
|
1264
|
+
len = bytes.byteslice(4, 4).unpack1("l<")
|
|
1265
|
+
start = 8
|
|
1266
|
+
end
|
|
1267
|
+
raise Error, "IPC message length #{len} does not fit in #{bytes.bytesize} bytes" if len <= 0 || start + len > bytes.bytesize
|
|
1268
|
+
bytes.byteslice(start, len)
|
|
1269
|
+
end
|
|
1270
|
+
|
|
1271
|
+
# Describes one Arrow Field table and, recursively, its children
|
|
1272
|
+
# @param t [Table] the Field table
|
|
1273
|
+
# @param depth [Integer] nesting depth of the field, 0 at the top level
|
|
1274
|
+
# @param count [Array<Integer>] one-element counter of the fields seen so far, shared across
|
|
1275
|
+
# the recursion
|
|
1276
|
+
# @return [Hash{Symbol => Object}] see the module description for the keys
|
|
1277
|
+
# @raise [Error] past MAX_DEPTH or MAX_FIELDS, or on malformed input
|
|
1278
|
+
def field(t, depth, count)
|
|
1279
|
+
raise Error, "fields nested deeper than #{MAX_DEPTH} levels" if depth > MAX_DEPTH
|
|
1280
|
+
raise Error, "more than #{MAX_FIELDS} fields" if (count[0] += 1) > MAX_FIELDS
|
|
1281
|
+
children = t.tables(5).map { |c| field(c, depth + 1, count) }
|
|
1282
|
+
metadata = key_values(t.tables(6))
|
|
1283
|
+
type = type_name(t.u8(2), t.table(3), children)
|
|
1284
|
+
h = {name: t.string(0).to_s, type: type, nullable: t.bool(1)}
|
|
1285
|
+
if (d = t.table(4))
|
|
1286
|
+
index = d.table(1)
|
|
1287
|
+
index_type = index ? int_name(index) : "int32"
|
|
1288
|
+
ordered = d.bool(2)
|
|
1289
|
+
h[:dictionary] = {index_type: index_type, ordered: ordered, id: d.i64(0)}
|
|
1290
|
+
h[:type] = "dictionary<values=#{type}, indices=#{index_type}, ordered=#{ordered ? 1 : 0}>"
|
|
1291
|
+
end
|
|
1292
|
+
h[:children] = children unless children.empty?
|
|
1293
|
+
if metadata
|
|
1294
|
+
h[:extension] = metadata["ARROW:extension:name"] if metadata["ARROW:extension:name"]
|
|
1295
|
+
h[:metadata] = metadata
|
|
1296
|
+
end
|
|
1297
|
+
h
|
|
1298
|
+
end
|
|
1299
|
+
|
|
1300
|
+
# @param tables [Array<Table>] KeyValue tables (custom_metadata of a Schema or Field)
|
|
1301
|
+
# @return [Hash{String => String}, nil] nil when there are none
|
|
1302
|
+
def key_values(tables)
|
|
1303
|
+
return nil if tables.empty?
|
|
1304
|
+
tables.to_h { |kv| [kv.string(0).to_s, kv.string(1).to_s] }
|
|
1305
|
+
end
|
|
1306
|
+
|
|
1307
|
+
# A child as pyarrow prints it inside a nested type: "name: type" plus " not null"
|
|
1308
|
+
# @param c [Hash{Symbol => Object}] a field as returned by #field
|
|
1309
|
+
# @return [String]
|
|
1310
|
+
def child_text(c) = "#{c[:name]}: #{c[:type]}#{" not null" unless c[:nullable]}"
|
|
1311
|
+
|
|
1312
|
+
# @param t [Table] an Arrow Int table (bitWidth, is_signed)
|
|
1313
|
+
# @return [String] e.g. "int32" or "uint8"
|
|
1314
|
+
def int_name(t) = "#{"u" unless t.bool(1)}int#{t.i32(0)}"
|
|
1315
|
+
|
|
1316
|
+
# @param u [Integer] Arrow TimeUnit value
|
|
1317
|
+
# @return [String] "s", "ms", "us" or "ns" ("unitN" for unknown values)
|
|
1318
|
+
def unit(u) = TIME_UNITS[u] || "unit#{u}"
|
|
1319
|
+
|
|
1320
|
+
# Type names follow Arrow's DataType::ToString (what pyarrow prints)
|
|
1321
|
+
# @param kind [Integer] the Field's Type union tag (Schema.fbs)
|
|
1322
|
+
# @param t [Table, nil] the union's type table, nil when absent
|
|
1323
|
+
# @param children [Array<Hash{Symbol => Object}>] the already described child fields
|
|
1324
|
+
# @return [String] e.g. "timestamp[us, tz=UTC]" or "list<item: string>"
|
|
1325
|
+
# @raise [Error] for a Decimal type without its parameters table
|
|
1326
|
+
def type_name(kind, t, children)
|
|
1327
|
+
case kind
|
|
1328
|
+
when 1 then "null"
|
|
1329
|
+
when 2 then t ? int_name(t) : "int"
|
|
1330
|
+
when 3 then %w[halffloat float double][t ? t.i16(0) : 0] || "float?"
|
|
1331
|
+
when 4 then "binary"
|
|
1332
|
+
when 5 then "string"
|
|
1333
|
+
when 6 then "bool"
|
|
1334
|
+
when 7
|
|
1335
|
+
raise Error, "decimal type without parameters" unless t
|
|
1336
|
+
bits = t.i32(2, 128)
|
|
1337
|
+
"decimal#{bits}(#{t.i32(0)}, #{t.i32(1)})"
|
|
1338
|
+
when 8 then t&.i16(0, 1)&.zero? ? "date32[day]" : "date64[ms]"
|
|
1339
|
+
when 9
|
|
1340
|
+
bits = t ? t.i32(1, 32) : 32
|
|
1341
|
+
"time#{bits}[#{unit(t ? t.i16(0, 1) : 1)}]"
|
|
1342
|
+
when 10
|
|
1343
|
+
tz = t&.string(1)
|
|
1344
|
+
"timestamp[#{unit(t ? t.i16(0) : 0)}#{", tz=#{tz}" if tz}]"
|
|
1345
|
+
when 11 then %w[month_interval day_time_interval month_day_nano_interval][t ? t.i16(0) : 0] || "interval"
|
|
1346
|
+
when 12 then "list<#{children.map { |c| child_text(c) }.join(", ")}>"
|
|
1347
|
+
when 13 then "struct<#{children.map { |c| child_text(c) }.join(", ")}>"
|
|
1348
|
+
when 14
|
|
1349
|
+
mode = t&.i16(0)&.positive? ? "dense" : "sparse"
|
|
1350
|
+
ids = t ? t.i32s(1) : []
|
|
1351
|
+
members = children.each_with_index.map { |c, i| "#{child_text(c)}=#{ids[i] || i}" }
|
|
1352
|
+
"#{mode}_union<#{members.join(", ")}>"
|
|
1353
|
+
when 15 then "fixed_size_binary[#{t ? t.i32(0) : 0}]"
|
|
1354
|
+
when 16 then "fixed_size_list<#{children.map { |c| child_text(c) }.join(", ")}>[#{t ? t.i32(0) : 0}]"
|
|
1355
|
+
when 17 then map_name(t, children)
|
|
1356
|
+
when 18 then "duration[#{unit(t ? t.i16(0, 1) : 1)}]"
|
|
1357
|
+
when 19 then "large_binary"
|
|
1358
|
+
when 20 then "large_string"
|
|
1359
|
+
when 21 then "large_list<#{children.map { |c| child_text(c) }.join(", ")}>"
|
|
1360
|
+
when 22 then "run_end_encoded<#{children.map { |c| "#{(c[:name] == "values") ? "values" : "run_ends"}: #{c[:type]}" }.join(", ")}>"
|
|
1361
|
+
when 23 then "binary_view"
|
|
1362
|
+
when 24 then "string_view"
|
|
1363
|
+
when 25 then "list_view<#{children.map { |c| child_text(c) }.join(", ")}>"
|
|
1364
|
+
when 26 then "large_list_view<#{children.map { |c| child_text(c) }.join(", ")}>"
|
|
1365
|
+
else "unknown type #{kind}"
|
|
1366
|
+
end
|
|
1367
|
+
end
|
|
1368
|
+
|
|
1369
|
+
# map<key, value> with non-standard field names in parentheses, as Arrow prints it
|
|
1370
|
+
# @param t [Table, nil] the Map table (keysSorted), nil when absent
|
|
1371
|
+
# @param children [Array<Hash{Symbol => Object}>] the map's single entries struct field
|
|
1372
|
+
# @return [String]
|
|
1373
|
+
def map_name(t, children)
|
|
1374
|
+
entries = children.first
|
|
1375
|
+
kv = entries && entries[:children] || []
|
|
1376
|
+
named = ->(f, std) { f ? "#{f[:type]}#{" ('#{f[:name]}')" unless f[:name] == std}" : "?" }
|
|
1377
|
+
sorted = t&.bool(0) ? ", keys_sorted" : ""
|
|
1378
|
+
entries_name = (entries && entries[:name] != "entries") ? " ('#{entries[:name]}')" : ""
|
|
1379
|
+
"map<#{named.call(kv[0], "key")}, #{named.call(kv[1], "value")}#{sorted}#{entries_name}>"
|
|
1380
|
+
end
|
|
1381
|
+
|
|
1382
|
+
# "name: type" lines for a field and its children, indented, for text output
|
|
1383
|
+
# Children nested deeper than 8 levels are left out.
|
|
1384
|
+
# @param fields [Array<Hash{Symbol => Object}>] fields as returned by #decode under +:fields+
|
|
1385
|
+
# @param depth [Integer] indentation level of +fields+
|
|
1386
|
+
# @param out [Array<String>] accumulator the lines are appended to
|
|
1387
|
+
# @return [Array<String>] +out+
|
|
1388
|
+
def lines(fields, depth = 0, out = [])
|
|
1389
|
+
fields.each do |f|
|
|
1390
|
+
notes = []
|
|
1391
|
+
notes << "not null" unless f[:nullable]
|
|
1392
|
+
notes << "extension #{f[:extension]}" if f[:extension]
|
|
1393
|
+
meta = (f[:metadata] || {}).reject { |k, _| k.start_with?("ARROW:extension:") }
|
|
1394
|
+
notes << "metadata #{meta.map { |k, v| "#{k}=#{v.to_s[0, 60].inspect}" }.join(", ")}" unless meta.empty?
|
|
1395
|
+
out << "#{" " * depth}#{f[:name]}: #{f[:type]}#{" (#{notes.join("; ")})" unless notes.empty?}"
|
|
1396
|
+
lines(f[:children], depth + 1, out) if f[:children] && depth < 8
|
|
1397
|
+
end
|
|
1398
|
+
out
|
|
1399
|
+
end
|
|
1400
|
+
end
|
|
1401
|
+
|
|
1402
|
+
# ---- class helpers ----
|
|
1403
|
+
|
|
1404
|
+
# @param e [Integer] Parquet Encoding value
|
|
1405
|
+
# @return [String] its name, e.g. "RLE_DICTIONARY" (the number as a String when unknown)
|
|
1406
|
+
def self.encoding_name(e) = Format::Encoding::NAMES[e]&.to_s || e.to_s
|
|
1407
|
+
|
|
1408
|
+
# @param column [Schema::Column] leaf column
|
|
1409
|
+
# @return [String] physical type with its logical annotation, e.g. "BYTE_ARRAY STRING" or
|
|
1410
|
+
# "FIXED_LEN_BYTE_ARRAY(16) UUID"
|
|
1411
|
+
def self.type_name(column)
|
|
1412
|
+
node = column.node
|
|
1413
|
+
phys = T::NAMES[node.type].to_s
|
|
1414
|
+
phys += "(#{node.type_length})" if node.type == T::FIXED_LEN_BYTE_ARRAY && node.type_length
|
|
1415
|
+
logical = logical_type_name(node)
|
|
1416
|
+
logical ? "#{phys} #{logical}" : phys
|
|
1417
|
+
end
|
|
1418
|
+
|
|
1419
|
+
# The node's LogicalType, else its ConvertedType, as text
|
|
1420
|
+
# @param node [Schema::Node] schema node (leaf or group)
|
|
1421
|
+
# @return [String, nil] e.g. "INTEGER(8, unsigned)", "TIMESTAMP(MICROS, UTC)" or "DECIMAL(10, 2)";
|
|
1422
|
+
# nil when the node has no annotation
|
|
1423
|
+
def self.logical_type_name(node)
|
|
1424
|
+
if (kind = node.logical_type&.kind)
|
|
1425
|
+
name, payload = kind
|
|
1426
|
+
case name
|
|
1427
|
+
when :integer then "INTEGER(#{payload.bit_width}, #{payload.is_signed ? "signed" : "unsigned"})"
|
|
1428
|
+
when :decimal then "DECIMAL(#{payload.precision}, #{payload.scale})"
|
|
1429
|
+
when :timestamp, :time
|
|
1430
|
+
"#{name.upcase}(#{payload.unit&.to_sym&.upcase}, #{payload.is_adjusted_to_utc ? "UTC" : "local"})"
|
|
1431
|
+
when :unknown then "NULL"
|
|
1432
|
+
else name.to_s.upcase
|
|
1433
|
+
end
|
|
1434
|
+
elsif node.converted_type
|
|
1435
|
+
c = Format::ConvertedType::NAMES[node.converted_type].to_s
|
|
1436
|
+
(c == "DECIMAL") ? "DECIMAL(#{node.precision}, #{node.scale || 0})" : c
|
|
1437
|
+
end
|
|
1438
|
+
end
|
|
1439
|
+
|
|
1440
|
+
# :signed, :unsigned or :unknown, per the Parquet sort order rules for the column's type
|
|
1441
|
+
# @param column [Schema::Column] leaf column
|
|
1442
|
+
# @return [Symbol]
|
|
1443
|
+
def self.sort_order(column)
|
|
1444
|
+
kind, _a, signed = Types.logical_of(column.node)
|
|
1445
|
+
case kind
|
|
1446
|
+
when :integer then (signed == false) ? :unsigned : :signed
|
|
1447
|
+
when :decimal, :date, :time, :timestamp, :float16 then :signed
|
|
1448
|
+
when :string, :enum, :json, :bson, :uuid then :unsigned
|
|
1449
|
+
else
|
|
1450
|
+
case column.type
|
|
1451
|
+
when T::BYTE_ARRAY, T::FIXED_LEN_BYTE_ARRAY then :unsigned
|
|
1452
|
+
when T::INT96 then :unknown
|
|
1453
|
+
else :signed
|
|
1454
|
+
end
|
|
1455
|
+
end
|
|
1456
|
+
end
|
|
1457
|
+
|
|
1458
|
+
# Converts Ruby values (Time, BigDecimal, binary Strings, non-finite Floats...) to JSON-safe ones
|
|
1459
|
+
# Recurses into Hashes, Arrays and Structs; Hash keys other than Symbols become Strings.
|
|
1460
|
+
# @param v [Object] value to convert
|
|
1461
|
+
# @return [Hash, Array, String, Integer, Float, Boolean, nil]
|
|
1462
|
+
def self.jsonable(v)
|
|
1463
|
+
case v
|
|
1464
|
+
when Hash then v.each_with_object({}) { |(k, x), h| h[k.is_a?(Symbol) ? k : k.to_s] = jsonable(x) }
|
|
1465
|
+
when Array then v.map { |x| jsonable(x) }
|
|
1466
|
+
when Struct then v.respond_to?(:to_h) ? jsonable(v.to_h) : v.to_s
|
|
1467
|
+
when String then text(v)
|
|
1468
|
+
when Symbol then v.to_s
|
|
1469
|
+
when Float then v.finite? ? v : v.to_s
|
|
1470
|
+
when Time then v.utc.strftime(v.nsec.zero? ? "%Y-%m-%dT%H:%M:%SZ" : "%Y-%m-%dT%H:%M:%S.%NZ")
|
|
1471
|
+
when Date then v.iso8601
|
|
1472
|
+
when BigDecimal then v.to_s("F")
|
|
1473
|
+
when Integer, true, false, nil then v
|
|
1474
|
+
else v.to_s
|
|
1475
|
+
end
|
|
1476
|
+
end
|
|
1477
|
+
|
|
1478
|
+
# A String as readable text when it is valid UTF-8 without control characters, else as hex
|
|
1479
|
+
# Strings already tagged as valid UTF-8 are returned as they are, control characters and all.
|
|
1480
|
+
# @param s [String] string in any encoding
|
|
1481
|
+
# @return [String]
|
|
1482
|
+
def self.text(s)
|
|
1483
|
+
return s if s.encoding == Encoding::UTF_8 && s.valid_encoding?
|
|
1484
|
+
u = s.dup.force_encoding(Encoding::UTF_8)
|
|
1485
|
+
return u if u.valid_encoding? && !u.match?(/[\x00-\x08\x0e-\x1f\x7f]/)
|
|
1486
|
+
hex(s)
|
|
1487
|
+
end
|
|
1488
|
+
|
|
1489
|
+
# @param bytes [String] bytes to show
|
|
1490
|
+
# @return [String] "0x" and lowercase hex digits; only the first 64 bytes, followed by the
|
|
1491
|
+
# total size, for longer input
|
|
1492
|
+
def self.hex(bytes)
|
|
1493
|
+
b = bytes.b
|
|
1494
|
+
(b.bytesize > 64) ? "0x#{b.byteslice(0, 64).unpack1("H*")}… (#{b.bytesize} bytes)" : "0x#{b.unpack1("H*")}"
|
|
1495
|
+
end
|
|
1496
|
+
|
|
1497
|
+
# A short display form of a decoded value
|
|
1498
|
+
# Strings are quoted (binary ones go through Inspector.text first), nil shows as "null".
|
|
1499
|
+
# @param v [Object] decoded value
|
|
1500
|
+
# @param max [Integer] longest result, in characters; longer ones are cut and end in an ellipsis
|
|
1501
|
+
# @return [String]
|
|
1502
|
+
def self.display(v, max: 40)
|
|
1503
|
+
s = case v
|
|
1504
|
+
when String then (v.encoding == Encoding::BINARY) ? text(v) : v
|
|
1505
|
+
when nil then "null"
|
|
1506
|
+
else jsonable(v).to_s
|
|
1507
|
+
end
|
|
1508
|
+
s = s.inspect if v.is_a?(String)
|
|
1509
|
+
(s.size > max) ? "#{s[0, max - 1]}…" : s
|
|
1510
|
+
end
|
|
1511
|
+
|
|
1512
|
+
# @param n [Integer, nil] byte count
|
|
1513
|
+
# @return [String] e.g. "512 B", "1.50 KB" or "12.3 MB" (binary units); "?" for nil
|
|
1514
|
+
def self.human_bytes(n)
|
|
1515
|
+
return "?" unless n
|
|
1516
|
+
units = %w[B KB MB GB TB]
|
|
1517
|
+
f = n.to_f
|
|
1518
|
+
i = 0
|
|
1519
|
+
while f >= 1024 && i < units.size - 1
|
|
1520
|
+
f /= 1024
|
|
1521
|
+
i += 1
|
|
1522
|
+
end
|
|
1523
|
+
if i.zero?
|
|
1524
|
+
"#{n} B"
|
|
1525
|
+
else
|
|
1526
|
+
format("%.#{(f < 10) ? 2 : 1}f %s", f, units[i])
|
|
1527
|
+
end
|
|
1528
|
+
end
|
|
1529
|
+
|
|
1530
|
+
private
|
|
1531
|
+
|
|
1532
|
+
# @param uncompressed [Integer, nil] uncompressed byte count
|
|
1533
|
+
# @param compressed [Integer, nil] compressed byte count
|
|
1534
|
+
# @return [String] " (3.21x)" for #report, empty when the ratio is unknown
|
|
1535
|
+
def ratio_text(uncompressed, compressed)
|
|
1536
|
+
(compressed.to_i.positive? && uncompressed) ? format(" (%.2fx)", uncompressed.to_f / compressed) : ""
|
|
1537
|
+
end
|
|
1538
|
+
|
|
1539
|
+
# @param page [PageInfo] page whose CRC status to show
|
|
1540
|
+
# @return [String] the status for a #report page line: " crc ok", " CRC MISMATCH", " crc"
|
|
1541
|
+
# (has a CRC, not verified) or empty
|
|
1542
|
+
def crc_text(page)
|
|
1543
|
+
case page.checksum
|
|
1544
|
+
when :ok then " crc ok"
|
|
1545
|
+
when :mismatch then " CRC MISMATCH"
|
|
1546
|
+
else page.crc ? " crc" : ""
|
|
1547
|
+
end
|
|
1548
|
+
end
|
|
1549
|
+
|
|
1550
|
+
# @param m [Hash{Symbol => Object}] an entry of #index_mismatches
|
|
1551
|
+
# @return [String] one #report line describing it
|
|
1552
|
+
def index_mismatch_text(m)
|
|
1553
|
+
where = "row group #{m[:row_group]} #{m[:column]}"
|
|
1554
|
+
if m[:field] == :page_count
|
|
1555
|
+
"#{where}: #{m[:page_value]} data pages but #{m[:index_value]} column index entries"
|
|
1556
|
+
else
|
|
1557
|
+
"#{where} page #{m[:page]}: #{m[:field]} in page header #{Inspector.display(m[:page_value])}, " \
|
|
1558
|
+
"in column index #{Inspector.display(m[:index_value])}"
|
|
1559
|
+
end
|
|
1560
|
+
end
|
|
1561
|
+
|
|
1562
|
+
# Adds :arrow_type to schema nodes with a same-named Arrow field (top level, and struct members)
|
|
1563
|
+
# @param nodes [Array<Hash{Symbol => Object}>] #schema_tree nodes, updated in place
|
|
1564
|
+
# @param fields [Array<Hash{Symbol => Object}>] Arrow fields at the same level
|
|
1565
|
+
# @param depth [Integer] nesting depth; recursion stops at 32
|
|
1566
|
+
# @return [void]
|
|
1567
|
+
def annotate_arrow_types(nodes, fields, depth = 0)
|
|
1568
|
+
by_name = fields.to_h { |f| [f[:name], f] }
|
|
1569
|
+
nodes.each do |n|
|
|
1570
|
+
f = by_name[n[:name]] or next
|
|
1571
|
+
n[:arrow_type] = f[:type]
|
|
1572
|
+
if n[:children] && f[:children] && f[:type].start_with?("struct<") && depth < 32
|
|
1573
|
+
annotate_arrow_types(n[:children], f[:children], depth + 1)
|
|
1574
|
+
end
|
|
1575
|
+
end
|
|
1576
|
+
end
|
|
1577
|
+
|
|
1578
|
+
# Whether the index bound +idx+ excludes values the page's bound +page+ says are present
|
|
1579
|
+
# (index min above the page min, or index max below the page max). Truncated binary bounds
|
|
1580
|
+
# (one a prefix of the other) and values that can't be compared are never reported.
|
|
1581
|
+
# @param idx [Object, nil] decoded ColumnIndex bound
|
|
1582
|
+
# @param page [Object, nil] decoded page header bound
|
|
1583
|
+
# @param order [Symbol] :signed, :unsigned or :unknown (see Inspector.sort_order)
|
|
1584
|
+
# @param which [Symbol] :min or :max
|
|
1585
|
+
# @return [Boolean]
|
|
1586
|
+
def narrower?(idx, page, order, which)
|
|
1587
|
+
return false if idx.nil? || page.nil? || order == :unknown
|
|
1588
|
+
a, b = (which == :min) ? [page, idx] : [idx, page] # true when a < b
|
|
1589
|
+
if a.is_a?(String) && b.is_a?(String)
|
|
1590
|
+
a = a.b
|
|
1591
|
+
b = b.b
|
|
1592
|
+
return false if a.start_with?(b) || b.start_with?(a)
|
|
1593
|
+
return (order == :unsigned) ? a < b : false
|
|
1594
|
+
end
|
|
1595
|
+
return false if a.is_a?(Float) && a.nan? || b.is_a?(Float) && b.nan?
|
|
1596
|
+
a = a ? 1 : 0 if a == true || a == false
|
|
1597
|
+
b = b ? 1 : 0 if b == true || b == false
|
|
1598
|
+
(a <=> b) == -1
|
|
1599
|
+
rescue
|
|
1600
|
+
false
|
|
1601
|
+
end
|
|
1602
|
+
|
|
1603
|
+
# Overall min or max of per-chunk bounds, when they can be compared
|
|
1604
|
+
# @param values [Array<Object, nil>] decoded per-chunk minimums or maximums
|
|
1605
|
+
# @param which [Symbol] :min or :max
|
|
1606
|
+
# @return [Object, nil] nil when there are no values or they are of mixed or incomparable types
|
|
1607
|
+
def safe_extreme(values, which)
|
|
1608
|
+
vals = values.compact
|
|
1609
|
+
return nil if vals.empty?
|
|
1610
|
+
if vals.all? { |v| v == true || v == false }
|
|
1611
|
+
return (which == :min) ? vals.all? : vals.any?
|
|
1612
|
+
end
|
|
1613
|
+
return nil unless vals.map(&:class).uniq.size == 1
|
|
1614
|
+
(which == :min) ? vals.min : vals.max
|
|
1615
|
+
rescue ArgumentError, NoMethodError
|
|
1616
|
+
nil
|
|
1617
|
+
end
|
|
1618
|
+
|
|
1619
|
+
# Reads the file size, the footer length and magic, and decodes the FileMetaData into @metadata
|
|
1620
|
+
# @return [void]
|
|
1621
|
+
# @raise [FormatError] when the file is too small, lacks the magic bytes or has a corrupt footer
|
|
1622
|
+
# @raise [UnsupportedError] when the file is encrypted (PARE magic)
|
|
1623
|
+
def read_footer
|
|
1624
|
+
@io.seek(0, IO::SEEK_END)
|
|
1625
|
+
@file_size = @io.pos
|
|
1626
|
+
raise FormatError, "File too small to be Parquet (#{@file_size} bytes)" if @file_size < 12
|
|
1627
|
+
tail = read_at(@file_size - 8, 8)
|
|
1628
|
+
raise UnsupportedError, "Encrypted Parquet files are not supported" if tail.byteslice(4, 4) == "PARE"
|
|
1629
|
+
raise FormatError, "Missing PAR1 footer magic" unless tail.byteslice(4, 4) == MAGIC
|
|
1630
|
+
@footer_size = tail.unpack1("V")
|
|
1631
|
+
raise FormatError, "Footer length #{@footer_size} exceeds file size" if @footer_size + 12 > @file_size
|
|
1632
|
+
@metadata = Format::FileMetaData.decode(read_at(footer_offset, @footer_size)).first
|
|
1633
|
+
rescue Thrift::Error => e
|
|
1634
|
+
raise FormatError, "Corrupt file metadata: #{e.message}"
|
|
1635
|
+
end
|
|
1636
|
+
|
|
1637
|
+
# Reads +len+ bytes at +pos+ through a small read-ahead window, so walking many small pages
|
|
1638
|
+
# does not cost a syscall per header
|
|
1639
|
+
# @param pos [Integer] file offset
|
|
1640
|
+
# @param len [Integer] bytes wanted
|
|
1641
|
+
# @return [String] binary String; shorter than +len+ at the end of the file
|
|
1642
|
+
def read_at(pos, len)
|
|
1643
|
+
if @window && pos >= @window_pos && pos + len <= @window_pos + @window.bytesize
|
|
1644
|
+
return @window.byteslice(pos - @window_pos, len)
|
|
1645
|
+
end
|
|
1646
|
+
@io.seek(pos)
|
|
1647
|
+
if len >= WINDOW
|
|
1648
|
+
(@io.read(len) || "".b).b
|
|
1649
|
+
else
|
|
1650
|
+
@window_pos = pos
|
|
1651
|
+
@window = (@io.read(WINDOW) || "".b).b
|
|
1652
|
+
@window.byteslice(0, len)
|
|
1653
|
+
end
|
|
1654
|
+
end
|
|
1655
|
+
|
|
1656
|
+
# Decodes the page header at +pos+, reading more bytes when it is larger than the first guess
|
|
1657
|
+
# (page statistics of long strings can make headers big)
|
|
1658
|
+
# @param pos [Integer] file offset of the header
|
|
1659
|
+
# @param limit [Integer] offset the header must not extend past (the footer's start)
|
|
1660
|
+
# @return [Array(Format::PageHeader, Integer)] the header and its encoded size in bytes
|
|
1661
|
+
# @raise [FormatError] when there is no room for a header before +limit+
|
|
1662
|
+
# @raise [Thrift::Error] when it does not decode within +limit+ or 16 MiB
|
|
1663
|
+
def read_page_header(pos, limit)
|
|
1664
|
+
want = 256
|
|
1665
|
+
while true
|
|
1666
|
+
avail = [want, limit - pos].min
|
|
1667
|
+
raise FormatError, "no room for a page header" if avail <= 0
|
|
1668
|
+
buf = read_at(pos, avail)
|
|
1669
|
+
begin
|
|
1670
|
+
header, size = Format::PageHeader.decode(buf)
|
|
1671
|
+
return [header, size]
|
|
1672
|
+
rescue Thrift::Error
|
|
1673
|
+
raise if avail < want || want >= 16 * 1024 * 1024
|
|
1674
|
+
want *= 8
|
|
1675
|
+
end
|
|
1676
|
+
end
|
|
1677
|
+
end
|
|
1678
|
+
|
|
1679
|
+
# Builds a PageInfo from a decoded page header
|
|
1680
|
+
# @param index [Integer] position of the page in its chunk
|
|
1681
|
+
# @param h [Format::PageHeader] decoded header
|
|
1682
|
+
# @param pos [Integer] file offset of the header
|
|
1683
|
+
# @param header_size [Integer] encoded size of the header in bytes
|
|
1684
|
+
# @param column [Schema::Column] column the page belongs to (for statistics and row counts)
|
|
1685
|
+
# @return [PageInfo]
|
|
1686
|
+
# @raise [FormatError] when the header declares a negative compressed size
|
|
1687
|
+
def page_info(index, h, pos, header_size, column)
|
|
1688
|
+
info = PageInfo.new(
|
|
1689
|
+
index: index,
|
|
1690
|
+
type: Format::PageType::NAMES[h.type] || h.type,
|
|
1691
|
+
offset: pos,
|
|
1692
|
+
header_size: header_size,
|
|
1693
|
+
compressed_size: h.compressed_page_size.to_i,
|
|
1694
|
+
uncompressed_size: h.uncompressed_page_size.to_i,
|
|
1695
|
+
crc: h.crc
|
|
1696
|
+
)
|
|
1697
|
+
raise FormatError, "negative page size #{info.compressed_size}" if info.compressed_size.negative?
|
|
1698
|
+
if (d = h.data_page_header)
|
|
1699
|
+
info.num_values = d.num_values
|
|
1700
|
+
info.encoding = Inspector.encoding_name(d.encoding)
|
|
1701
|
+
info.definition_level_encoding = Inspector.encoding_name(d.definition_level_encoding) if d.definition_level_encoding
|
|
1702
|
+
info.repetition_level_encoding = Inspector.encoding_name(d.repetition_level_encoding) if d.repetition_level_encoding
|
|
1703
|
+
info.statistics = decode_statistics(d.statistics, column)
|
|
1704
|
+
info.num_nulls = info.statistics&.null_count
|
|
1705
|
+
# Without repetition every value is a row
|
|
1706
|
+
info.num_rows = d.num_values if column.max_repetition_level.zero?
|
|
1707
|
+
elsif (d = h.data_page_header_v2)
|
|
1708
|
+
info.num_values = d.num_values
|
|
1709
|
+
info.num_nulls = d.num_nulls
|
|
1710
|
+
info.num_rows = d.num_rows
|
|
1711
|
+
info.encoding = Inspector.encoding_name(d.encoding)
|
|
1712
|
+
info.definition_levels_byte_length = d.definition_levels_byte_length
|
|
1713
|
+
info.repetition_levels_byte_length = d.repetition_levels_byte_length
|
|
1714
|
+
info.is_compressed = d.is_compressed.nil? || d.is_compressed
|
|
1715
|
+
info.statistics = decode_statistics(d.statistics, column)
|
|
1716
|
+
elsif (d = h.dictionary_page_header)
|
|
1717
|
+
info.num_values = d.num_values
|
|
1718
|
+
info.encoding = Inspector.encoding_name(d.encoding)
|
|
1719
|
+
info.is_sorted = d.is_sorted
|
|
1720
|
+
end
|
|
1721
|
+
info
|
|
1722
|
+
end
|
|
1723
|
+
|
|
1724
|
+
# One #key_value_metadata entry
|
|
1725
|
+
# @param key [String] metadata key
|
|
1726
|
+
# @param value [String, nil] metadata value
|
|
1727
|
+
# @return [Hash{Symbol => Object}] see #key_value_metadata
|
|
1728
|
+
def describe_key_value(key, value)
|
|
1729
|
+
value = value.to_s
|
|
1730
|
+
h = {key: key, bytesize: value.bytesize}
|
|
1731
|
+
if key == "ARROW:schema"
|
|
1732
|
+
h[:format] = "arrow_schema"
|
|
1733
|
+
h[:value] = (value.size > 120) ? "#{value[0, 120]}…" : value
|
|
1734
|
+
if (arrow = arrow_schema)
|
|
1735
|
+
h[:summary] = "Arrow schema, #{arrow[:fields].size} field#{"s" unless arrow[:fields].size == 1}"
|
|
1736
|
+
h[:arrow_fields] = arrow[:fields].map { |f| f[:name] }
|
|
1737
|
+
meta = arrow[:metadata]&.transform_values { |v| (v.size > 4000) ? "#{v[0, 4000]}…" : v }
|
|
1738
|
+
h[:arrow_schema] = arrow.merge(metadata: meta).compact
|
|
1739
|
+
else
|
|
1740
|
+
h[:summary] = "Arrow IPC schema message, base64-encoded (#{value.bytesize} bytes)"
|
|
1741
|
+
h[:arrow_error] = arrow_schema_error
|
|
1742
|
+
names = arrow_field_names(value)
|
|
1743
|
+
h[:arrow_fields] = names if names
|
|
1744
|
+
end
|
|
1745
|
+
elsif value.lstrip.start_with?("{", "[") && value.bytesize < 4 * 1024 * 1024
|
|
1746
|
+
begin
|
|
1747
|
+
h[:json] = JSON.parse(value)
|
|
1748
|
+
h[:format] = "json"
|
|
1749
|
+
h[:summary] = (key == "pandas") ? pandas_summary(h[:json]) : "JSON"
|
|
1750
|
+
rescue JSON::ParserError
|
|
1751
|
+
h[:format] = "text"
|
|
1752
|
+
end
|
|
1753
|
+
h[:value] = (value.size > 4000) ? "#{value[0, 4000]}…" : value
|
|
1754
|
+
else
|
|
1755
|
+
t = Inspector.text(value.b)
|
|
1756
|
+
h[:format] = (t.start_with?("0x") && value.bytesize.positive?) ? "binary" : "text"
|
|
1757
|
+
h[:value] = (t.size > 4000) ? "#{t[0, 4000]}…" : t
|
|
1758
|
+
end
|
|
1759
|
+
h
|
|
1760
|
+
end
|
|
1761
|
+
|
|
1762
|
+
# @param json [Object] the parsed "pandas" metadata value
|
|
1763
|
+
# @return [String] e.g. "pandas 2.2.0 metadata, 3 columns"
|
|
1764
|
+
def pandas_summary(json)
|
|
1765
|
+
return "pandas metadata" unless json.is_a?(Hash)
|
|
1766
|
+
cols = json["columns"]&.size
|
|
1767
|
+
"pandas #{json["pandas_version"]} metadata, #{cols || "?"} columns"
|
|
1768
|
+
end
|
|
1769
|
+
|
|
1770
|
+
# Field names from an Arrow IPC schema message, found by scanning for its flatbuffer strings.
|
|
1771
|
+
# Best effort: only used to label the blob, nil when nothing sensible is found.
|
|
1772
|
+
# Only top-level Parquet column names that occur in the decoded bytes are reported.
|
|
1773
|
+
# @param b64 [String] the base64 ARROW:schema value
|
|
1774
|
+
# @return [Array<String>, nil]
|
|
1775
|
+
def arrow_field_names(b64)
|
|
1776
|
+
bytes = b64.unpack1("m")
|
|
1777
|
+
schema_cols = columns.map { |c| c.path.first }.uniq
|
|
1778
|
+
found = schema_cols.select { |n| bytes.include?(n.b) }
|
|
1779
|
+
found.empty? ? nil : found
|
|
1780
|
+
rescue ArgumentError
|
|
1781
|
+
nil
|
|
1782
|
+
end
|
|
1783
|
+
end
|
|
1784
|
+
end
|