herringbone 0.2.0 → 0.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/README.md +56 -4
- data/lib/herringbone/active_record.rb +44 -12
- data/lib/herringbone/bloom_filter.rb +112 -12
- data/lib/herringbone/byte_values.rb +29 -0
- data/lib/herringbone/codecs/lz4.rb +66 -3
- data/lib/herringbone/codecs/snappy.rb +132 -3
- data/lib/herringbone/compression.rb +48 -0
- data/lib/herringbone/encodings/delta.rb +70 -13
- data/lib/herringbone/encodings/plain.rb +22 -0
- data/lib/herringbone/encodings/rle.rb +53 -1
- data/lib/herringbone/format.rb +150 -0
- data/lib/herringbone/inspector.rb +477 -81
- data/lib/herringbone/io_buffer_support.rb +3 -1
- data/lib/herringbone/reader/column_chunk_reader.rb +98 -6
- data/lib/herringbone/reader/column_cursor.rb +46 -2
- data/lib/herringbone/reader/numo.rb +365 -0
- data/lib/herringbone/reader/page_stream.rb +280 -4
- data/lib/herringbone/reader/scan.rb +103 -9
- data/lib/herringbone/reader.rb +308 -17
- data/lib/herringbone/schema.rb +306 -15
- data/lib/herringbone/thrift.rb +155 -4
- data/lib/herringbone/types.rb +176 -41
- data/lib/herringbone/version.rb +2 -1
- data/lib/herringbone/visualizer.rb +51 -11
- data/lib/herringbone/writer.rb +308 -31
- data/lib/herringbone/xxhash.rb +108 -2
- data/lib/herringbone.rb +28 -0
- metadata +3 -2
|
@@ -21,8 +21,11 @@ module Herringbone
|
|
|
21
21
|
# inspector no longer needs the IO. The IO is never closed. After #verify_checksums, page CRC
|
|
22
22
|
# results are included in every output.
|
|
23
23
|
class Inspector
|
|
24
|
+
# Shorthand for the Parquet physical type constants
|
|
24
25
|
T = Format::Type
|
|
26
|
+
# Magic bytes at the start and end of every (unencrypted) Parquet file
|
|
25
27
|
MAGIC = "PAR1"
|
|
28
|
+
# Size of the read-ahead window used by #read_at, in bytes
|
|
26
29
|
WINDOW = 64 * 1024
|
|
27
30
|
|
|
28
31
|
# Just enough of parquet.thrift's BloomFilterHeader to learn the filter's size
|
|
@@ -30,29 +33,50 @@ module Herringbone
|
|
|
30
33
|
field 1, :num_bytes, :i32
|
|
31
34
|
end
|
|
32
35
|
|
|
33
|
-
# Decoded min/max statistics (from a column chunk, a page header or a page index entry)
|
|
36
|
+
# Decoded min/max statistics (from a column chunk, a page header or a page index entry).
|
|
37
|
+
# +min+/+max+ are converted with the column's type converter (hex Strings when they could
|
|
38
|
+
# not be decoded); +min_exact+/+max_exact+ mirror is_min_value_exact/is_max_value_exact;
|
|
39
|
+
# +source+ says which Thrift fields they came from and +caveat+ why they may be unreliable.
|
|
34
40
|
Stats = Struct.new(:min, :max, :null_count, :distinct_count, :min_exact, :max_exact, :source, :caveat,
|
|
35
41
|
keyword_init: true) do
|
|
42
|
+
# @return [Hash{Symbol => Object}] the set members, JSON-safe (see Inspector.jsonable)
|
|
36
43
|
def to_h = Inspector.jsonable(super.compact)
|
|
37
44
|
end
|
|
38
45
|
|
|
39
46
|
# One page header of a column chunk. +checksum+ is nil until the CRCs are verified
|
|
40
|
-
# (Inspector#verify_checksums), then :ok, :mismatch or :absent (the page has no CRC)
|
|
47
|
+
# (Inspector#verify_checksums), then :ok, :mismatch or :absent (the page has no CRC);
|
|
48
|
+
# +actual_crc+ is then the CRC32 of the page's stored bytes (nil when it has no CRC).
|
|
49
|
+
# +index+ is the page's position in the chunk, +offset+ where its header starts; +type+ is
|
|
50
|
+
# a PageType name such as :DATA_PAGE (the raw Integer when unknown). +first_row_index+ and,
|
|
51
|
+
# for v1 pages of repeated columns, +num_rows+ are only known from the OffsetIndex.
|
|
41
52
|
PageInfo = Struct.new(:index, :type, :offset, :header_size, :compressed_size, :uncompressed_size,
|
|
42
53
|
:num_values, :num_nulls, :num_rows, :first_row_index, :encoding, :definition_level_encoding,
|
|
43
54
|
:repetition_level_encoding, :definition_levels_byte_length, :repetition_levels_byte_length,
|
|
44
|
-
:is_compressed, :is_sorted, :statistics, :crc, :checksum, keyword_init: true) do
|
|
55
|
+
:is_compressed, :is_sorted, :statistics, :crc, :checksum, :actual_crc, keyword_init: true) do
|
|
56
|
+
# @return [Integer] bytes taken by the page in the file: header plus compressed body
|
|
45
57
|
def total_size = header_size + compressed_size
|
|
58
|
+
|
|
59
|
+
# @return [Integer] file offset just past the page's body
|
|
46
60
|
def end_offset = offset + total_size
|
|
61
|
+
|
|
62
|
+
# @return [Integer] file offset where the page's body (after the header) starts
|
|
47
63
|
def body_offset = offset + header_size
|
|
64
|
+
|
|
65
|
+
# @return [Boolean] whether this is a dictionary page
|
|
48
66
|
def dictionary? = type == :DICTIONARY_PAGE
|
|
67
|
+
|
|
68
|
+
# @return [Boolean] whether this is a v1 or v2 data page
|
|
49
69
|
def data? = type == :DATA_PAGE || type == :DATA_PAGE_V2
|
|
50
70
|
|
|
51
71
|
# The CRC from the header as an unsigned 32-bit value (Thrift stores it as a signed i32)
|
|
72
|
+
# @return [Integer, nil] nil when the header has no CRC
|
|
52
73
|
def expected_crc = crc && (crc & 0xFFFF_FFFF)
|
|
53
74
|
|
|
75
|
+
# @return [Hash{Symbol => Object}] the set members, JSON-safe; +:crc+ becomes a Boolean
|
|
76
|
+
# saying whether the header has a CRC, +:actual_crc+ is left out
|
|
54
77
|
def to_h
|
|
55
78
|
h = super
|
|
79
|
+
h.delete(:actual_crc)
|
|
56
80
|
h[:statistics] = statistics&.to_h
|
|
57
81
|
h[:crc] = !crc.nil?
|
|
58
82
|
Inspector.jsonable(h.compact)
|
|
@@ -62,13 +86,19 @@ module Herringbone
|
|
|
62
86
|
# ColumnIndex of one column chunk, with min/max decoded per page (nil for all-null pages)
|
|
63
87
|
ColumnIndexInfo = Struct.new(:offset, :length, :null_pages, :min_values, :max_values, :boundary_order,
|
|
64
88
|
:null_counts, :repetition_level_histograms, :definition_level_histograms, keyword_init: true) do
|
|
89
|
+
# @return [Hash{Symbol => Object}] the set members, JSON-safe (see Inspector.jsonable)
|
|
65
90
|
def to_h = Inspector.jsonable(super.compact)
|
|
66
91
|
end
|
|
67
92
|
|
|
93
|
+
# One OffsetIndex entry: where a data page starts, its size (header included) and the index
|
|
94
|
+
# of its first row within the row group
|
|
68
95
|
PageLocation = Struct.new(:offset, :compressed_page_size, :first_row_index, keyword_init: true)
|
|
69
96
|
|
|
97
|
+
# OffsetIndex of one column chunk: its own +offset+/+length+ in the file, the PageLocations
|
|
98
|
+
# of its data pages and the optional per-page unencoded BYTE_ARRAY sizes
|
|
70
99
|
OffsetIndexInfo = Struct.new(:offset, :length, :page_locations, :unencoded_byte_array_data_bytes,
|
|
71
100
|
keyword_init: true) do
|
|
101
|
+
# @return [Hash{Symbol => Object}] the set members, with +:page_locations+ as Hashes
|
|
72
102
|
def to_h
|
|
73
103
|
h = super
|
|
74
104
|
h[:page_locations] = page_locations.map(&:to_h)
|
|
@@ -80,6 +110,10 @@ module Herringbone
|
|
|
80
110
|
class ColumnChunkInfo
|
|
81
111
|
attr_reader :inspector, :row_group, :column, :chunk, :meta, :error
|
|
82
112
|
|
|
113
|
+
# @param inspector [Inspector] owner, used to read page headers and indexes lazily
|
|
114
|
+
# @param row_group [RowGroupInfo] row group the chunk belongs to
|
|
115
|
+
# @param column [Schema::Column] leaf column the chunk stores
|
|
116
|
+
# @param chunk [Format::ColumnChunk] chunk as decoded from the footer
|
|
83
117
|
def initialize(inspector, row_group, column, chunk)
|
|
84
118
|
@inspector = inspector
|
|
85
119
|
@row_group = row_group
|
|
@@ -88,63 +122,95 @@ module Herringbone
|
|
|
88
122
|
@meta = chunk.meta_data
|
|
89
123
|
end
|
|
90
124
|
|
|
125
|
+
# @return [String] dotted path of the column, e.g. +"address.city"+
|
|
91
126
|
def path = @column.dotted_path
|
|
127
|
+
|
|
128
|
+
# @return [Integer] position of the column among the schema's leaf columns
|
|
92
129
|
def column_index_number = @column.index
|
|
130
|
+
|
|
131
|
+
# @return [Symbol, Integer] codec name such as :SNAPPY (the raw Integer when unknown)
|
|
93
132
|
def codec = Format::Codec::NAMES[@meta.codec] || @meta.codec
|
|
133
|
+
|
|
134
|
+
# @return [Array<String>] encoding names listed in the column metadata
|
|
94
135
|
def encodings = (@meta.encodings || []).map { |e| Inspector.encoding_name(e) }
|
|
136
|
+
|
|
137
|
+
# @return [Integer] values in the chunk, nulls and repeated entries included
|
|
95
138
|
def num_values = @meta.num_values
|
|
139
|
+
|
|
140
|
+
# @return [Integer] total_compressed_size from the column metadata (page headers included)
|
|
96
141
|
def compressed_size = @meta.total_compressed_size
|
|
142
|
+
|
|
143
|
+
# @return [Integer] total_uncompressed_size from the column metadata (page headers included)
|
|
97
144
|
def uncompressed_size = @meta.total_uncompressed_size
|
|
145
|
+
|
|
146
|
+
# @return [Integer] offset of the first data page, as declared in the column metadata
|
|
98
147
|
def data_page_offset = @meta.data_page_offset
|
|
148
|
+
|
|
149
|
+
# @return [String, nil] path of the file holding the chunk when it is not stored in this one
|
|
99
150
|
def external_file = @chunk.file_path
|
|
100
151
|
|
|
152
|
+
# @return [Float, nil] uncompressed size divided by compressed size; nil when nothing is compressed
|
|
101
153
|
def compression_ratio
|
|
102
154
|
compressed_size.to_i.positive? ? uncompressed_size.to_f / compressed_size : nil
|
|
103
155
|
end
|
|
104
156
|
|
|
105
157
|
# Some writers store 0 when there is no dictionary page, and some store a data_page_offset
|
|
106
158
|
# of 0 for empty chunks (which would point at the magic bytes)
|
|
159
|
+
# @return [Integer, nil] the declared dictionary page offset, nil when absent or implausible
|
|
107
160
|
def dictionary_page_offset
|
|
108
161
|
d = @meta.dictionary_page_offset
|
|
109
|
-
d && d >= 4 && (d < @meta.data_page_offset.to_i || @meta.data_page_offset.to_i < 4) ? d : nil
|
|
162
|
+
(d && d >= 4 && (d < @meta.data_page_offset.to_i || @meta.data_page_offset.to_i < 4)) ? d : nil
|
|
110
163
|
end
|
|
111
164
|
|
|
112
165
|
# Where the chunk's first page starts
|
|
166
|
+
# @return [Integer] file offset
|
|
113
167
|
def start_offset = dictionary_page_offset || data_page_offset
|
|
114
168
|
|
|
115
169
|
# The end according to the metadata; some writers under-report it (see #end_offset)
|
|
170
|
+
# @return [Integer] file offset just past the chunk
|
|
116
171
|
def declared_end_offset = start_offset + compressed_size
|
|
117
172
|
|
|
118
173
|
# The end of the last page actually found (the declared end if the pages could not be walked)
|
|
174
|
+
# @return [Integer] file offset just past the chunk, never before #declared_end_offset
|
|
119
175
|
def end_offset
|
|
120
176
|
last = pages.last
|
|
121
177
|
last ? [last.end_offset, declared_end_offset].max : declared_end_offset
|
|
122
178
|
end
|
|
123
179
|
|
|
180
|
+
# Page counts per page type and encoding, from the column metadata's encoding_stats
|
|
181
|
+
# @return [Array<Hash{Symbol => Object}>] each { page_type:, encoding:, count: }; empty when
|
|
182
|
+
# the writer stored none
|
|
124
183
|
def encoding_stats
|
|
125
184
|
(@meta.encoding_stats || []).map do |s|
|
|
126
|
-
{
|
|
127
|
-
|
|
185
|
+
{page_type: Format::PageType::NAMES[s.page_type] || s.page_type,
|
|
186
|
+
encoding: Inspector.encoding_name(s.encoding), count: s.count}
|
|
128
187
|
end
|
|
129
188
|
end
|
|
130
189
|
|
|
190
|
+
# Chunk-level statistics from the column metadata, decoded (memoized)
|
|
191
|
+
# @return [Stats, nil] nil when the chunk has no statistics
|
|
131
192
|
def statistics
|
|
132
193
|
return @statistics if defined?(@statistics)
|
|
133
194
|
@statistics = @inspector.decode_statistics(@meta.statistics, @column)
|
|
134
195
|
end
|
|
135
196
|
|
|
197
|
+
# @return [Hash{Symbol => Object}, nil] the SizeStatistics (unencoded byte array sizes and
|
|
198
|
+
# level histograms) as a Hash, nil when absent
|
|
136
199
|
def size_statistics
|
|
137
200
|
s = @meta.size_statistics
|
|
138
201
|
s&.to_h
|
|
139
202
|
end
|
|
140
203
|
|
|
204
|
+
# @return [Hash{String => String}] the chunk's own key/value metadata (rarely used by writers)
|
|
141
205
|
def key_value_metadata
|
|
142
206
|
(@meta.key_value_metadata || []).to_h { |kv| [kv.key, kv.value] }
|
|
143
207
|
end
|
|
144
208
|
|
|
209
|
+
# @return [Integer, nil] file offset of the chunk's bloom filter, nil when it has none
|
|
145
210
|
def bloom_filter_offset = @meta.bloom_filter_offset
|
|
146
211
|
|
|
147
212
|
# Bytes taken by the bloom filter (header + bitset); read from its header when the footer has no length
|
|
213
|
+
# @return [Integer, nil] nil when there is no bloom filter or its header can't be decoded
|
|
148
214
|
def bloom_filter_length
|
|
149
215
|
return nil unless bloom_filter_offset
|
|
150
216
|
return @meta.bloom_filter_length if @meta.bloom_filter_length
|
|
@@ -152,18 +218,23 @@ module Herringbone
|
|
|
152
218
|
@bloom_filter_length = @inspector.bloom_filter_size(bloom_filter_offset)
|
|
153
219
|
end
|
|
154
220
|
|
|
221
|
+
# @return [Array(Integer, Integer), nil] [offset, length] of the chunk's ColumnIndex, nil when
|
|
222
|
+
# it has none
|
|
155
223
|
def column_index_range
|
|
156
224
|
o = @chunk.column_index_offset
|
|
157
|
-
o && @chunk.column_index_length ? [o, @chunk.column_index_length] : nil
|
|
225
|
+
(o && @chunk.column_index_length) ? [o, @chunk.column_index_length] : nil
|
|
158
226
|
end
|
|
159
227
|
|
|
228
|
+
# @return [Array(Integer, Integer), nil] [offset, length] of the chunk's OffsetIndex, nil when
|
|
229
|
+
# it has none
|
|
160
230
|
def offset_index_range
|
|
161
231
|
o = @chunk.offset_index_offset
|
|
162
|
-
o && @chunk.offset_index_length ? [o, @chunk.offset_index_length] : nil
|
|
232
|
+
(o && @chunk.offset_index_length) ? [o, @chunk.offset_index_length] : nil
|
|
163
233
|
end
|
|
164
234
|
|
|
165
235
|
# All page headers, in file order. Errors while walking (corrupt or truncated headers) end
|
|
166
236
|
# the walk; #error then says what went wrong and #pages holds the pages found before it.
|
|
237
|
+
# @return [Array<PageInfo>]
|
|
167
238
|
def pages
|
|
168
239
|
@pages ||= begin
|
|
169
240
|
list, @error = @inspector.walk_pages(self)
|
|
@@ -172,22 +243,32 @@ module Herringbone
|
|
|
172
243
|
end
|
|
173
244
|
end
|
|
174
245
|
|
|
246
|
+
# @return [Array<PageInfo>] the v1 and v2 data pages, in file order
|
|
175
247
|
def data_pages = pages.select(&:data?)
|
|
248
|
+
|
|
249
|
+
# @return [PageInfo, nil] the dictionary page, nil when the chunk has none
|
|
176
250
|
def dictionary_page = pages.find(&:dictionary?)
|
|
177
251
|
|
|
178
252
|
# Number of entries in the dictionary (from its page header)
|
|
253
|
+
# @return [Integer, nil] nil when the chunk has no dictionary page
|
|
179
254
|
def dictionary_size = dictionary_page&.num_values
|
|
180
255
|
|
|
256
|
+
# The chunk's ColumnIndex, read and decoded on first use
|
|
257
|
+
# @return [ColumnIndexInfo, nil] nil when there is none or it is corrupt
|
|
181
258
|
def column_index
|
|
182
259
|
return @column_index if defined?(@column_index)
|
|
183
260
|
@column_index = @inspector.read_column_index(self)
|
|
184
261
|
end
|
|
185
262
|
|
|
263
|
+
# The chunk's OffsetIndex, read and decoded on first use
|
|
264
|
+
# @return [OffsetIndexInfo, nil] nil when there is none or it is corrupt
|
|
186
265
|
def offset_index
|
|
187
266
|
return @offset_index if defined?(@offset_index)
|
|
188
267
|
@offset_index = @inspector.read_offset_index(self)
|
|
189
268
|
end
|
|
190
269
|
|
|
270
|
+
# Null count from the chunk statistics, else the sum over the data page headers
|
|
271
|
+
# @return [Integer, nil] nil when neither the statistics nor every data page has it
|
|
191
272
|
def null_count
|
|
192
273
|
return statistics.null_count if statistics&.null_count
|
|
193
274
|
counts = data_pages.map(&:num_nulls)
|
|
@@ -196,6 +277,7 @@ module Herringbone
|
|
|
196
277
|
|
|
197
278
|
# Reads every page body (compressed bytes, as stored) and checks it against the CRC in its
|
|
198
279
|
# page header. Sets PageInfo#checksum on each page and returns the pages' statuses.
|
|
280
|
+
# @return [Array<Symbol>] :ok, :mismatch or :absent per page, in #pages order
|
|
199
281
|
def verify_checksums
|
|
200
282
|
pages.map { |p| p.checksum = @inspector.page_checksum(p) }
|
|
201
283
|
end
|
|
@@ -207,10 +289,14 @@ module Herringbone
|
|
|
207
289
|
# A ColumnIndex bound that is wider than the page's (e.g. a truncated string prefix) is
|
|
208
290
|
# allowed; one that is narrower, or a differing null count, is reported. Empty when the
|
|
209
291
|
# chunk has no ColumnIndex or its pages carry no statistics.
|
|
292
|
+
# @return [Array<Hash{Symbol => Object}>]
|
|
210
293
|
def index_mismatches
|
|
211
294
|
@index_mismatches ||= @inspector.compare_page_index(self)
|
|
212
295
|
end
|
|
213
296
|
|
|
297
|
+
# Everything known about the chunk, including its pages and page indexes (reads them if
|
|
298
|
+
# they were not read yet)
|
|
299
|
+
# @return [Hash{Symbol => Object}] JSON-safe; keys without a value are left out
|
|
214
300
|
def to_h
|
|
215
301
|
Inspector.jsonable({
|
|
216
302
|
path: path,
|
|
@@ -229,13 +315,13 @@ module Herringbone
|
|
|
229
315
|
end_offset: end_offset,
|
|
230
316
|
data_page_offset: data_page_offset,
|
|
231
317
|
dictionary_page_offset: dictionary_page_offset,
|
|
232
|
-
dictionary: dictionary_page && {
|
|
233
|
-
|
|
234
|
-
|
|
318
|
+
dictionary: dictionary_page && {offset: dictionary_page.offset, num_values: dictionary_page.num_values,
|
|
319
|
+
compressed_size: dictionary_page.compressed_size, uncompressed_size: dictionary_page.uncompressed_size,
|
|
320
|
+
is_sorted: dictionary_page.is_sorted},
|
|
235
321
|
statistics: statistics&.to_h,
|
|
236
322
|
size_statistics: size_statistics,
|
|
237
323
|
key_value_metadata: key_value_metadata.empty? ? nil : key_value_metadata,
|
|
238
|
-
bloom_filter: bloom_filter_offset && {
|
|
324
|
+
bloom_filter: bloom_filter_offset && {offset: bloom_filter_offset, length: bloom_filter_length},
|
|
239
325
|
column_index_offset: column_index_range&.first,
|
|
240
326
|
column_index_length: column_index_range&.last,
|
|
241
327
|
offset_index_offset: offset_index_range&.first,
|
|
@@ -251,6 +337,7 @@ module Herringbone
|
|
|
251
337
|
}.compact)
|
|
252
338
|
end
|
|
253
339
|
|
|
340
|
+
# @return [String] short description: path, row group, codec and sizes
|
|
254
341
|
def inspect
|
|
255
342
|
"#<#{self.class.name} #{path} rg=#{@row_group.index} #{codec} #{compressed_size}/#{uncompressed_size} bytes>"
|
|
256
343
|
end
|
|
@@ -258,6 +345,10 @@ module Herringbone
|
|
|
258
345
|
private
|
|
259
346
|
|
|
260
347
|
# Page row counts come from the offset index when there is one (v1 pages don't carry them)
|
|
348
|
+
# Sets +first_row_index+ on the data pages the index points at, and +num_rows+ where the
|
|
349
|
+
# page header did not provide it.
|
|
350
|
+
# @param list [Array<PageInfo>] pages of this chunk, updated in place
|
|
351
|
+
# @return [void]
|
|
261
352
|
def apply_offset_index(list)
|
|
262
353
|
oi = offset_index or return
|
|
263
354
|
by_offset = list.select(&:data?).to_h { |p| [p.offset, p] }
|
|
@@ -275,6 +366,11 @@ module Herringbone
|
|
|
275
366
|
class RowGroupInfo
|
|
276
367
|
attr_reader :index, :row_group, :columns
|
|
277
368
|
|
|
369
|
+
# @param inspector [Inspector] owner, passed on to the column chunks
|
|
370
|
+
# @param index [Integer] position of the row group in the file
|
|
371
|
+
# @param row_group [Format::RowGroup] row group as decoded from the footer
|
|
372
|
+
# @param first_row [Integer] file-wide index of the row group's first row
|
|
373
|
+
# @raise [FormatError] when the row group has more column chunks than the schema has leaves
|
|
278
374
|
def initialize(inspector, index, row_group, first_row)
|
|
279
375
|
@index = index
|
|
280
376
|
@row_group = row_group
|
|
@@ -286,21 +382,41 @@ module Herringbone
|
|
|
286
382
|
end
|
|
287
383
|
end
|
|
288
384
|
|
|
385
|
+
# @return [Integer] rows in the row group
|
|
289
386
|
def num_rows = @row_group.num_rows
|
|
290
|
-
|
|
387
|
+
# @return [Integer] file-wide index of the row group's first row
|
|
388
|
+
attr_reader :first_row
|
|
389
|
+
|
|
390
|
+
# @return [Integer] total_byte_size from the footer (uncompressed size of all column data)
|
|
291
391
|
def total_byte_size = @row_group.total_byte_size
|
|
392
|
+
|
|
393
|
+
# @return [Integer] total_compressed_size from the footer, else the sum over the chunks
|
|
292
394
|
def compressed_size = @row_group.total_compressed_size || @columns.sum(&:compressed_size)
|
|
395
|
+
|
|
396
|
+
# @return [Integer] sum of the chunks' total_uncompressed_size
|
|
293
397
|
def uncompressed_size = @columns.sum { |c| c.uncompressed_size.to_i }
|
|
398
|
+
|
|
399
|
+
# @return [Integer, nil] lowest start offset of the row group's chunks; nil without chunks
|
|
294
400
|
def start_offset = @columns.map(&:start_offset).min
|
|
401
|
+
|
|
402
|
+
# @return [Integer, nil] highest end offset of the row group's chunks; nil without chunks
|
|
295
403
|
def end_offset = @columns.map(&:end_offset).max
|
|
404
|
+
|
|
405
|
+
# @param path [String, Array<String>] dotted path (+"a.b"+) or path segments (+["a", "b"]+)
|
|
406
|
+
# @return [ColumnChunkInfo, nil] the chunk of that column, nil when there is none
|
|
296
407
|
def column(path) = @columns.find { |c| c.path == path.to_s || c.column.path == Array(path) }
|
|
297
408
|
|
|
409
|
+
# The row group's declared sort order
|
|
410
|
+
# @return [Array<Hash{Symbol => Object}>] each { column:, descending:, nulls_first: }, where
|
|
411
|
+
# +column+ is the column's path (its index when out of range)
|
|
298
412
|
def sorting_columns
|
|
299
413
|
(@row_group.sorting_columns || []).map do |s|
|
|
300
|
-
{
|
|
414
|
+
{column: @columns[s.column_idx]&.path || s.column_idx, descending: s.descending, nulls_first: s.nulls_first}
|
|
301
415
|
end
|
|
302
416
|
end
|
|
303
417
|
|
|
418
|
+
# Everything known about the row group, with each column chunk's #to_h
|
|
419
|
+
# @return [Hash{Symbol => Object}] JSON-safe; keys without a value are left out
|
|
304
420
|
def to_h
|
|
305
421
|
Inspector.jsonable({
|
|
306
422
|
index: index,
|
|
@@ -318,6 +434,7 @@ module Herringbone
|
|
|
318
434
|
}.compact)
|
|
319
435
|
end
|
|
320
436
|
|
|
437
|
+
# @return [String] short description: index, row count and number of columns
|
|
321
438
|
def inspect
|
|
322
439
|
"#<#{self.class.name} #{index} rows=#{num_rows} columns=#{@columns.size}>"
|
|
323
440
|
end
|
|
@@ -330,23 +447,37 @@ module Herringbone
|
|
|
330
447
|
|
|
331
448
|
# +io+ is a random-access IO (responds to #seek and #read, e.g. File.open(path, "rb")). It is
|
|
332
449
|
# left open.
|
|
450
|
+
# @param io [IO] random-access IO positioned anywhere; only the footer is read here
|
|
451
|
+
# @raise [ArgumentError] when +io+ does not support #seek and #read
|
|
452
|
+
# @raise [FormatError] when the file is too small, lacks the magic bytes or has a corrupt footer
|
|
453
|
+
# @raise [UnsupportedError] when the file is encrypted (PARE magic)
|
|
333
454
|
def initialize(io)
|
|
334
455
|
@io = io
|
|
335
456
|
unless @io.respond_to?(:read) && @io.respond_to?(:seek)
|
|
336
457
|
raise ArgumentError, "Herringbone::Inspector expects an IO that supports #seek and #read " \
|
|
337
458
|
"(e.g. File.open(path, \"rb\")), got #{io.class}"
|
|
338
459
|
end
|
|
339
|
-
@name = @io.respond_to?(:path) && @io.path ? File.basename(@io.path.to_s) : nil
|
|
460
|
+
@name = (@io.respond_to?(:path) && @io.path) ? File.basename(@io.path.to_s) : nil
|
|
340
461
|
read_footer
|
|
341
462
|
@schema = Schema.from_elements(@metadata.schema)
|
|
342
463
|
end
|
|
343
464
|
|
|
465
|
+
# @return [Integer] row count declared in the footer
|
|
344
466
|
def num_rows = @metadata.num_rows
|
|
467
|
+
|
|
468
|
+
# @return [String, nil] the writer's created_by string, e.g. +"parquet-cpp-arrow version 15.0.0"+
|
|
345
469
|
def created_by = @metadata.created_by
|
|
470
|
+
|
|
471
|
+
# @return [Integer] format version from the footer (1 or 2; says little about the features used)
|
|
346
472
|
def version = @metadata.version
|
|
473
|
+
|
|
474
|
+
# @return [Array<Schema::Column>] the schema's leaf columns
|
|
347
475
|
def columns = @schema.columns
|
|
476
|
+
|
|
477
|
+
# @return [Integer] file offset where the Thrift-encoded FileMetaData starts
|
|
348
478
|
def footer_offset = @file_size - 8 - @footer_size
|
|
349
479
|
|
|
480
|
+
# @return [Array<RowGroupInfo>] the row groups, in footer order (built on first use)
|
|
350
481
|
def row_groups
|
|
351
482
|
@row_groups ||= begin
|
|
352
483
|
first = 0
|
|
@@ -358,9 +489,11 @@ module Herringbone
|
|
|
358
489
|
end
|
|
359
490
|
end
|
|
360
491
|
|
|
492
|
+
# @return [Array<ColumnChunkInfo>] the column chunks of all row groups, row group by row group
|
|
361
493
|
def column_chunks = row_groups.flat_map(&:columns)
|
|
362
494
|
|
|
363
495
|
# Walks every page header and page index now (e.g. before closing the file)
|
|
496
|
+
# @return [Inspector] self
|
|
364
497
|
def load_all
|
|
365
498
|
column_chunks.each do |c|
|
|
366
499
|
c.pages
|
|
@@ -371,18 +504,32 @@ module Herringbone
|
|
|
371
504
|
self
|
|
372
505
|
end
|
|
373
506
|
|
|
507
|
+
# @return [Boolean] whether any column chunk has a ColumnIndex or an OffsetIndex
|
|
374
508
|
def page_index? = column_chunks.any? { |c| c.column_index_range || c.offset_index_range }
|
|
509
|
+
|
|
510
|
+
# @return [Boolean] whether any column chunk has a bloom filter
|
|
375
511
|
def bloom_filters? = column_chunks.any?(&:bloom_filter_offset)
|
|
376
512
|
|
|
513
|
+
# The file-level key/value metadata, each entry described: ARROW:schema is decoded, JSON
|
|
514
|
+
# values are parsed (pandas metadata summarized), binary values are shown as hex
|
|
515
|
+
# @return [Array<Hash{Symbol => Object}>] each with :key, :bytesize, :format ("arrow_schema",
|
|
516
|
+
# "json", "text" or "binary"), :value (truncated) and, depending on the format, :summary,
|
|
517
|
+
# :json, :arrow_schema, :arrow_fields, :arrow_error
|
|
377
518
|
def key_value_metadata
|
|
378
519
|
(@metadata.key_value_metadata || []).map { |kv| describe_key_value(kv.key, kv.value) }
|
|
379
520
|
end
|
|
380
521
|
|
|
522
|
+
# @return [Array<String>, nil] "TYPE_DEFINED_ORDER" or "UNKNOWN" per leaf column; nil when the
|
|
523
|
+
# footer has no column_orders
|
|
381
524
|
def column_orders
|
|
382
525
|
orders = @metadata.column_orders or return nil
|
|
383
526
|
orders.map { |o| o.type_order ? "TYPE_DEFINED_ORDER" : "UNKNOWN" }
|
|
384
527
|
end
|
|
385
528
|
|
|
529
|
+
# File-wide facts: sizes, row and column counts, codecs, whether page indexes and bloom
|
|
530
|
+
# filters are present, and (after #verify_checksums) the CRC tallies under :checksums.
|
|
531
|
+
# Walks no page headers.
|
|
532
|
+
# @return [Hash{Symbol => Object}]
|
|
386
533
|
def summary
|
|
387
534
|
codecs = column_chunks.map(&:codec).uniq
|
|
388
535
|
{
|
|
@@ -408,15 +555,20 @@ module Herringbone
|
|
|
408
555
|
# the page data as stored: compressed, and for v2 pages the levels plus the compressed
|
|
409
556
|
# values). Nothing is decompressed. Sets PageInfo#checksum on every page and returns
|
|
410
557
|
# #checksum_summary. Needs the IO, so call it before closing the file.
|
|
558
|
+
# @return [Hash{Symbol => Object}] see #checksum_summary
|
|
411
559
|
def verify_checksums
|
|
412
560
|
column_chunks.each(&:verify_checksums)
|
|
413
561
|
@checksums_verified = true
|
|
414
562
|
checksum_summary
|
|
415
563
|
end
|
|
416
564
|
|
|
565
|
+
# @return [Boolean] whether #verify_checksums has run
|
|
417
566
|
def checksums_verified? = @checksums_verified == true
|
|
418
567
|
|
|
419
568
|
# After #verify_checksums: { ok:, mismatch:, absent:, mismatches: [{ row_group:, column:, page:, type:, offset:, crc:, actual: }] }
|
|
569
|
+
# The counts are pages per status; in each mismatch +crc+ is the CRC from the page header and
|
|
570
|
+
# +actual+ the CRC32 of the stored bytes (kept from the verification, so the IO is not needed).
|
|
571
|
+
# @return [Hash{Symbol => Object}, nil] nil before #verify_checksums
|
|
420
572
|
def checksum_summary
|
|
421
573
|
return nil unless checksums_verified?
|
|
422
574
|
all = column_chunks.flat_map { |c| c.pages.map { |p| [c, p] } }
|
|
@@ -424,44 +576,50 @@ module Herringbone
|
|
|
424
576
|
{
|
|
425
577
|
ok: tally.fetch(:ok, 0), mismatch: tally.fetch(:mismatch, 0), absent: tally.fetch(:absent, 0),
|
|
426
578
|
mismatches: all.select { |_, p| p.checksum == :mismatch }.map do |c, p|
|
|
427
|
-
{
|
|
428
|
-
|
|
579
|
+
{row_group: c.row_group.index, column: c.path, page: p.index, type: p.type, offset: p.offset,
|
|
580
|
+
crc: p.expected_crc, actual: p.actual_crc}
|
|
429
581
|
end
|
|
430
582
|
}
|
|
431
583
|
end
|
|
432
584
|
|
|
433
585
|
# Every disagreement between page header statistics and the ColumnIndex, across the file,
|
|
434
586
|
# each with :row_group and :column added (see ColumnChunkInfo#index_mismatches)
|
|
587
|
+
# @return [Array<Hash{Symbol => Object}>]
|
|
435
588
|
def index_mismatches
|
|
436
589
|
column_chunks.flat_map do |c|
|
|
437
|
-
c.index_mismatches.map { |m| {
|
|
590
|
+
c.index_mismatches.map { |m| {row_group: c.row_group.index, column: c.path}.merge(m) }
|
|
438
591
|
end
|
|
439
592
|
end
|
|
440
593
|
|
|
441
594
|
# The decoded ARROW:schema key/value (see ArrowSchema.decode); nil when the file has none or
|
|
442
595
|
# it could not be decoded, and #arrow_schema_error then says why
|
|
596
|
+
# @return [Hash{Symbol => Object}, nil] see ArrowSchema.decode
|
|
443
597
|
def arrow_schema
|
|
444
598
|
return @arrow_schema if defined?(@arrow_schema)
|
|
445
599
|
@arrow_schema_error = nil
|
|
446
600
|
kv = (@metadata.key_value_metadata || []).find { |x| x.key == "ARROW:schema" }
|
|
447
601
|
@arrow_schema = kv && begin
|
|
448
602
|
ArrowSchema.decode(kv.value)
|
|
449
|
-
rescue
|
|
603
|
+
rescue => e
|
|
450
604
|
@arrow_schema_error = "could not decode ARROW:schema: #{e.message}"
|
|
451
605
|
nil
|
|
452
606
|
end
|
|
453
607
|
end
|
|
454
608
|
|
|
609
|
+
# @return [String, nil] why the ARROW:schema value could not be decoded; nil when it decoded
|
|
610
|
+
# fine or the file has none
|
|
455
611
|
def arrow_schema_error
|
|
456
612
|
arrow_schema
|
|
457
613
|
@arrow_schema_error
|
|
458
614
|
end
|
|
459
615
|
|
|
460
616
|
# Schema tree: Hashes with name, repetition, types, levels (leaves) and children (groups)
|
|
617
|
+
# Nodes that match a field of the ARROW:schema also get its type as :arrow_type.
|
|
618
|
+
# @return [Array<Hash{Symbol => Object}>] the root's children
|
|
461
619
|
def schema_tree
|
|
462
620
|
leaf_by_node = @schema.columns.to_h { |c| [c.node, c] }
|
|
463
621
|
build = lambda do |node|
|
|
464
|
-
h = {
|
|
622
|
+
h = {name: node.name, repetition: node.repetition}
|
|
465
623
|
if node.leaf?
|
|
466
624
|
col = leaf_by_node[node]
|
|
467
625
|
h[:physical_type] = T::NAMES[node.type]&.to_s
|
|
@@ -483,6 +641,8 @@ module Herringbone
|
|
|
483
641
|
end
|
|
484
642
|
|
|
485
643
|
# Per leaf column: sums over all row groups, plus overall min/max where comparable
|
|
644
|
+
# Walks every page header (for the page counts).
|
|
645
|
+
# @return [Array<Hash{Symbol => Object}>] one per leaf column, in schema order
|
|
486
646
|
def column_totals
|
|
487
647
|
columns.map do |col|
|
|
488
648
|
chunks = row_groups.map { |rg| rg.columns[col.index] }.compact
|
|
@@ -507,8 +667,8 @@ module Herringbone
|
|
|
507
667
|
num_data_pages: chunks.sum { |c| c.data_pages.size },
|
|
508
668
|
dictionary_pages: chunks.count(&:dictionary_page),
|
|
509
669
|
dictionary_bytes: chunks.sum { |c| c.dictionary_page&.total_size.to_i },
|
|
510
|
-
min: mins.
|
|
511
|
-
max: maxes.
|
|
670
|
+
min: mins.include?(nil) ? nil : safe_extreme(mins, :min),
|
|
671
|
+
max: maxes.include?(nil) ? nil : safe_extreme(maxes, :max)
|
|
512
672
|
}.compact
|
|
513
673
|
end
|
|
514
674
|
end
|
|
@@ -516,33 +676,36 @@ module Herringbone
|
|
|
516
676
|
# Byte ranges of the whole file, in offset order: magic, pages (or whole chunks when their pages
|
|
517
677
|
# could not be walked), bloom filters, page indexes, footer. Gaps right after a chunk at its
|
|
518
678
|
# file_offset are inline :column_metadata copies; other gaps are reported as :unknown. Each entry: { kind:, start:, length:, row_group:, column:, page: }
|
|
679
|
+
# +row_group+, +column+ and +page+ are indexes, present only where they apply. Segments can
|
|
680
|
+
# overlap when the file is damaged; gaps are only computed past the furthest end seen so far.
|
|
681
|
+
# @return [Array<Hash{Symbol => Object}>]
|
|
519
682
|
def layout
|
|
520
|
-
segs = [{
|
|
683
|
+
segs = [{kind: :magic, start: 0, length: 4}]
|
|
521
684
|
column_chunks.each do |c|
|
|
522
685
|
rg = c.row_group.index
|
|
523
686
|
col = c.column.index
|
|
524
687
|
if c.pages.empty?
|
|
525
|
-
segs << {
|
|
688
|
+
segs << {kind: :chunk, start: c.start_offset, length: c.compressed_size, row_group: rg, column: col}
|
|
526
689
|
else
|
|
527
690
|
c.pages.each_with_index do |p, i|
|
|
528
|
-
segs << {
|
|
529
|
-
row_group: rg, column: col, page: i
|
|
691
|
+
segs << {kind: p.dictionary? ? :dictionary_page : :data_page, start: p.offset, length: p.total_size,
|
|
692
|
+
row_group: rg, column: col, page: i}
|
|
530
693
|
end
|
|
531
694
|
end
|
|
532
695
|
if c.bloom_filter_offset
|
|
533
|
-
segs << {
|
|
534
|
-
row_group: rg, column: col
|
|
696
|
+
segs << {kind: :bloom_filter, start: c.bloom_filter_offset, length: c.bloom_filter_length.to_i,
|
|
697
|
+
row_group: rg, column: col}
|
|
535
698
|
end
|
|
536
699
|
if (r = c.column_index_range)
|
|
537
|
-
segs << {
|
|
700
|
+
segs << {kind: :column_index, start: r[0], length: r[1], row_group: rg, column: col}
|
|
538
701
|
end
|
|
539
702
|
if (r = c.offset_index_range)
|
|
540
|
-
segs << {
|
|
703
|
+
segs << {kind: :offset_index, start: r[0], length: r[1], row_group: rg, column: col}
|
|
541
704
|
end
|
|
542
705
|
end
|
|
543
|
-
segs << {
|
|
544
|
-
segs << {
|
|
545
|
-
segs << {
|
|
706
|
+
segs << {kind: :footer, start: footer_offset, length: @footer_size}
|
|
707
|
+
segs << {kind: :footer_length, start: @file_size - 8, length: 4}
|
|
708
|
+
segs << {kind: :magic, start: @file_size - 4, length: 4}
|
|
546
709
|
segs.sort_by! { |s| s[:start] }
|
|
547
710
|
# Some writers (old parquet-rs, parquet-mr for a while) store a copy of the ColumnMetaData
|
|
548
711
|
# right after the chunk, at ColumnChunk.file_offset
|
|
@@ -556,9 +719,9 @@ module Herringbone
|
|
|
556
719
|
if s[:start] > pos
|
|
557
720
|
c = meta_copies[pos]
|
|
558
721
|
out << if c
|
|
559
|
-
{
|
|
722
|
+
{kind: :column_metadata, start: pos, length: s[:start] - pos, row_group: c.row_group.index, column: c.column.index}
|
|
560
723
|
else
|
|
561
|
-
{
|
|
724
|
+
{kind: :unknown, start: pos, length: s[:start] - pos}
|
|
562
725
|
end
|
|
563
726
|
end
|
|
564
727
|
out << s
|
|
@@ -567,6 +730,9 @@ module Herringbone
|
|
|
567
730
|
out
|
|
568
731
|
end
|
|
569
732
|
|
|
733
|
+
# Everything: summary, key/value metadata, schema tree, row groups with their chunks and
|
|
734
|
+
# pages, column totals, and any checksum and page index mismatches. Walks every page header.
|
|
735
|
+
# @return [Hash{Symbol => Object}] JSON-safe; keys without a value are left out
|
|
570
736
|
def to_h
|
|
571
737
|
Inspector.jsonable({
|
|
572
738
|
summary: summary,
|
|
@@ -579,12 +745,17 @@ module Herringbone
|
|
|
579
745
|
}.compact)
|
|
580
746
|
end
|
|
581
747
|
|
|
748
|
+
# @param args [Array] passed on to Hash#to_json (e.g. a JSON::State)
|
|
749
|
+
# @return [String] #to_h as JSON
|
|
582
750
|
def to_json(*args) = to_h.to_json(*args)
|
|
583
751
|
|
|
584
752
|
# A self-contained HTML page showing the file's layout, see Visualizer
|
|
753
|
+
# @return [String] HTML document
|
|
585
754
|
def to_html = Visualizer.new(self).to_html
|
|
586
755
|
|
|
587
756
|
# Readable text summary. With pages: true, lists every page header too.
|
|
757
|
+
# @param pages [Boolean] whether to add a line per page header under each column chunk
|
|
758
|
+
# @return [String] multi-line text, as printed by +bin/herringbone inspect+
|
|
588
759
|
def report(pages: false)
|
|
589
760
|
s = summary
|
|
590
761
|
out = []
|
|
@@ -604,7 +775,7 @@ module Herringbone
|
|
|
604
775
|
end
|
|
605
776
|
mismatches = index_mismatches
|
|
606
777
|
unless mismatches.empty?
|
|
607
|
-
out << "page statistics vs column index: #{mismatches.size} disagreement#{mismatches.size == 1
|
|
778
|
+
out << "page statistics vs column index: #{mismatches.size} disagreement#{"s" unless mismatches.size == 1}"
|
|
608
779
|
mismatches.first(50).each { |m| out << " #{index_mismatch_text(m)}" }
|
|
609
780
|
out << " ..." if mismatches.size > 50
|
|
610
781
|
end
|
|
@@ -623,7 +794,7 @@ module Herringbone
|
|
|
623
794
|
ann = n[:logical_type] || n[:converted_type]
|
|
624
795
|
levels = n[:children] ? "" : " [def #{n[:max_definition_level]}, rep #{n[:max_repetition_level]}]"
|
|
625
796
|
arrow = n[:arrow_type] ? " arrow: #{n[:arrow_type]}" : ""
|
|
626
|
-
out << "#{" " * depth}#{n[:repetition]} #{type} #{n[:name]}#{
|
|
797
|
+
out << "#{" " * depth}#{n[:repetition]} #{type} #{n[:name]}#{" (#{ann})" if ann}#{levels}#{arrow}"
|
|
627
798
|
(n[:children] || []).each { |c| walk.call(c, depth + 1) }
|
|
628
799
|
end
|
|
629
800
|
schema_tree.each { |n| walk.call(n, 1) }
|
|
@@ -633,15 +804,15 @@ module Herringbone
|
|
|
633
804
|
out << " #{t[:path]}: #{t[:type]} #{t[:codecs].join(",")} #{t[:encodings].join(",")} " \
|
|
634
805
|
"#{Inspector.human_bytes(t[:compressed_size])}/#{Inspector.human_bytes(t[:uncompressed_size])}" \
|
|
635
806
|
"#{ratio_text(t[:uncompressed_size], t[:compressed_size])}, #{t[:num_values]} values" \
|
|
636
|
-
"#{
|
|
807
|
+
"#{", #{t[:null_count]} nulls" if t[:null_count]}, #{t[:num_data_pages]} data pages#{range}"
|
|
637
808
|
end
|
|
638
809
|
row_groups.each do |rg|
|
|
639
|
-
sorting = rg.sorting_columns.map { |c| "#{c[:column]}#{
|
|
810
|
+
sorting = rg.sorting_columns.map { |c| "#{c[:column]}#{" desc" if c[:descending]}" }
|
|
640
811
|
out << "row group #{rg.index}: #{rg.num_rows} rows, #{Inspector.human_bytes(rg.compressed_size)} " \
|
|
641
|
-
"at #{rg.start_offset}..#{rg.end_offset}#{
|
|
812
|
+
"at #{rg.start_offset}..#{rg.end_offset}#{", sorted by #{sorting.join(", ")}" unless sorting.empty?}"
|
|
642
813
|
rg.columns.each do |c|
|
|
643
814
|
st = c.statistics
|
|
644
|
-
range = st && (st.min
|
|
815
|
+
range = (st && !(st.min.nil? && st.max.nil?)) ? " [#{Inspector.display(st.min)} .. #{Inspector.display(st.max)}]" : ""
|
|
645
816
|
extras = []
|
|
646
817
|
extras << "dict #{c.dictionary_size} entries" if c.dictionary_page
|
|
647
818
|
extras << "column index" if c.column_index_range
|
|
@@ -650,20 +821,21 @@ module Herringbone
|
|
|
650
821
|
extras << "ERROR: #{c.error}" if c.error
|
|
651
822
|
out << " #{c.path}: #{c.codec} #{c.encodings.join(",")} " \
|
|
652
823
|
"#{c.compressed_size}/#{c.uncompressed_size} bytes#{ratio_text(c.uncompressed_size, c.compressed_size)}, " \
|
|
653
|
-
"#{c.num_values} values, #{c.pages.size} pages#{
|
|
824
|
+
"#{c.num_values} values, #{c.pages.size} pages#{", #{extras.join(", ")}" unless extras.empty?}#{range}"
|
|
654
825
|
next unless pages
|
|
655
826
|
c.pages.each do |p|
|
|
656
827
|
st = p.statistics
|
|
657
828
|
out << " #{p.index}: #{p.type} @#{p.offset} header #{p.header_size} + #{p.compressed_size}/#{p.uncompressed_size} bytes, " \
|
|
658
|
-
"#{p.num_values} values#{
|
|
659
|
-
"#{
|
|
660
|
-
"#{
|
|
829
|
+
"#{p.num_values} values#{", #{p.num_nulls} nulls" if p.num_nulls}#{", #{p.num_rows} rows" if p.num_rows}" \
|
|
830
|
+
"#{" #{p.encoding}" if p.encoding}#{crc_text(p)}" \
|
|
831
|
+
"#{" [#{Inspector.display(st.min)} .. #{Inspector.display(st.max)}]" if st && !(st.min.nil? && st.max.nil?)}"
|
|
661
832
|
end
|
|
662
833
|
end
|
|
663
834
|
end
|
|
664
835
|
out.join("\n")
|
|
665
836
|
end
|
|
666
837
|
|
|
838
|
+
# @return [String] short description: name, rows, row group count and file size
|
|
667
839
|
def inspect
|
|
668
840
|
"#<#{self.class.name} #{@name || "(IO)"} rows=#{num_rows} row_groups=#{row_groups.size} size=#{@file_size}>"
|
|
669
841
|
end
|
|
@@ -672,6 +844,9 @@ module Herringbone
|
|
|
672
844
|
|
|
673
845
|
# Walks page headers from the chunk's first page. Returns [pages, error_message_or_nil].
|
|
674
846
|
# Mirrors the reader's tolerance: a chunk may extend past its declared total_compressed_size.
|
|
847
|
+
# @param chunk [ColumnChunkInfo] chunk whose pages to walk
|
|
848
|
+
# @return [Array(Array<PageInfo>, String), Array(Array<PageInfo>, nil)] the pages found, and why
|
|
849
|
+
# the walk stopped early (nil when every value was accounted for)
|
|
675
850
|
def walk_pages(chunk)
|
|
676
851
|
return [[], "column chunk stored in external file #{chunk.external_file}"] if chunk.external_file
|
|
677
852
|
pages = []
|
|
@@ -699,9 +874,12 @@ module Herringbone
|
|
|
699
874
|
seen += page.num_values.to_i if page.data?
|
|
700
875
|
pos = page.end_offset
|
|
701
876
|
end
|
|
702
|
-
[pages, seen < total ? "found #{seen} of #{total} values in page headers" : nil]
|
|
877
|
+
[pages, (seen < total) ? "found #{seen} of #{total} values in page headers" : nil]
|
|
703
878
|
end
|
|
704
879
|
|
|
880
|
+
# Reads and decodes a chunk's ColumnIndex, decoding the per-page min/max with the column type
|
|
881
|
+
# @param chunk [ColumnChunkInfo] chunk whose ColumnIndex to read
|
|
882
|
+
# @return [ColumnIndexInfo, nil] nil when the chunk has none or it fails to decode
|
|
705
883
|
def read_column_index(chunk)
|
|
706
884
|
offset, length = chunk.column_index_range
|
|
707
885
|
return nil unless offset && length.positive?
|
|
@@ -724,6 +902,9 @@ module Herringbone
|
|
|
724
902
|
nil
|
|
725
903
|
end
|
|
726
904
|
|
|
905
|
+
# Reads and decodes a chunk's OffsetIndex
|
|
906
|
+
# @param chunk [ColumnChunkInfo] chunk whose OffsetIndex to read
|
|
907
|
+
# @return [OffsetIndexInfo, nil] nil when the chunk has none or it fails to decode
|
|
727
908
|
def read_offset_index(chunk)
|
|
728
909
|
offset, length = chunk.offset_index_range
|
|
729
910
|
return nil unless offset && length.positive?
|
|
@@ -740,6 +921,8 @@ module Herringbone
|
|
|
740
921
|
end
|
|
741
922
|
|
|
742
923
|
# Size in bytes of the bloom filter at +offset+ (its Thrift header plus the bitset)
|
|
924
|
+
# @param offset [Integer] file offset of the BloomFilterHeader
|
|
925
|
+
# @return [Integer, nil] nil when the header can't be decoded or has no num_bytes
|
|
743
926
|
def bloom_filter_size(offset)
|
|
744
927
|
buf = read_at(offset, 64)
|
|
745
928
|
header, size = BloomFilterHeader.decode(buf)
|
|
@@ -748,28 +931,36 @@ module Herringbone
|
|
|
748
931
|
nil
|
|
749
932
|
end
|
|
750
933
|
|
|
751
|
-
# :ok, :mismatch or :absent for one page (reads its body)
|
|
934
|
+
# :ok, :mismatch or :absent for one page (reads its body). Sets PageInfo#actual_crc.
|
|
935
|
+
# @param page [PageInfo] page to check
|
|
936
|
+
# @return [Symbol]
|
|
752
937
|
def page_checksum(page)
|
|
753
938
|
return :absent unless page.crc
|
|
754
|
-
page_crc(page)
|
|
939
|
+
page.actual_crc = page_crc(page)
|
|
940
|
+
(page.actual_crc == page.expected_crc) ? :ok : :mismatch
|
|
755
941
|
end
|
|
756
942
|
|
|
757
943
|
# CRC32 of a page's body as stored
|
|
944
|
+
# @param page [PageInfo] page whose compressed body to read
|
|
945
|
+
# @return [Integer] unsigned 32-bit CRC
|
|
758
946
|
def page_crc(page)
|
|
759
947
|
Zlib.crc32(read_at(page.body_offset, page.compressed_size))
|
|
760
948
|
end
|
|
761
949
|
|
|
762
950
|
# See ColumnChunkInfo#index_mismatches
|
|
951
|
+
# @param chunk [ColumnChunkInfo] chunk whose page headers to compare with its ColumnIndex
|
|
952
|
+
# @return [Array<Hash{Symbol => Object}>] one entry per disagreement; a single :page_count
|
|
953
|
+
# entry when the ColumnIndex and the data pages differ in number
|
|
763
954
|
def compare_page_index(chunk)
|
|
764
955
|
ci = chunk.column_index or return []
|
|
765
956
|
data = chunk.data_pages
|
|
766
957
|
if ci.null_pages.size != data.size
|
|
767
|
-
return [{
|
|
958
|
+
return [{page: nil, data_page: nil, field: :page_count, page_value: data.size, index_value: ci.null_pages.size}]
|
|
768
959
|
end
|
|
769
960
|
order = Inspector.sort_order(chunk.column)
|
|
770
961
|
out = []
|
|
771
962
|
data.each_with_index do |p, k|
|
|
772
|
-
report = ->(field, pv, iv) { out << {
|
|
963
|
+
report = ->(field, pv, iv) { out << {page: p.index, data_page: k, field: field, page_value: pv, index_value: iv} }
|
|
773
964
|
idx_nulls = ci.null_counts&.[](k)
|
|
774
965
|
report.call(:null_count, p.num_nulls, idx_nulls) if idx_nulls && p.num_nulls && idx_nulls != p.num_nulls
|
|
775
966
|
st = p.statistics
|
|
@@ -788,6 +979,11 @@ module Herringbone
|
|
|
788
979
|
end
|
|
789
980
|
|
|
790
981
|
# Decodes a Format::Statistics into Ruby values via the column's type converter
|
|
982
|
+
# Prefers min_value/max_value; falls back to the deprecated min/max, with a caveat when their
|
|
983
|
+
# ordering can't be trusted for the column's type.
|
|
984
|
+
# @param st [Format::Statistics, nil] statistics from column metadata or a page header
|
|
985
|
+
# @param column [Schema::Column] column the statistics describe
|
|
986
|
+
# @return [Stats, nil] nil when +st+ is nil
|
|
791
987
|
def decode_statistics(st, column)
|
|
792
988
|
return nil unless st
|
|
793
989
|
order = Inspector.sort_order(column)
|
|
@@ -821,6 +1017,12 @@ module Herringbone
|
|
|
821
1017
|
end
|
|
822
1018
|
|
|
823
1019
|
# Decodes one PLAIN-encoded statistics value (no length prefix for byte arrays)
|
|
1020
|
+
# INT96 decodes to [nanoseconds, julian_day] before conversion. Values of the wrong width,
|
|
1021
|
+
# or that fail to convert, come back as hex (see Inspector.hex).
|
|
1022
|
+
# @param bytes [String, nil] encoded value
|
|
1023
|
+
# @param column [Schema::Column] column whose physical type and converter apply
|
|
1024
|
+
# @return [Object, nil] the converted value, a hex String when it could not be decoded, nil
|
|
1025
|
+
# for nil (or empty BOOLEAN) input
|
|
824
1026
|
def decode_value(bytes, column)
|
|
825
1027
|
return nil if bytes.nil?
|
|
826
1028
|
bytes = bytes.b
|
|
@@ -840,7 +1042,7 @@ module Herringbone
|
|
|
840
1042
|
end
|
|
841
1043
|
conv = column.converter
|
|
842
1044
|
conv ? conv.call(raw) : raw
|
|
843
|
-
rescue
|
|
1045
|
+
rescue
|
|
844
1046
|
Inspector.hex(bytes)
|
|
845
1047
|
end
|
|
846
1048
|
|
|
@@ -854,43 +1056,84 @@ module Herringbone
|
|
|
854
1056
|
# Fields carry :name, :type, :nullable and, when present, :children, :dictionary
|
|
855
1057
|
# ({ index_type:, ordered:, id: }), :extension (ARROW:extension:name) and :metadata.
|
|
856
1058
|
module ArrowSchema
|
|
1059
|
+
# Raised for malformed or unsupported ARROW:schema values
|
|
857
1060
|
class Error < StandardError; end
|
|
858
1061
|
|
|
1062
|
+
# Deepest field nesting accepted, against stack exhaustion on hostile input
|
|
859
1063
|
MAX_DEPTH = 64
|
|
1064
|
+
# Most fields (nested ones included) accepted in one schema
|
|
860
1065
|
MAX_FIELDS = 100_000
|
|
1066
|
+
# Arrow TimeUnit values (SECOND, MILLISECOND, MICROSECOND, NANOSECOND) as pyarrow abbreviates them
|
|
861
1067
|
TIME_UNITS = %w[s ms us ns].freeze
|
|
1068
|
+
# MessageHeader union tag of a Schema in Message.fbs
|
|
862
1069
|
MESSAGE_SCHEMA = 1
|
|
863
1070
|
|
|
864
1071
|
# A minimal flatbuffer reader: tables (through their vtables), scalars, strings, vectors of
|
|
865
1072
|
# scalars and tables, and unions. Every read is bounds-checked; malformed input raises Error.
|
|
866
1073
|
class FlatBuffer
|
|
1074
|
+
# @param bytes [String] the flatbuffer (copied as binary)
|
|
867
1075
|
def initialize(bytes)
|
|
868
1076
|
@b = bytes.b
|
|
869
1077
|
end
|
|
870
1078
|
|
|
1079
|
+
# @return [Table] the root table, whose offset is stored in the first 4 bytes
|
|
871
1080
|
def root = table_at(u32(0))
|
|
1081
|
+
|
|
1082
|
+
# @param pos [Integer] absolute position of a table
|
|
1083
|
+
# @return [Table]
|
|
1084
|
+
# @raise [Error] when its vtable is out of bounds or malformed
|
|
872
1085
|
def table_at(pos) = Table.new(self, pos)
|
|
873
1086
|
|
|
1087
|
+
# @param pos [Integer] absolute start of the read
|
|
1088
|
+
# @param len [Integer] bytes to read
|
|
1089
|
+
# @return [void]
|
|
1090
|
+
# @raise [Error] when the range is not inside the buffer
|
|
874
1091
|
def check(pos, len)
|
|
875
1092
|
return if pos >= 0 && len >= 0 && pos + len <= @b.bytesize
|
|
876
1093
|
raise Error, "flatbuffer read of #{len} bytes at #{pos} is out of bounds (#{@b.bytesize} bytes)"
|
|
877
1094
|
end
|
|
878
1095
|
|
|
1096
|
+
# @param pos [Integer] absolute start of the read
|
|
1097
|
+
# @param len [Integer] bytes to read
|
|
1098
|
+
# @param fmt [String] String#unpack1 directive for the bytes
|
|
1099
|
+
# @return [Integer] the unpacked scalar
|
|
1100
|
+
# @raise [Error] when the range is not inside the buffer
|
|
879
1101
|
def read(pos, len, fmt)
|
|
880
1102
|
check(pos, len)
|
|
881
1103
|
@b.byteslice(pos, len).unpack1(fmt)
|
|
882
1104
|
end
|
|
883
1105
|
|
|
1106
|
+
# @param pos [Integer] absolute position
|
|
1107
|
+
# @return [Integer] unsigned 8-bit value at +pos+
|
|
884
1108
|
def u8(pos) = read(pos, 1, "C")
|
|
1109
|
+
|
|
1110
|
+
# @param pos [Integer] absolute position
|
|
1111
|
+
# @return [Integer] unsigned little-endian 16-bit value at +pos+
|
|
885
1112
|
def u16(pos) = read(pos, 2, "S<")
|
|
1113
|
+
|
|
1114
|
+
# @param pos [Integer] absolute position
|
|
1115
|
+
# @return [Integer] signed little-endian 16-bit value at +pos+
|
|
886
1116
|
def i16(pos) = read(pos, 2, "s<")
|
|
1117
|
+
|
|
1118
|
+
# @param pos [Integer] absolute position
|
|
1119
|
+
# @return [Integer] unsigned little-endian 32-bit value at +pos+
|
|
887
1120
|
def u32(pos) = read(pos, 4, "L<")
|
|
1121
|
+
|
|
1122
|
+
# @param pos [Integer] absolute position
|
|
1123
|
+
# @return [Integer] signed little-endian 32-bit value at +pos+
|
|
888
1124
|
def i32(pos) = read(pos, 4, "l<")
|
|
1125
|
+
|
|
1126
|
+
# @param pos [Integer] absolute position
|
|
1127
|
+
# @return [Integer] signed little-endian 64-bit value at +pos+
|
|
889
1128
|
def i64(pos) = read(pos, 8, "q<")
|
|
890
1129
|
|
|
891
1130
|
# Offsets are relative to where they are stored
|
|
1131
|
+
# @param pos [Integer] absolute position of a uoffset
|
|
1132
|
+
# @return [Integer] absolute position it points to
|
|
892
1133
|
def deref(pos) = pos + u32(pos)
|
|
893
1134
|
|
|
1135
|
+
# @param pos [Integer] absolute position of the string's length prefix
|
|
1136
|
+
# @return [String] the string's bytes as UTF-8 (not validated)
|
|
894
1137
|
def string(pos)
|
|
895
1138
|
len = u32(pos)
|
|
896
1139
|
check(pos + 4, len)
|
|
@@ -898,6 +1141,10 @@ module Herringbone
|
|
|
898
1141
|
end
|
|
899
1142
|
|
|
900
1143
|
# [start, length] of the vector at +pos+ with +size+-byte elements
|
|
1144
|
+
# @param pos [Integer] absolute position of the vector's length prefix
|
|
1145
|
+
# @param size [Integer] bytes per element
|
|
1146
|
+
# @return [Array(Integer, Integer)] start of the first element and the element count
|
|
1147
|
+
# @raise [Error] when the elements do not fit in the buffer
|
|
901
1148
|
def vector(pos, size)
|
|
902
1149
|
len = u32(pos)
|
|
903
1150
|
check(pos + 4, len * size)
|
|
@@ -907,6 +1154,9 @@ module Herringbone
|
|
|
907
1154
|
|
|
908
1155
|
# One flatbuffer table; fields are addressed by their slot (declaration order in the .fbs)
|
|
909
1156
|
class Table
|
|
1157
|
+
# @param fb [FlatBuffer] buffer the table lives in
|
|
1158
|
+
# @param pos [Integer] absolute position of the table (where its vtable offset is stored)
|
|
1159
|
+
# @raise [Error] when the vtable is out of bounds or malformed
|
|
910
1160
|
def initialize(fb, pos)
|
|
911
1161
|
@fb = fb
|
|
912
1162
|
@pos = pos
|
|
@@ -917,6 +1167,8 @@ module Herringbone
|
|
|
917
1167
|
end
|
|
918
1168
|
|
|
919
1169
|
# Absolute position of a field's value, nil when absent
|
|
1170
|
+
# @param slot [Integer] field slot, counting from 0
|
|
1171
|
+
# @return [Integer, nil]
|
|
920
1172
|
def field(slot)
|
|
921
1173
|
o = 4 + (slot * 2)
|
|
922
1174
|
return nil if o + 2 > @vtable_size
|
|
@@ -924,20 +1176,49 @@ module Herringbone
|
|
|
924
1176
|
off.zero? ? nil : @pos + off
|
|
925
1177
|
end
|
|
926
1178
|
|
|
1179
|
+
# @param slot [Integer] field slot, counting from 0
|
|
1180
|
+
# @param default [Integer] returned when the field is absent (the .fbs default)
|
|
1181
|
+
# @return [Integer] the unsigned 8-bit field (also used for union type tags)
|
|
927
1182
|
def u8(slot, default = 0) = (p = field(slot)) ? @fb.u8(p) : default
|
|
1183
|
+
|
|
1184
|
+
# @param slot [Integer] field slot, counting from 0
|
|
1185
|
+
# @param default [Boolean] returned when the field is absent (the .fbs default)
|
|
1186
|
+
# @return [Boolean]
|
|
928
1187
|
def bool(slot, default = false) = (p = field(slot)) ? @fb.u8(p) != 0 : default
|
|
1188
|
+
|
|
1189
|
+
# @param slot [Integer] field slot, counting from 0
|
|
1190
|
+
# @param default [Integer] returned when the field is absent (the .fbs default)
|
|
1191
|
+
# @return [Integer] the signed 16-bit field (also used for enums such as TimeUnit)
|
|
929
1192
|
def i16(slot, default = 0) = (p = field(slot)) ? @fb.i16(p) : default
|
|
1193
|
+
|
|
1194
|
+
# @param slot [Integer] field slot, counting from 0
|
|
1195
|
+
# @param default [Integer] returned when the field is absent (the .fbs default)
|
|
1196
|
+
# @return [Integer] the signed 32-bit field
|
|
930
1197
|
def i32(slot, default = 0) = (p = field(slot)) ? @fb.i32(p) : default
|
|
1198
|
+
|
|
1199
|
+
# @param slot [Integer] field slot, counting from 0
|
|
1200
|
+
# @param default [Integer] returned when the field is absent (the .fbs default)
|
|
1201
|
+
# @return [Integer] the signed 64-bit field
|
|
931
1202
|
def i64(slot, default = 0) = (p = field(slot)) ? @fb.i64(p) : default
|
|
1203
|
+
|
|
1204
|
+
# @param slot [Integer] field slot, counting from 0
|
|
1205
|
+
# @return [String, nil] the string field, nil when absent
|
|
932
1206
|
def string(slot) = (p = field(slot)) && @fb.string(@fb.deref(p))
|
|
1207
|
+
|
|
1208
|
+
# @param slot [Integer] field slot, counting from 0 (also the value of a union)
|
|
1209
|
+
# @return [Table, nil] the sub-table, nil when absent
|
|
933
1210
|
def table(slot) = (p = field(slot)) && @fb.table_at(@fb.deref(p))
|
|
934
1211
|
|
|
1212
|
+
# @param slot [Integer] field slot, counting from 0
|
|
1213
|
+
# @return [Array<Table>] the vector of tables, empty when absent
|
|
935
1214
|
def tables(slot)
|
|
936
1215
|
p = field(slot) or return []
|
|
937
1216
|
start, len = @fb.vector(@fb.deref(p), 4)
|
|
938
1217
|
Array.new(len) { |i| @fb.table_at(@fb.deref(start + (4 * i))) }
|
|
939
1218
|
end
|
|
940
1219
|
|
|
1220
|
+
# @param slot [Integer] field slot, counting from 0
|
|
1221
|
+
# @return [Array<Integer>] the vector of signed 32-bit values, empty when absent
|
|
941
1222
|
def i32s(slot)
|
|
942
1223
|
p = field(slot) or return []
|
|
943
1224
|
start, len = @fb.vector(@fb.deref(p), 4)
|
|
@@ -948,6 +1229,10 @@ module Herringbone
|
|
|
948
1229
|
module_function
|
|
949
1230
|
|
|
950
1231
|
# Decodes the base64 ARROW:schema value; raises ArrowSchema::Error when it can't
|
|
1232
|
+
# @param b64 [String] the key/value metadata value (base64 of an IPC Schema message)
|
|
1233
|
+
# @return [Hash{Symbol => Object}] { endianness:, fields:, metadata: }; +metadata+ is a
|
|
1234
|
+
# Hash{String => String}, left out when the schema has none
|
|
1235
|
+
# @raise [Error] when the value is empty, malformed or not a Schema message
|
|
951
1236
|
def decode(b64)
|
|
952
1237
|
bytes = b64.to_s.unpack1("m")
|
|
953
1238
|
raise Error, "empty value" if bytes.empty?
|
|
@@ -968,6 +1253,9 @@ module Herringbone
|
|
|
968
1253
|
|
|
969
1254
|
# The flatbuffer inside an encapsulated IPC message: [0xFFFFFFFF] int32 length, flatbuffer
|
|
970
1255
|
# (the continuation marker is missing in files from before Arrow 0.15)
|
|
1256
|
+
# @param bytes [String] the decoded (binary) ARROW:schema value
|
|
1257
|
+
# @return [String] the message's flatbuffer bytes
|
|
1258
|
+
# @raise [Error] when the length prefix is missing or does not fit
|
|
971
1259
|
def message_bytes(bytes)
|
|
972
1260
|
raise Error, "too short for an IPC message (#{bytes.bytesize} bytes)" if bytes.bytesize < 8
|
|
973
1261
|
len = bytes.unpack1("l<")
|
|
@@ -980,18 +1268,25 @@ module Herringbone
|
|
|
980
1268
|
bytes.byteslice(start, len)
|
|
981
1269
|
end
|
|
982
1270
|
|
|
1271
|
+
# Describes one Arrow Field table and, recursively, its children
|
|
1272
|
+
# @param t [Table] the Field table
|
|
1273
|
+
# @param depth [Integer] nesting depth of the field, 0 at the top level
|
|
1274
|
+
# @param count [Array<Integer>] one-element counter of the fields seen so far, shared across
|
|
1275
|
+
# the recursion
|
|
1276
|
+
# @return [Hash{Symbol => Object}] see the module description for the keys
|
|
1277
|
+
# @raise [Error] past MAX_DEPTH or MAX_FIELDS, or on malformed input
|
|
983
1278
|
def field(t, depth, count)
|
|
984
1279
|
raise Error, "fields nested deeper than #{MAX_DEPTH} levels" if depth > MAX_DEPTH
|
|
985
1280
|
raise Error, "more than #{MAX_FIELDS} fields" if (count[0] += 1) > MAX_FIELDS
|
|
986
1281
|
children = t.tables(5).map { |c| field(c, depth + 1, count) }
|
|
987
1282
|
metadata = key_values(t.tables(6))
|
|
988
1283
|
type = type_name(t.u8(2), t.table(3), children)
|
|
989
|
-
h = {
|
|
1284
|
+
h = {name: t.string(0).to_s, type: type, nullable: t.bool(1)}
|
|
990
1285
|
if (d = t.table(4))
|
|
991
1286
|
index = d.table(1)
|
|
992
1287
|
index_type = index ? int_name(index) : "int32"
|
|
993
1288
|
ordered = d.bool(2)
|
|
994
|
-
h[:dictionary] = {
|
|
1289
|
+
h[:dictionary] = {index_type: index_type, ordered: ordered, id: d.i64(0)}
|
|
995
1290
|
h[:type] = "dictionary<values=#{type}, indices=#{index_type}, ordered=#{ordered ? 1 : 0}>"
|
|
996
1291
|
end
|
|
997
1292
|
h[:children] = children unless children.empty?
|
|
@@ -1002,19 +1297,32 @@ module Herringbone
|
|
|
1002
1297
|
h
|
|
1003
1298
|
end
|
|
1004
1299
|
|
|
1300
|
+
# @param tables [Array<Table>] KeyValue tables (custom_metadata of a Schema or Field)
|
|
1301
|
+
# @return [Hash{String => String}, nil] nil when there are none
|
|
1005
1302
|
def key_values(tables)
|
|
1006
1303
|
return nil if tables.empty?
|
|
1007
1304
|
tables.to_h { |kv| [kv.string(0).to_s, kv.string(1).to_s] }
|
|
1008
1305
|
end
|
|
1009
1306
|
|
|
1010
1307
|
# A child as pyarrow prints it inside a nested type: "name: type" plus " not null"
|
|
1011
|
-
|
|
1308
|
+
# @param c [Hash{Symbol => Object}] a field as returned by #field
|
|
1309
|
+
# @return [String]
|
|
1310
|
+
def child_text(c) = "#{c[:name]}: #{c[:type]}#{" not null" unless c[:nullable]}"
|
|
1012
1311
|
|
|
1013
|
-
|
|
1312
|
+
# @param t [Table] an Arrow Int table (bitWidth, is_signed)
|
|
1313
|
+
# @return [String] e.g. "int32" or "uint8"
|
|
1314
|
+
def int_name(t) = "#{"u" unless t.bool(1)}int#{t.i32(0)}"
|
|
1014
1315
|
|
|
1316
|
+
# @param u [Integer] Arrow TimeUnit value
|
|
1317
|
+
# @return [String] "s", "ms", "us" or "ns" ("unitN" for unknown values)
|
|
1015
1318
|
def unit(u) = TIME_UNITS[u] || "unit#{u}"
|
|
1016
1319
|
|
|
1017
1320
|
# Type names follow Arrow's DataType::ToString (what pyarrow prints)
|
|
1321
|
+
# @param kind [Integer] the Field's Type union tag (Schema.fbs)
|
|
1322
|
+
# @param t [Table, nil] the union's type table, nil when absent
|
|
1323
|
+
# @param children [Array<Hash{Symbol => Object}>] the already described child fields
|
|
1324
|
+
# @return [String] e.g. "timestamp[us, tz=UTC]" or "list<item: string>"
|
|
1325
|
+
# @raise [Error] for a Decimal type without its parameters table
|
|
1018
1326
|
def type_name(kind, t, children)
|
|
1019
1327
|
case kind
|
|
1020
1328
|
when 1 then "null"
|
|
@@ -1033,7 +1341,7 @@ module Herringbone
|
|
|
1033
1341
|
"time#{bits}[#{unit(t ? t.i16(0, 1) : 1)}]"
|
|
1034
1342
|
when 10
|
|
1035
1343
|
tz = t&.string(1)
|
|
1036
|
-
"timestamp[#{unit(t ? t.i16(0) : 0)}#{
|
|
1344
|
+
"timestamp[#{unit(t ? t.i16(0) : 0)}#{", tz=#{tz}" if tz}]"
|
|
1037
1345
|
when 11 then %w[month_interval day_time_interval month_day_nano_interval][t ? t.i16(0) : 0] || "interval"
|
|
1038
1346
|
when 12 then "list<#{children.map { |c| child_text(c) }.join(", ")}>"
|
|
1039
1347
|
when 13 then "struct<#{children.map { |c| child_text(c) }.join(", ")}>"
|
|
@@ -1049,7 +1357,7 @@ module Herringbone
|
|
|
1049
1357
|
when 19 then "large_binary"
|
|
1050
1358
|
when 20 then "large_string"
|
|
1051
1359
|
when 21 then "large_list<#{children.map { |c| child_text(c) }.join(", ")}>"
|
|
1052
|
-
when 22 then "run_end_encoded<#{children.map { |c| "#{c[:name] == "values" ? "values" : "run_ends"}: #{c[:type]}" }.join(", ")}>"
|
|
1360
|
+
when 22 then "run_end_encoded<#{children.map { |c| "#{(c[:name] == "values") ? "values" : "run_ends"}: #{c[:type]}" }.join(", ")}>"
|
|
1053
1361
|
when 23 then "binary_view"
|
|
1054
1362
|
when 24 then "string_view"
|
|
1055
1363
|
when 25 then "list_view<#{children.map { |c| child_text(c) }.join(", ")}>"
|
|
@@ -1059,16 +1367,24 @@ module Herringbone
|
|
|
1059
1367
|
end
|
|
1060
1368
|
|
|
1061
1369
|
# map<key, value> with non-standard field names in parentheses, as Arrow prints it
|
|
1370
|
+
# @param t [Table, nil] the Map table (keysSorted), nil when absent
|
|
1371
|
+
# @param children [Array<Hash{Symbol => Object}>] the map's single entries struct field
|
|
1372
|
+
# @return [String]
|
|
1062
1373
|
def map_name(t, children)
|
|
1063
1374
|
entries = children.first
|
|
1064
1375
|
kv = entries && entries[:children] || []
|
|
1065
|
-
named = ->(f, std) { f ? "#{f[:type]}#{
|
|
1376
|
+
named = ->(f, std) { f ? "#{f[:type]}#{" ('#{f[:name]}')" unless f[:name] == std}" : "?" }
|
|
1066
1377
|
sorted = t&.bool(0) ? ", keys_sorted" : ""
|
|
1067
|
-
entries_name = entries && entries[:name] != "entries" ? " ('#{entries[:name]}')" : ""
|
|
1378
|
+
entries_name = (entries && entries[:name] != "entries") ? " ('#{entries[:name]}')" : ""
|
|
1068
1379
|
"map<#{named.call(kv[0], "key")}, #{named.call(kv[1], "value")}#{sorted}#{entries_name}>"
|
|
1069
1380
|
end
|
|
1070
1381
|
|
|
1071
1382
|
# "name: type" lines for a field and its children, indented, for text output
|
|
1383
|
+
# Children nested deeper than 8 levels are left out.
|
|
1384
|
+
# @param fields [Array<Hash{Symbol => Object}>] fields as returned by #decode under +:fields+
|
|
1385
|
+
# @param depth [Integer] indentation level of +fields+
|
|
1386
|
+
# @param out [Array<String>] accumulator the lines are appended to
|
|
1387
|
+
# @return [Array<String>] +out+
|
|
1072
1388
|
def lines(fields, depth = 0, out = [])
|
|
1073
1389
|
fields.each do |f|
|
|
1074
1390
|
notes = []
|
|
@@ -1076,7 +1392,7 @@ module Herringbone
|
|
|
1076
1392
|
notes << "extension #{f[:extension]}" if f[:extension]
|
|
1077
1393
|
meta = (f[:metadata] || {}).reject { |k, _| k.start_with?("ARROW:extension:") }
|
|
1078
1394
|
notes << "metadata #{meta.map { |k, v| "#{k}=#{v.to_s[0, 60].inspect}" }.join(", ")}" unless meta.empty?
|
|
1079
|
-
out << "#{" " * depth}#{f[:name]}: #{f[:type]}#{
|
|
1395
|
+
out << "#{" " * depth}#{f[:name]}: #{f[:type]}#{" (#{notes.join("; ")})" unless notes.empty?}"
|
|
1080
1396
|
lines(f[:children], depth + 1, out) if f[:children] && depth < 8
|
|
1081
1397
|
end
|
|
1082
1398
|
out
|
|
@@ -1085,8 +1401,13 @@ module Herringbone
|
|
|
1085
1401
|
|
|
1086
1402
|
# ---- class helpers ----
|
|
1087
1403
|
|
|
1404
|
+
# @param e [Integer] Parquet Encoding value
|
|
1405
|
+
# @return [String] its name, e.g. "RLE_DICTIONARY" (the number as a String when unknown)
|
|
1088
1406
|
def self.encoding_name(e) = Format::Encoding::NAMES[e]&.to_s || e.to_s
|
|
1089
1407
|
|
|
1408
|
+
# @param column [Schema::Column] leaf column
|
|
1409
|
+
# @return [String] physical type with its logical annotation, e.g. "BYTE_ARRAY STRING" or
|
|
1410
|
+
# "FIXED_LEN_BYTE_ARRAY(16) UUID"
|
|
1090
1411
|
def self.type_name(column)
|
|
1091
1412
|
node = column.node
|
|
1092
1413
|
phys = T::NAMES[node.type].to_s
|
|
@@ -1095,6 +1416,10 @@ module Herringbone
|
|
|
1095
1416
|
logical ? "#{phys} #{logical}" : phys
|
|
1096
1417
|
end
|
|
1097
1418
|
|
|
1419
|
+
# The node's LogicalType, else its ConvertedType, as text
|
|
1420
|
+
# @param node [Schema::Node] schema node (leaf or group)
|
|
1421
|
+
# @return [String, nil] e.g. "INTEGER(8, unsigned)", "TIMESTAMP(MICROS, UTC)" or "DECIMAL(10, 2)";
|
|
1422
|
+
# nil when the node has no annotation
|
|
1098
1423
|
def self.logical_type_name(node)
|
|
1099
1424
|
if (kind = node.logical_type&.kind)
|
|
1100
1425
|
name, payload = kind
|
|
@@ -1108,15 +1433,17 @@ module Herringbone
|
|
|
1108
1433
|
end
|
|
1109
1434
|
elsif node.converted_type
|
|
1110
1435
|
c = Format::ConvertedType::NAMES[node.converted_type].to_s
|
|
1111
|
-
c == "DECIMAL" ? "DECIMAL(#{node.precision}, #{node.scale || 0})" : c
|
|
1436
|
+
(c == "DECIMAL") ? "DECIMAL(#{node.precision}, #{node.scale || 0})" : c
|
|
1112
1437
|
end
|
|
1113
1438
|
end
|
|
1114
1439
|
|
|
1115
1440
|
# :signed, :unsigned or :unknown, per the Parquet sort order rules for the column's type
|
|
1441
|
+
# @param column [Schema::Column] leaf column
|
|
1442
|
+
# @return [Symbol]
|
|
1116
1443
|
def self.sort_order(column)
|
|
1117
1444
|
kind, _a, signed = Types.logical_of(column.node)
|
|
1118
1445
|
case kind
|
|
1119
|
-
when :integer then signed == false ? :unsigned : :signed
|
|
1446
|
+
when :integer then (signed == false) ? :unsigned : :signed
|
|
1120
1447
|
when :decimal, :date, :time, :timestamp, :float16 then :signed
|
|
1121
1448
|
when :string, :enum, :json, :bson, :uuid then :unsigned
|
|
1122
1449
|
else
|
|
@@ -1129,6 +1456,9 @@ module Herringbone
|
|
|
1129
1456
|
end
|
|
1130
1457
|
|
|
1131
1458
|
# Converts Ruby values (Time, BigDecimal, binary Strings, non-finite Floats...) to JSON-safe ones
|
|
1459
|
+
# Recurses into Hashes, Arrays and Structs; Hash keys other than Symbols become Strings.
|
|
1460
|
+
# @param v [Object] value to convert
|
|
1461
|
+
# @return [Hash, Array, String, Integer, Float, Boolean, nil]
|
|
1132
1462
|
def self.jsonable(v)
|
|
1133
1463
|
case v
|
|
1134
1464
|
when Hash then v.each_with_object({}) { |(k, x), h| h[k.is_a?(Symbol) ? k : k.to_s] = jsonable(x) }
|
|
@@ -1146,6 +1476,9 @@ module Herringbone
|
|
|
1146
1476
|
end
|
|
1147
1477
|
|
|
1148
1478
|
# A String as readable text when it is valid UTF-8 without control characters, else as hex
|
|
1479
|
+
# Strings already tagged as valid UTF-8 are returned as they are, control characters and all.
|
|
1480
|
+
# @param s [String] string in any encoding
|
|
1481
|
+
# @return [String]
|
|
1149
1482
|
def self.text(s)
|
|
1150
1483
|
return s if s.encoding == Encoding::UTF_8 && s.valid_encoding?
|
|
1151
1484
|
u = s.dup.force_encoding(Encoding::UTF_8)
|
|
@@ -1153,22 +1486,31 @@ module Herringbone
|
|
|
1153
1486
|
hex(s)
|
|
1154
1487
|
end
|
|
1155
1488
|
|
|
1489
|
+
# @param bytes [String] bytes to show
|
|
1490
|
+
# @return [String] "0x" and lowercase hex digits; only the first 64 bytes, followed by the
|
|
1491
|
+
# total size, for longer input
|
|
1156
1492
|
def self.hex(bytes)
|
|
1157
1493
|
b = bytes.b
|
|
1158
|
-
b.bytesize > 64 ? "0x#{b.byteslice(0, 64).unpack1("H*")}… (#{b.bytesize} bytes)" : "0x#{b.unpack1("H*")}"
|
|
1494
|
+
(b.bytesize > 64) ? "0x#{b.byteslice(0, 64).unpack1("H*")}… (#{b.bytesize} bytes)" : "0x#{b.unpack1("H*")}"
|
|
1159
1495
|
end
|
|
1160
1496
|
|
|
1161
1497
|
# A short display form of a decoded value
|
|
1498
|
+
# Strings are quoted (binary ones go through Inspector.text first), nil shows as "null".
|
|
1499
|
+
# @param v [Object] decoded value
|
|
1500
|
+
# @param max [Integer] longest result, in characters; longer ones are cut and end in an ellipsis
|
|
1501
|
+
# @return [String]
|
|
1162
1502
|
def self.display(v, max: 40)
|
|
1163
1503
|
s = case v
|
|
1164
|
-
when String then v.encoding == Encoding::BINARY ? text(v) : v
|
|
1504
|
+
when String then (v.encoding == Encoding::BINARY) ? text(v) : v
|
|
1165
1505
|
when nil then "null"
|
|
1166
1506
|
else jsonable(v).to_s
|
|
1167
1507
|
end
|
|
1168
1508
|
s = s.inspect if v.is_a?(String)
|
|
1169
|
-
s.size > max ? "#{s[0, max - 1]}…" : s
|
|
1509
|
+
(s.size > max) ? "#{s[0, max - 1]}…" : s
|
|
1170
1510
|
end
|
|
1171
1511
|
|
|
1512
|
+
# @param n [Integer, nil] byte count
|
|
1513
|
+
# @return [String] e.g. "512 B", "1.50 KB" or "12.3 MB" (binary units); "?" for nil
|
|
1172
1514
|
def self.human_bytes(n)
|
|
1173
1515
|
return "?" unless n
|
|
1174
1516
|
units = %w[B KB MB GB TB]
|
|
@@ -1178,15 +1520,25 @@ module Herringbone
|
|
|
1178
1520
|
f /= 1024
|
|
1179
1521
|
i += 1
|
|
1180
1522
|
end
|
|
1181
|
-
i.zero?
|
|
1523
|
+
if i.zero?
|
|
1524
|
+
"#{n} B"
|
|
1525
|
+
else
|
|
1526
|
+
format("%.#{(f < 10) ? 2 : 1}f %s", f, units[i])
|
|
1527
|
+
end
|
|
1182
1528
|
end
|
|
1183
1529
|
|
|
1184
1530
|
private
|
|
1185
1531
|
|
|
1532
|
+
# @param uncompressed [Integer, nil] uncompressed byte count
|
|
1533
|
+
# @param compressed [Integer, nil] compressed byte count
|
|
1534
|
+
# @return [String] " (3.21x)" for #report, empty when the ratio is unknown
|
|
1186
1535
|
def ratio_text(uncompressed, compressed)
|
|
1187
|
-
compressed.to_i.positive? && uncompressed ? format(" (%.2fx)", uncompressed.to_f / compressed) : ""
|
|
1536
|
+
(compressed.to_i.positive? && uncompressed) ? format(" (%.2fx)", uncompressed.to_f / compressed) : ""
|
|
1188
1537
|
end
|
|
1189
1538
|
|
|
1539
|
+
# @param page [PageInfo] page whose CRC status to show
|
|
1540
|
+
# @return [String] the status for a #report page line: " crc ok", " CRC MISMATCH", " crc"
|
|
1541
|
+
# (has a CRC, not verified) or empty
|
|
1190
1542
|
def crc_text(page)
|
|
1191
1543
|
case page.checksum
|
|
1192
1544
|
when :ok then " crc ok"
|
|
@@ -1195,6 +1547,8 @@ module Herringbone
|
|
|
1195
1547
|
end
|
|
1196
1548
|
end
|
|
1197
1549
|
|
|
1550
|
+
# @param m [Hash{Symbol => Object}] an entry of #index_mismatches
|
|
1551
|
+
# @return [String] one #report line describing it
|
|
1198
1552
|
def index_mismatch_text(m)
|
|
1199
1553
|
where = "row group #{m[:row_group]} #{m[:column]}"
|
|
1200
1554
|
if m[:field] == :page_count
|
|
@@ -1206,6 +1560,10 @@ module Herringbone
|
|
|
1206
1560
|
end
|
|
1207
1561
|
|
|
1208
1562
|
# Adds :arrow_type to schema nodes with a same-named Arrow field (top level, and struct members)
|
|
1563
|
+
# @param nodes [Array<Hash{Symbol => Object}>] #schema_tree nodes, updated in place
|
|
1564
|
+
# @param fields [Array<Hash{Symbol => Object}>] Arrow fields at the same level
|
|
1565
|
+
# @param depth [Integer] nesting depth; recursion stops at 32
|
|
1566
|
+
# @return [void]
|
|
1209
1567
|
def annotate_arrow_types(nodes, fields, depth = 0)
|
|
1210
1568
|
by_name = fields.to_h { |f| [f[:name], f] }
|
|
1211
1569
|
nodes.each do |n|
|
|
@@ -1220,35 +1578,48 @@ module Herringbone
|
|
|
1220
1578
|
# Whether the index bound +idx+ excludes values the page's bound +page+ says are present
|
|
1221
1579
|
# (index min above the page min, or index max below the page max). Truncated binary bounds
|
|
1222
1580
|
# (one a prefix of the other) and values that can't be compared are never reported.
|
|
1581
|
+
# @param idx [Object, nil] decoded ColumnIndex bound
|
|
1582
|
+
# @param page [Object, nil] decoded page header bound
|
|
1583
|
+
# @param order [Symbol] :signed, :unsigned or :unknown (see Inspector.sort_order)
|
|
1584
|
+
# @param which [Symbol] :min or :max
|
|
1585
|
+
# @return [Boolean]
|
|
1223
1586
|
def narrower?(idx, page, order, which)
|
|
1224
1587
|
return false if idx.nil? || page.nil? || order == :unknown
|
|
1225
|
-
a, b = which == :min ? [page, idx] : [idx, page] # true when a < b
|
|
1588
|
+
a, b = (which == :min) ? [page, idx] : [idx, page] # true when a < b
|
|
1226
1589
|
if a.is_a?(String) && b.is_a?(String)
|
|
1227
1590
|
a = a.b
|
|
1228
1591
|
b = b.b
|
|
1229
1592
|
return false if a.start_with?(b) || b.start_with?(a)
|
|
1230
|
-
return order == :unsigned ? a < b : false
|
|
1593
|
+
return (order == :unsigned) ? a < b : false
|
|
1231
1594
|
end
|
|
1232
1595
|
return false if a.is_a?(Float) && a.nan? || b.is_a?(Float) && b.nan?
|
|
1233
1596
|
a = a ? 1 : 0 if a == true || a == false
|
|
1234
1597
|
b = b ? 1 : 0 if b == true || b == false
|
|
1235
1598
|
(a <=> b) == -1
|
|
1236
|
-
rescue
|
|
1599
|
+
rescue
|
|
1237
1600
|
false
|
|
1238
1601
|
end
|
|
1239
1602
|
|
|
1603
|
+
# Overall min or max of per-chunk bounds, when they can be compared
|
|
1604
|
+
# @param values [Array<Object, nil>] decoded per-chunk minimums or maximums
|
|
1605
|
+
# @param which [Symbol] :min or :max
|
|
1606
|
+
# @return [Object, nil] nil when there are no values or they are of mixed or incomparable types
|
|
1240
1607
|
def safe_extreme(values, which)
|
|
1241
1608
|
vals = values.compact
|
|
1242
1609
|
return nil if vals.empty?
|
|
1243
1610
|
if vals.all? { |v| v == true || v == false }
|
|
1244
|
-
return which == :min ? vals.all? : vals.any?
|
|
1611
|
+
return (which == :min) ? vals.all? : vals.any?
|
|
1245
1612
|
end
|
|
1246
1613
|
return nil unless vals.map(&:class).uniq.size == 1
|
|
1247
|
-
which == :min ? vals.min : vals.max
|
|
1614
|
+
(which == :min) ? vals.min : vals.max
|
|
1248
1615
|
rescue ArgumentError, NoMethodError
|
|
1249
1616
|
nil
|
|
1250
1617
|
end
|
|
1251
1618
|
|
|
1619
|
+
# Reads the file size, the footer length and magic, and decodes the FileMetaData into @metadata
|
|
1620
|
+
# @return [void]
|
|
1621
|
+
# @raise [FormatError] when the file is too small, lacks the magic bytes or has a corrupt footer
|
|
1622
|
+
# @raise [UnsupportedError] when the file is encrypted (PARE magic)
|
|
1252
1623
|
def read_footer
|
|
1253
1624
|
@io.seek(0, IO::SEEK_END)
|
|
1254
1625
|
@file_size = @io.pos
|
|
@@ -1265,6 +1636,9 @@ module Herringbone
|
|
|
1265
1636
|
|
|
1266
1637
|
# Reads +len+ bytes at +pos+ through a small read-ahead window, so walking many small pages
|
|
1267
1638
|
# does not cost a syscall per header
|
|
1639
|
+
# @param pos [Integer] file offset
|
|
1640
|
+
# @param len [Integer] bytes wanted
|
|
1641
|
+
# @return [String] binary String; shorter than +len+ at the end of the file
|
|
1268
1642
|
def read_at(pos, len)
|
|
1269
1643
|
if @window && pos >= @window_pos && pos + len <= @window_pos + @window.bytesize
|
|
1270
1644
|
return @window.byteslice(pos - @window_pos, len)
|
|
@@ -1281,6 +1655,11 @@ module Herringbone
|
|
|
1281
1655
|
|
|
1282
1656
|
# Decodes the page header at +pos+, reading more bytes when it is larger than the first guess
|
|
1283
1657
|
# (page statistics of long strings can make headers big)
|
|
1658
|
+
# @param pos [Integer] file offset of the header
|
|
1659
|
+
# @param limit [Integer] offset the header must not extend past (the footer's start)
|
|
1660
|
+
# @return [Array(Format::PageHeader, Integer)] the header and its encoded size in bytes
|
|
1661
|
+
# @raise [FormatError] when there is no room for a header before +limit+
|
|
1662
|
+
# @raise [Thrift::Error] when it does not decode within +limit+ or 16 MiB
|
|
1284
1663
|
def read_page_header(pos, limit)
|
|
1285
1664
|
want = 256
|
|
1286
1665
|
while true
|
|
@@ -1297,6 +1676,14 @@ module Herringbone
|
|
|
1297
1676
|
end
|
|
1298
1677
|
end
|
|
1299
1678
|
|
|
1679
|
+
# Builds a PageInfo from a decoded page header
|
|
1680
|
+
# @param index [Integer] position of the page in its chunk
|
|
1681
|
+
# @param h [Format::PageHeader] decoded header
|
|
1682
|
+
# @param pos [Integer] file offset of the header
|
|
1683
|
+
# @param header_size [Integer] encoded size of the header in bytes
|
|
1684
|
+
# @param column [Schema::Column] column the page belongs to (for statistics and row counts)
|
|
1685
|
+
# @return [PageInfo]
|
|
1686
|
+
# @raise [FormatError] when the header declares a negative compressed size
|
|
1300
1687
|
def page_info(index, h, pos, header_size, column)
|
|
1301
1688
|
info = PageInfo.new(
|
|
1302
1689
|
index: index,
|
|
@@ -1324,7 +1711,7 @@ module Herringbone
|
|
|
1324
1711
|
info.encoding = Inspector.encoding_name(d.encoding)
|
|
1325
1712
|
info.definition_levels_byte_length = d.definition_levels_byte_length
|
|
1326
1713
|
info.repetition_levels_byte_length = d.repetition_levels_byte_length
|
|
1327
|
-
info.is_compressed = d.is_compressed.nil?
|
|
1714
|
+
info.is_compressed = d.is_compressed.nil? || d.is_compressed
|
|
1328
1715
|
info.statistics = decode_statistics(d.statistics, column)
|
|
1329
1716
|
elsif (d = h.dictionary_page_header)
|
|
1330
1717
|
info.num_values = d.num_values
|
|
@@ -1334,16 +1721,20 @@ module Herringbone
|
|
|
1334
1721
|
info
|
|
1335
1722
|
end
|
|
1336
1723
|
|
|
1724
|
+
# One #key_value_metadata entry
|
|
1725
|
+
# @param key [String] metadata key
|
|
1726
|
+
# @param value [String, nil] metadata value
|
|
1727
|
+
# @return [Hash{Symbol => Object}] see #key_value_metadata
|
|
1337
1728
|
def describe_key_value(key, value)
|
|
1338
1729
|
value = value.to_s
|
|
1339
|
-
h = {
|
|
1730
|
+
h = {key: key, bytesize: value.bytesize}
|
|
1340
1731
|
if key == "ARROW:schema"
|
|
1341
1732
|
h[:format] = "arrow_schema"
|
|
1342
|
-
h[:value] = value.size > 120 ? "#{value[0, 120]}…" : value
|
|
1733
|
+
h[:value] = (value.size > 120) ? "#{value[0, 120]}…" : value
|
|
1343
1734
|
if (arrow = arrow_schema)
|
|
1344
|
-
h[:summary] = "Arrow schema, #{arrow[:fields].size} field#{arrow[:fields].size == 1
|
|
1735
|
+
h[:summary] = "Arrow schema, #{arrow[:fields].size} field#{"s" unless arrow[:fields].size == 1}"
|
|
1345
1736
|
h[:arrow_fields] = arrow[:fields].map { |f| f[:name] }
|
|
1346
|
-
meta = arrow[:metadata]&.transform_values { |v| v.size > 4000 ? "#{v[0, 4000]}…" : v }
|
|
1737
|
+
meta = arrow[:metadata]&.transform_values { |v| (v.size > 4000) ? "#{v[0, 4000]}…" : v }
|
|
1347
1738
|
h[:arrow_schema] = arrow.merge(metadata: meta).compact
|
|
1348
1739
|
else
|
|
1349
1740
|
h[:summary] = "Arrow IPC schema message, base64-encoded (#{value.bytesize} bytes)"
|
|
@@ -1355,19 +1746,21 @@ module Herringbone
|
|
|
1355
1746
|
begin
|
|
1356
1747
|
h[:json] = JSON.parse(value)
|
|
1357
1748
|
h[:format] = "json"
|
|
1358
|
-
h[:summary] = key == "pandas" ? pandas_summary(h[:json]) : "JSON"
|
|
1749
|
+
h[:summary] = (key == "pandas") ? pandas_summary(h[:json]) : "JSON"
|
|
1359
1750
|
rescue JSON::ParserError
|
|
1360
1751
|
h[:format] = "text"
|
|
1361
1752
|
end
|
|
1362
|
-
h[:value] = value.size > 4000 ? "#{value[0, 4000]}…" : value
|
|
1753
|
+
h[:value] = (value.size > 4000) ? "#{value[0, 4000]}…" : value
|
|
1363
1754
|
else
|
|
1364
1755
|
t = Inspector.text(value.b)
|
|
1365
|
-
h[:format] = t.start_with?("0x") && value.bytesize.positive? ? "binary" : "text"
|
|
1366
|
-
h[:value] = t.size > 4000 ? "#{t[0, 4000]}…" : t
|
|
1756
|
+
h[:format] = (t.start_with?("0x") && value.bytesize.positive?) ? "binary" : "text"
|
|
1757
|
+
h[:value] = (t.size > 4000) ? "#{t[0, 4000]}…" : t
|
|
1367
1758
|
end
|
|
1368
1759
|
h
|
|
1369
1760
|
end
|
|
1370
1761
|
|
|
1762
|
+
# @param json [Object] the parsed "pandas" metadata value
|
|
1763
|
+
# @return [String] e.g. "pandas 2.2.0 metadata, 3 columns"
|
|
1371
1764
|
def pandas_summary(json)
|
|
1372
1765
|
return "pandas metadata" unless json.is_a?(Hash)
|
|
1373
1766
|
cols = json["columns"]&.size
|
|
@@ -1376,6 +1769,9 @@ module Herringbone
|
|
|
1376
1769
|
|
|
1377
1770
|
# Field names from an Arrow IPC schema message, found by scanning for its flatbuffer strings.
|
|
1378
1771
|
# Best effort: only used to label the blob, nil when nothing sensible is found.
|
|
1772
|
+
# Only top-level Parquet column names that occur in the decoded bytes are reported.
|
|
1773
|
+
# @param b64 [String] the base64 ARROW:schema value
|
|
1774
|
+
# @return [Array<String>, nil]
|
|
1379
1775
|
def arrow_field_names(b64)
|
|
1380
1776
|
bytes = b64.unpack1("m")
|
|
1381
1777
|
schema_cols = columns.map { |c| c.path.first }.uniq
|