herringbone 0.1.0 → 0.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/README.md +198 -74
- data/bin/herringbone +38 -26
- data/lib/herringbone/active_record.rb +48 -19
- data/lib/herringbone/bloom_filter.rb +370 -0
- data/lib/herringbone/byte_values.rb +29 -0
- data/lib/herringbone/codecs/lz4.rb +66 -3
- data/lib/herringbone/codecs/snappy.rb +132 -3
- data/lib/herringbone/compression.rb +121 -16
- data/lib/herringbone/encodings/delta.rb +76 -19
- data/lib/herringbone/encodings/plain.rb +29 -7
- data/lib/herringbone/encodings/rle.rb +55 -3
- data/lib/herringbone/format.rb +203 -0
- data/lib/herringbone/inspector.rb +1784 -0
- data/lib/herringbone/io_buffer_support.rb +3 -1
- data/lib/herringbone/reader/column_chunk_reader.rb +388 -0
- data/lib/herringbone/reader/column_cursor.rb +225 -0
- data/lib/herringbone/reader/numo.rb +365 -0
- data/lib/herringbone/reader/page_stream.rb +615 -0
- data/lib/herringbone/reader/scan.rb +350 -0
- data/lib/herringbone/reader.rb +585 -272
- data/lib/herringbone/schema.rb +326 -90
- data/lib/herringbone/thrift.rb +155 -4
- data/lib/herringbone/types.rb +177 -42
- data/lib/herringbone/version.rb +2 -1
- data/lib/herringbone/visualizer.rb +1136 -0
- data/lib/herringbone/writer.rb +550 -105
- data/lib/herringbone/xxhash.rb +425 -0
- data/lib/herringbone.rb +64 -9
- metadata +14 -33
data/lib/herringbone/reader.rb
CHANGED
|
@@ -1,133 +1,595 @@
|
|
|
1
1
|
# frozen_string_literal: true
|
|
2
2
|
|
|
3
3
|
module Herringbone
|
|
4
|
-
# Reads Parquet files.
|
|
4
|
+
# Reads Parquet files from a random-access IO. Herringbone never opens files by path: the
|
|
5
|
+
# caller opens (and closes) the IO.
|
|
5
6
|
#
|
|
6
|
-
#
|
|
7
|
-
# reader
|
|
8
|
-
# reader.
|
|
9
|
-
# reader.
|
|
7
|
+
# File.open("data.parquet", "rb") do |file|
|
|
8
|
+
# reader = Herringbone::Reader.new(file)
|
|
9
|
+
# reader.each_row { |row| p row } # rows as Hashes with String keys
|
|
10
|
+
# reader.each_batch(1000, as: :columns) { |batch| ... } # { "id" => [...], ... } per batch
|
|
11
|
+
# reader.read(columns: ["id"], where: { id: 1..10 }) # everything at once
|
|
10
12
|
# end
|
|
13
|
+
#
|
|
14
|
+
# Rows are read in batches: pages are read and decoded one at a time per column, so memory use
|
|
15
|
+
# depends on the batch size and the page size, not on the size of the row groups.
|
|
16
|
+
#
|
|
17
|
+
# Options:
|
|
18
|
+
# keys: :string (default) or :symbol, for row Hashes and Hashes built from structs.
|
|
19
|
+
# Map keys are always the stored values.
|
|
20
|
+
# time_zone: return timestamps in this zone instead of UTC. A UTC offset ("+02:00", or
|
|
21
|
+
# seconds as an Integer), a timezone object Time#getlocal accepts (e.g. a
|
|
22
|
+
# TZInfo::Timezone), or anything responding to #at such as an
|
|
23
|
+
# ActiveSupport::TimeZone (Time.zone), which yields ActiveSupport::TimeWithZone.
|
|
11
24
|
class Reader
|
|
12
|
-
|
|
13
|
-
|
|
25
|
+
# The 4 bytes a Parquet file starts and ends with
|
|
14
26
|
MAGIC = "PAR1"
|
|
27
|
+
# Rows per batch in #each_batch when no size is given
|
|
28
|
+
DEFAULT_BATCH_SIZE = 1024
|
|
29
|
+
# Accepted values of the +keys:+ option
|
|
30
|
+
KEY_MODES = %i[string symbol].freeze
|
|
31
|
+
# Accepted values of the +as:+ option
|
|
32
|
+
AS_MODES = %i[rows columns numo].freeze
|
|
33
|
+
|
|
34
|
+
# @return [Schema] the schema, built from the footer's flattened SchemaElements
|
|
35
|
+
attr_reader :schema
|
|
36
|
+
|
|
37
|
+
# @return [Format::FileMetaData] the file's FileMetaData (the decoded Thrift footer)
|
|
38
|
+
attr_reader :file_metadata
|
|
39
|
+
|
|
40
|
+
# +io+ must support #seek and #read (a File opened with "rb", StringIO, Tempfile...)
|
|
41
|
+
#
|
|
42
|
+
# @param io [IO, StringIO] random-access source of the Parquet bytes; the caller closes it
|
|
43
|
+
# @param keys [Symbol, String] +:string+ or +:symbol+, the key type of row and struct Hashes
|
|
44
|
+
# @param time_zone [String, Integer, Object, nil] zone timestamps are returned in (see the
|
|
45
|
+
# class docs); nil keeps them in UTC
|
|
46
|
+
# @raise [ArgumentError] when +io+ cannot seek and read, or +keys+ / +time_zone+ are invalid
|
|
47
|
+
# @raise [FormatError] when the footer is missing or cannot be decoded
|
|
48
|
+
def initialize(io, keys: :string, time_zone: nil)
|
|
49
|
+
unless io.respond_to?(:seek) && io.respond_to?(:read)
|
|
50
|
+
raise ArgumentError, "Herringbone::Reader expects an IO that supports #seek and #read " \
|
|
51
|
+
"(e.g. File.open(path, \"rb\")), got #{io.is_a?(String) ? "a String" : io.class}" \
|
|
52
|
+
"#{" (wrap Parquet bytes in a StringIO)" if io.is_a?(String)}"
|
|
53
|
+
end
|
|
54
|
+
keys = keys.to_sym if keys.is_a?(String)
|
|
55
|
+
raise ArgumentError, "keys: must be :string or :symbol, got #{keys.inspect}" unless KEY_MODES.include?(keys)
|
|
56
|
+
@symbolize = keys == :symbol
|
|
57
|
+
@zone_converter = zone_converter(time_zone)
|
|
58
|
+
@io = io
|
|
59
|
+
@file_metadata = read_footer
|
|
60
|
+
@schema = Schema.from_elements(@file_metadata.schema)
|
|
61
|
+
end
|
|
15
62
|
|
|
16
|
-
|
|
63
|
+
# Total number of rows, as stated in the footer (some writers store 0)
|
|
64
|
+
#
|
|
65
|
+
# @return [Integer] the footer's num_rows
|
|
66
|
+
def num_rows = @file_metadata.num_rows
|
|
67
|
+
|
|
68
|
+
# The row groups listed in the footer
|
|
69
|
+
#
|
|
70
|
+
# @return [Array<Format::RowGroup>] row group metadata, in file order
|
|
71
|
+
def row_groups = @file_metadata.row_groups
|
|
72
|
+
|
|
73
|
+
# The footer's key/value metadata as a Hash (what the writer's metadata: option stores)
|
|
74
|
+
#
|
|
75
|
+
# @return [Hash{String => String, nil}] key => value; empty when the footer has none
|
|
76
|
+
def metadata
|
|
77
|
+
(@file_metadata.key_value_metadata || []).to_h { |kv| [kv.key, kv.value] }
|
|
78
|
+
end
|
|
17
79
|
|
|
18
|
-
|
|
19
|
-
|
|
20
|
-
|
|
21
|
-
|
|
22
|
-
|
|
23
|
-
|
|
24
|
-
|
|
25
|
-
|
|
80
|
+
# Yields batches of up to +size+ rows (all batches are full except the last one; batches span
|
|
81
|
+
# row groups). Only one page per column and the current batch are held in memory.
|
|
82
|
+
#
|
|
83
|
+
# as: :rows (default) yields an Array of row Hashes; as: :columns yields a Hash of top-level
|
|
84
|
+
# field name => Array of that field's values in the batch, which skips building a Hash per row
|
|
85
|
+
# and is noticeably faster when you process data column by column. as: :numo yields a Hash of
|
|
86
|
+
# field name => Numo array (needs the numo-narray-alt or numo-narray gem; see NumoColumns
|
|
87
|
+
# for the type mapping). With as: :numo, whether an integer column becomes DFloat (nulls) or
|
|
88
|
+
# a list column 2-D is decided per batch, from the values in it.
|
|
89
|
+
#
|
|
90
|
+
# where: only yields rows matching all conditions (see Reader::Filter). Row groups and pages
|
|
91
|
+
# that cannot match are skipped using statistics, bloom filters and the page index, and the
|
|
92
|
+
# remaining rows are checked one by one. Filtered columns need not be in +columns+.
|
|
93
|
+
# from: skips the first rows of the file (jumping over pages with the page index), and
|
|
94
|
+
# limit: stops after yielding that many rows.
|
|
95
|
+
#
|
|
96
|
+
# @param size [Integer] maximum number of rows per batch
|
|
97
|
+
# @param columns [Array<String, Symbol>, String, Symbol, nil] top-level fields to return;
|
|
98
|
+
# nil returns all of them
|
|
99
|
+
# @param as [Symbol] +:rows+, +:columns+ or +:numo+, the shape of each batch
|
|
100
|
+
# @param where [Hash{String, Symbol => Object}, nil] column => condition, see Filter
|
|
101
|
+
# @param from [Integer, nil] number of rows at the start of the file to skip
|
|
102
|
+
# @param limit [Integer, nil] maximum number of rows to yield in total
|
|
103
|
+
# @yield [batch] once per batch
|
|
104
|
+
# @yieldparam batch [Array<Hash>, Hash{String, Symbol => Array}, Hash{String, Symbol => Numo::NArray}]
|
|
105
|
+
# row Hashes (+:rows+), or field name => values of the batch (+:columns+, +:numo+)
|
|
106
|
+
# @yieldreturn [void]
|
|
107
|
+
# @return [Reader, Enumerator] self, or an Enumerator of batches when no block is given
|
|
108
|
+
# @raise [ArgumentError] for a non-positive +size+, unknown +as+, negative +from+ / +limit+,
|
|
109
|
+
# or an unknown column in +columns+ or +where+
|
|
110
|
+
# @raise [UnsupportedError] with +as: :numo+ when no Numo gem can be loaded
|
|
111
|
+
def each_batch(size = DEFAULT_BATCH_SIZE, columns: nil, as: :rows, where: nil, from: nil, limit: nil)
|
|
112
|
+
return enum_for(:each_batch, size, columns: columns, as: as, where: where, from: from, limit: limit) unless block_given?
|
|
113
|
+
size = Integer(size)
|
|
114
|
+
raise ArgumentError, "Batch size must be positive, got #{size}" unless size.positive?
|
|
115
|
+
raise ArgumentError, "as: must be :rows, :columns or :numo, got #{as.inspect}" unless AS_MODES.include?(as)
|
|
116
|
+
raise ArgumentError, "limit: must not be negative" if limit&.negative?
|
|
117
|
+
return validate_read_options(columns, where, from) if limit&.zero?
|
|
118
|
+
if as == :numo
|
|
119
|
+
each_numo_batch(size, columns, where, from, limit) { |batch| yield batch }
|
|
120
|
+
return self
|
|
121
|
+
end
|
|
122
|
+
columnar = as == :columns
|
|
123
|
+
symbolize = @symbolize
|
|
124
|
+
out_fields = select_fields(columns)
|
|
125
|
+
filter = (where && !where.empty?) ? Filter.new(@schema, where) : nil
|
|
126
|
+
fields = filter ? out_fields | filter.fields : out_fields
|
|
127
|
+
names = row_keys(out_fields, symbolize)
|
|
128
|
+
nout = out_fields.size
|
|
129
|
+
converters = @schema.columns.map { |col| converter_for(col) }
|
|
130
|
+
assemblers = fields.map { |f| Assembler.new(f, symbolize) }
|
|
131
|
+
left_to_yield = limit
|
|
132
|
+
pending = pending_rows = nil
|
|
133
|
+
emit = lambda do |data, k|
|
|
134
|
+
if columnar
|
|
135
|
+
if pending
|
|
136
|
+
pending.each_with_index { |col, j| col.concat(data[j]) }
|
|
137
|
+
else
|
|
138
|
+
pending = data
|
|
139
|
+
end
|
|
140
|
+
pending_rows = (pending_rows || 0) + k
|
|
141
|
+
if pending_rows >= size
|
|
142
|
+
yield names.zip(pending).to_h
|
|
143
|
+
pending = pending_rows = nil
|
|
144
|
+
end
|
|
145
|
+
else
|
|
146
|
+
rows = build_rows(names, data, k)
|
|
147
|
+
if pending
|
|
148
|
+
pending.concat(rows)
|
|
149
|
+
else
|
|
150
|
+
pending = rows
|
|
151
|
+
end
|
|
152
|
+
pending_rows = pending.size
|
|
153
|
+
if pending_rows >= size
|
|
154
|
+
yield pending
|
|
155
|
+
pending = pending_rows = nil
|
|
156
|
+
end
|
|
157
|
+
end
|
|
26
158
|
end
|
|
159
|
+
|
|
160
|
+
plan_rows(filter, from).each do |rg_index, ranges|
|
|
161
|
+
rg = row_groups[rg_index]
|
|
162
|
+
partial = ranges != [[0, rg.num_rows]]
|
|
163
|
+
cursors = fields.map do |f|
|
|
164
|
+
f.leaves.map do |col|
|
|
165
|
+
reader = ColumnChunkReader.new(@io, rg.columns.fetch(col.index), col, converter: converters[col.index], lazy: true)
|
|
166
|
+
reader.locations = page_index(rg_index, col)[1]&.page_locations if partial
|
|
167
|
+
[col.index, ColumnCursor.new(reader)]
|
|
168
|
+
end
|
|
169
|
+
end
|
|
170
|
+
ranges.each do |first, stop|
|
|
171
|
+
cursors.each { |cs| cs.each { |_, cursor| cursor.seek(first) } }
|
|
172
|
+
left = stop - first
|
|
173
|
+
while left.positive?
|
|
174
|
+
# Rows still missing from the current batch (pending_rows counts rows in both modes)
|
|
175
|
+
k = pending ? size - pending_rows : size
|
|
176
|
+
raise Error, "Internal error: batch needs #{k} rows" unless k.positive? # never loop without progress
|
|
177
|
+
k = left if left < k
|
|
178
|
+
data = assemblers.each_with_index.map do |asm, j|
|
|
179
|
+
asm.read_rows(k, cursors[j].to_h { |idx, cursor| [idx, cursor.take(k)] })
|
|
180
|
+
end
|
|
181
|
+
left -= k
|
|
182
|
+
kept = k
|
|
183
|
+
if filter
|
|
184
|
+
keep = filter.matching_rows(data, fields, k, symbolize)
|
|
185
|
+
kept = keep.size
|
|
186
|
+
next if kept.zero?
|
|
187
|
+
data = data.first(nout).map { |col| keep.map { |i| col[i] } } if kept < k
|
|
188
|
+
end
|
|
189
|
+
data = data.first(nout) if data.size > nout
|
|
190
|
+
if left_to_yield && kept >= left_to_yield
|
|
191
|
+
emit.call(data.map { |col| col.first(left_to_yield) }, left_to_yield)
|
|
192
|
+
yield(columnar ? names.zip(pending).to_h : pending) if pending
|
|
193
|
+
return self
|
|
194
|
+
end
|
|
195
|
+
left_to_yield -= kept if left_to_yield
|
|
196
|
+
emit.call(data, kept)
|
|
197
|
+
end
|
|
198
|
+
end
|
|
199
|
+
end
|
|
200
|
+
yield(columnar ? names.zip(pending).to_h : pending) if pending
|
|
201
|
+
self
|
|
27
202
|
end
|
|
28
203
|
|
|
29
|
-
#
|
|
30
|
-
#
|
|
31
|
-
|
|
32
|
-
|
|
33
|
-
|
|
34
|
-
|
|
35
|
-
|
|
204
|
+
# What a read with +where:+ / +from:+ would touch, without reading any data: an Array of
|
|
205
|
+
# { row_group:, rows:, ranges: [[first_row, end_row), ...] } for the row groups that are
|
|
206
|
+
# read. Row groups ruled out entirely are left out.
|
|
207
|
+
#
|
|
208
|
+
# @param where [Hash{String, Symbol => Object}, nil] column => condition, as in #each_batch
|
|
209
|
+
# @param from [Integer, nil] number of rows at the start of the file to skip
|
|
210
|
+
# @return [Array<Hash{Symbol => Object}>] +{row_group: Integer, rows: Integer,
|
|
211
|
+
# ranges: Array<Array(Integer, Integer)>}+ per row group that would be read
|
|
212
|
+
# @raise [ArgumentError] for a negative +from+ or an unknown column in +where+
|
|
213
|
+
def scan_plan(where: nil, from: nil)
|
|
214
|
+
filter = (where && !where.empty?) ? Filter.new(@schema, where) : nil
|
|
215
|
+
plan_rows(filter, from).map do |rg_index, ranges|
|
|
216
|
+
{row_group: rg_index, rows: ranges.sum { |s, e| e - s }, ranges: ranges}
|
|
36
217
|
end
|
|
37
|
-
@metadata = read_footer
|
|
38
|
-
@schema = Schema.from_elements(@metadata.schema)
|
|
39
218
|
end
|
|
40
219
|
|
|
41
|
-
|
|
42
|
-
|
|
220
|
+
# Yields each row as a Hash of top-level field name => value. Takes the options of each_batch
|
|
221
|
+
# except as:.
|
|
222
|
+
#
|
|
223
|
+
# @param columns [Array<String, Symbol>, String, Symbol, nil] top-level fields to return;
|
|
224
|
+
# nil returns all of them
|
|
225
|
+
# @param where [Hash{String, Symbol => Object}, nil] column => condition, see Filter
|
|
226
|
+
# @param from [Integer, nil] number of rows at the start of the file to skip
|
|
227
|
+
# @param limit [Integer, nil] maximum number of rows to yield
|
|
228
|
+
# @yield [row] once per row
|
|
229
|
+
# @yieldparam row [Hash{String, Symbol => Object}] top-level field name => value
|
|
230
|
+
# @yieldreturn [void]
|
|
231
|
+
# @return [Reader, Enumerator] self, or an Enumerator of rows when no block is given
|
|
232
|
+
# @raise [ArgumentError] for a negative +from+ / +limit+ or an unknown column
|
|
233
|
+
def each_row(columns: nil, where: nil, from: nil, limit: nil, &block)
|
|
234
|
+
return enum_for(:each_row, columns: columns, where: where, from: from, limit: limit) unless block
|
|
235
|
+
each_batch(columns: columns, where: where, from: from, limit: limit) { |rows| rows.each(&block) }
|
|
236
|
+
self
|
|
43
237
|
end
|
|
44
238
|
|
|
45
|
-
|
|
46
|
-
|
|
239
|
+
# Reads the whole file (or the selected rows) at once: an Array of row Hashes, or with
|
|
240
|
+
# as: :columns a Hash of top-level field name => Array of values, or with as: :numo a Hash of
|
|
241
|
+
# top-level field name => Numo array. Takes the options of each_batch.
|
|
242
|
+
#
|
|
243
|
+
# @param columns [Array<String, Symbol>, String, Symbol, nil] top-level fields to return;
|
|
244
|
+
# nil returns all of them
|
|
245
|
+
# @param as [Symbol] +:rows+, +:columns+ or +:numo+, the shape of the result
|
|
246
|
+
# @param where [Hash{String, Symbol => Object}, nil] column => condition, see Filter
|
|
247
|
+
# @param from [Integer, nil] number of rows at the start of the file to skip
|
|
248
|
+
# @param limit [Integer, nil] maximum number of rows to return
|
|
249
|
+
# @return [Array<Hash>, Hash{String, Symbol => Array}, Hash{String, Symbol => Numo::NArray}]
|
|
250
|
+
# row Hashes (+:rows+), or field name => all values (+:columns+, +:numo+)
|
|
251
|
+
# @raise [ArgumentError] for an unknown +as+, negative +from+ / +limit+ or an unknown column
|
|
252
|
+
# @raise [UnsupportedError] with +as: :numo+ when no Numo gem can be loaded
|
|
253
|
+
def read(columns: nil, as: :rows, where: nil, from: nil, limit: nil)
|
|
254
|
+
if as == :numo
|
|
255
|
+
raise ArgumentError, "limit: must not be negative" if limit&.negative?
|
|
256
|
+
out = nil
|
|
257
|
+
if limit&.zero?
|
|
258
|
+
validate_read_options(columns, where, from)
|
|
259
|
+
else
|
|
260
|
+
# One batch (num_rows is not trusted: some writers store 0), so the column types are
|
|
261
|
+
# decided from all the rows read
|
|
262
|
+
each_numo_batch(1 << 62, columns, where, from, limit) { |batch| out = batch }
|
|
263
|
+
end
|
|
264
|
+
out ||= numo_empty(columns)
|
|
265
|
+
elsif as == :columns
|
|
266
|
+
out = row_keys(select_fields(columns), @symbolize).to_h { |name| [name, []] }
|
|
267
|
+
each_batch(65_536, columns: columns, as: :columns, where: where, from: from, limit: limit) do |batch|
|
|
268
|
+
batch.each { |name, values| out[name].concat(values) }
|
|
269
|
+
end
|
|
270
|
+
else
|
|
271
|
+
out = []
|
|
272
|
+
each_batch(columns: columns, as: as, where: where, from: from, limit: limit) { |rows| out.concat(rows) }
|
|
273
|
+
end
|
|
274
|
+
out
|
|
47
275
|
end
|
|
48
276
|
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
|
|
277
|
+
# Internal (used by reads with where:/from:): [ColumnIndex or nil, OffsetIndex or nil] of a
|
|
278
|
+
# leaf column in a row group. Cached per chunk; a missing or damaged index reads as nil.
|
|
279
|
+
#
|
|
280
|
+
# @param row_group_index [Integer] position of the row group in the footer
|
|
281
|
+
# @param column [Schema::Column, String, Array<String>] leaf column, or its dotted / Array path
|
|
282
|
+
# @return [Array(Format::ColumnIndex, Format::OffsetIndex)] either element may be nil
|
|
283
|
+
# @raise [ArgumentError] when +column+ does not name a leaf column
|
|
284
|
+
# @raise [IndexError] when the row group does not exist
|
|
285
|
+
def page_index(row_group_index, column)
|
|
286
|
+
column = @schema.column(column) unless column.is_a?(Schema::Column)
|
|
287
|
+
raise ArgumentError, "No such leaf column" unless column
|
|
288
|
+
@page_indexes ||= {}
|
|
289
|
+
@page_indexes[[row_group_index, column.index]] ||= begin
|
|
290
|
+
chunk = row_groups.fetch(row_group_index).columns.fetch(column.index)
|
|
291
|
+
[read_struct(Format::ColumnIndex, chunk.column_index_offset, chunk.column_index_length),
|
|
292
|
+
read_struct(Format::OffsetIndex, chunk.offset_index_offset, chunk.offset_index_length)]
|
|
293
|
+
end
|
|
294
|
+
end
|
|
53
295
|
|
|
54
|
-
|
|
55
|
-
|
|
296
|
+
# Short summary for the console, without the schema
|
|
297
|
+
#
|
|
298
|
+
# @return [String] row count, row group count and the writer's created_by
|
|
299
|
+
def inspect
|
|
300
|
+
"#<#{self.class.name} rows=#{num_rows} row_groups=#{row_groups.size} created_by=#{@file_metadata.created_by.inspect}>"
|
|
56
301
|
end
|
|
57
302
|
|
|
58
|
-
|
|
59
|
-
|
|
60
|
-
|
|
61
|
-
|
|
62
|
-
|
|
63
|
-
|
|
64
|
-
|
|
65
|
-
|
|
66
|
-
|
|
67
|
-
|
|
68
|
-
|
|
69
|
-
|
|
303
|
+
private
|
|
304
|
+
|
|
305
|
+
# as: :numo. Flat numeric/boolean output columns go through NumoCursors (no Ruby object per
|
|
306
|
+
# value); the other output columns, and every column a where: filter needs, are assembled as
|
|
307
|
+
# Ruby values like as: :columns and converted when a batch is complete.
|
|
308
|
+
#
|
|
309
|
+
# @param size [Integer] maximum number of rows per batch
|
|
310
|
+
# @param columns [Array<String, Symbol>, String, Symbol, nil] top-level fields to return
|
|
311
|
+
# @param where [Hash{String, Symbol => Object}, nil] column => condition, see Filter
|
|
312
|
+
# @param from [Integer, nil] number of rows at the start of the file to skip
|
|
313
|
+
# @param limit [Integer, nil] maximum number of rows to yield in total
|
|
314
|
+
# @yield [batch] once per batch
|
|
315
|
+
# @yieldparam batch [Hash{String, Symbol => Numo::NArray}] field name => values of the batch
|
|
316
|
+
# @yieldreturn [void]
|
|
317
|
+
# @return [void]
|
|
318
|
+
# @raise [UnsupportedError] when no Numo gem can be loaded
|
|
319
|
+
def each_numo_batch(size, columns, where, from, limit)
|
|
320
|
+
NumoColumns.load!
|
|
321
|
+
symbolize = @symbolize
|
|
322
|
+
out_fields = select_fields(columns)
|
|
323
|
+
filter = (where && !where.empty?) ? Filter.new(@schema, where) : nil
|
|
324
|
+
filter_fields = filter ? filter.fields : []
|
|
325
|
+
names = row_keys(out_fields, symbolize)
|
|
326
|
+
specs = out_fields.map { |f| NumoColumns.spec_for(f) }
|
|
327
|
+
fast = out_fields.each_index.select { |j| specs[j].fast? && !filter_fields.include?(out_fields[j]) }
|
|
328
|
+
ruby_fields = (out_fields.each_index.to_a - fast).map { |j| out_fields[j] } | filter_fields
|
|
329
|
+
# Where each output column comes from: index into the fast cursors or the Ruby fields
|
|
330
|
+
sources = out_fields.each_index.map { |j| (i = fast.index(j)) ? [:fast, i] : [:ruby, ruby_fields.index(out_fields[j])] }
|
|
331
|
+
converters = @schema.columns.map { |col| converter_for(col) }
|
|
332
|
+
assemblers = ruby_fields.map { |f| Assembler.new(f, symbolize) }
|
|
333
|
+
# Rows decoded per step: Ruby values are kept to slices of the batch, Numo arrays need not be
|
|
334
|
+
step = ruby_fields.empty? ? 1 << 20 : 65_536
|
|
335
|
+
left_to_yield = limit
|
|
336
|
+
pending = Array.new(out_fields.size) { [] }
|
|
337
|
+
pending_rows = 0
|
|
338
|
+
flush = lambda do
|
|
339
|
+
batch = sources.each_with_index.map do |(kind, _), j|
|
|
340
|
+
(kind == :fast) ? NumoColumns.finish_fixed(specs[j], pending[j]) : NumoColumns.finish_values(specs[j], pending[j])
|
|
70
341
|
end
|
|
342
|
+
pending = Array.new(out_fields.size) { [] }
|
|
343
|
+
pending_rows = 0
|
|
344
|
+
yield names.zip(batch).to_h
|
|
71
345
|
end
|
|
72
|
-
self
|
|
73
|
-
end
|
|
74
|
-
alias_method :each, :each_row
|
|
75
346
|
|
|
76
|
-
|
|
77
|
-
|
|
347
|
+
plan_rows(filter, from).each do |rg_index, ranges|
|
|
348
|
+
rg = row_groups[rg_index]
|
|
349
|
+
partial = ranges != [[0, rg.num_rows]]
|
|
350
|
+
open = lambda do |col|
|
|
351
|
+
reader = ColumnChunkReader.new(@io, rg.columns.fetch(col.index), col, converter: converters[col.index], lazy: true)
|
|
352
|
+
reader.locations = page_index(rg_index, col)[1]&.page_locations if partial
|
|
353
|
+
reader
|
|
354
|
+
end
|
|
355
|
+
ruby_cursors = ruby_fields.map { |f| f.leaves.map { |col| [col.index, ColumnCursor.new(open.call(col))] } }
|
|
356
|
+
fast_cursors = fast.map { |j| NumoCursor.new(open.call(out_fields[j].column), specs[j]) }
|
|
357
|
+
ranges.each do |first, stop|
|
|
358
|
+
ruby_cursors.each { |cs| cs.each { |_, cursor| cursor.seek(first) } }
|
|
359
|
+
fast_cursors.each { |cursor| cursor.seek(first) }
|
|
360
|
+
left = stop - first
|
|
361
|
+
while left.positive?
|
|
362
|
+
k = size - pending_rows
|
|
363
|
+
raise Error, "Internal error: batch needs #{k} rows" unless k.positive? # never loop without progress
|
|
364
|
+
k = step if k > step
|
|
365
|
+
k = left if left < k
|
|
366
|
+
data = assemblers.each_with_index.map do |asm, j|
|
|
367
|
+
asm.read_rows(k, ruby_cursors[j].to_h { |idx, cursor| [idx, cursor.take(k)] })
|
|
368
|
+
end
|
|
369
|
+
numo = fast_cursors.map { |cursor| cursor.take(k) }
|
|
370
|
+
left -= k
|
|
371
|
+
kept = k
|
|
372
|
+
if filter
|
|
373
|
+
keep = filter.matching_rows(data, ruby_fields, k, symbolize)
|
|
374
|
+
kept = keep.size
|
|
375
|
+
next if kept.zero?
|
|
376
|
+
if kept < k
|
|
377
|
+
data = data.map { |col| keep.map { |i| col[i] } }
|
|
378
|
+
index = Numo::Int64.cast(keep)
|
|
379
|
+
numo = numo.map { |values, valid| [values[index].dup, valid && valid[index].dup] }
|
|
380
|
+
end
|
|
381
|
+
end
|
|
382
|
+
if left_to_yield && kept > left_to_yield
|
|
383
|
+
kept = left_to_yield
|
|
384
|
+
data = data.map { |col| col.first(kept) }
|
|
385
|
+
numo = numo.map { |values, valid| [values[0...kept].dup, valid && valid[0...kept].dup] }
|
|
386
|
+
end
|
|
387
|
+
sources.each_with_index do |(kind, i), j|
|
|
388
|
+
pending[j] << ((kind == :fast) ? numo[i] : data[i])
|
|
389
|
+
end
|
|
390
|
+
pending_rows += kept
|
|
391
|
+
left_to_yield -= kept if left_to_yield
|
|
392
|
+
flush.call if pending_rows >= size || left_to_yield&.zero?
|
|
393
|
+
return if left_to_yield&.zero? # standard:disable Lint/NonLocalExitFromIterator
|
|
394
|
+
end
|
|
395
|
+
end
|
|
396
|
+
end
|
|
397
|
+
flush.call if pending_rows.positive?
|
|
398
|
+
end
|
|
78
399
|
|
|
79
|
-
#
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
|
|
400
|
+
# Checks the columns:, where: and from: options of a read that returns no rows (limit: 0),
|
|
401
|
+
# so it raises for bad options like any other read
|
|
402
|
+
#
|
|
403
|
+
# @param columns [Array<String, Symbol>, String, Symbol, nil] top-level fields to return
|
|
404
|
+
# @param where [Hash{String, Symbol => Object}, nil] column => condition, see Filter
|
|
405
|
+
# @param from [Integer, nil] number of rows at the start of the file to skip
|
|
406
|
+
# @return [Reader] self
|
|
407
|
+
# @raise [ArgumentError] for an unknown column, a bad condition or a negative +from+
|
|
408
|
+
def validate_read_options(columns, where, from)
|
|
409
|
+
select_fields(columns)
|
|
410
|
+
Filter.new(@schema, where) if where && !where.empty?
|
|
411
|
+
raise ArgumentError, "from: must not be negative" if Integer(from || 0).negative?
|
|
412
|
+
self
|
|
83
413
|
end
|
|
84
414
|
|
|
85
|
-
#
|
|
86
|
-
|
|
415
|
+
# read(as: :numo) of no rows: an empty array of each column's type
|
|
416
|
+
#
|
|
417
|
+
# @param columns [Array<String, Symbol>, String, Symbol, nil] top-level fields to return
|
|
418
|
+
# @return [Hash{String, Symbol => Numo::NArray}] field name => zero-length Numo array
|
|
419
|
+
def numo_empty(columns)
|
|
420
|
+
NumoColumns.load!
|
|
87
421
|
fields = select_fields(columns)
|
|
88
|
-
fields
|
|
422
|
+
row_keys(fields, @symbolize).zip(fields.map { |f|
|
|
423
|
+
spec = NumoColumns.spec_for(f)
|
|
424
|
+
(spec.kind == :fixed) ? spec.klass.new(0) : Numo::RObject.new(0)
|
|
425
|
+
}).to_h
|
|
89
426
|
end
|
|
90
427
|
|
|
91
|
-
#
|
|
92
|
-
#
|
|
93
|
-
|
|
94
|
-
|
|
95
|
-
|
|
96
|
-
|
|
97
|
-
|
|
428
|
+
# [[row_group_index, [[first_row, end_row), ...]], ...] to read, after ruling out row groups
|
|
429
|
+
# (statistics, bloom filters) and pages (page index), and skipping the first +from+ rows
|
|
430
|
+
#
|
|
431
|
+
# @param filter [Filter, nil] the where: conditions, if any
|
|
432
|
+
# @param from [Integer, nil] number of rows at the start of the file to skip
|
|
433
|
+
# @return [Array<Array(Integer, Array<Array(Integer, Integer)>)>] row group index and its
|
|
434
|
+
# half-open row ranges, for row groups that have rows left to read
|
|
435
|
+
# @raise [ArgumentError] for a negative +from+
|
|
436
|
+
def plan_rows(filter, from)
|
|
437
|
+
from = Integer(from || 0)
|
|
438
|
+
raise ArgumentError, "from: must not be negative" if from.negative?
|
|
439
|
+
offset = 0
|
|
440
|
+
plan = []
|
|
441
|
+
row_groups.each_with_index do |rg, i|
|
|
442
|
+
n = rg.num_rows
|
|
443
|
+
first = from - offset
|
|
444
|
+
offset += n
|
|
445
|
+
next if n.zero? || first >= n
|
|
446
|
+
ranges = [[first.positive? ? first : 0, n]]
|
|
447
|
+
if filter
|
|
448
|
+
next unless filter.row_group_may_match?(self, i)
|
|
449
|
+
ranges = Filter.intersect(ranges, filter.page_ranges(self, i))
|
|
450
|
+
end
|
|
451
|
+
plan << [i, ranges] unless ranges.empty?
|
|
452
|
+
end
|
|
453
|
+
plan
|
|
98
454
|
end
|
|
99
455
|
|
|
100
|
-
|
|
101
|
-
|
|
456
|
+
# Decodes one Thrift struct stored elsewhere in the file (ColumnIndex, OffsetIndex)
|
|
457
|
+
#
|
|
458
|
+
# @param klass [Class] Format struct class to decode with (responds to .decode)
|
|
459
|
+
# @param offset [Integer, nil] file offset of the struct
|
|
460
|
+
# @param length [Integer, nil] byte length of the struct
|
|
461
|
+
# @return [Object, nil] the decoded +klass+ instance, or nil when absent, truncated or corrupt
|
|
462
|
+
def read_struct(klass, offset, length)
|
|
463
|
+
return nil unless offset && length&.positive?
|
|
464
|
+
@io.seek(offset)
|
|
465
|
+
bytes = @io.read(length)
|
|
466
|
+
return nil unless bytes&.bytesize == length
|
|
467
|
+
klass.decode(bytes.b).first
|
|
468
|
+
rescue Thrift::Error
|
|
469
|
+
nil # a damaged index only means pages cannot be skipped
|
|
102
470
|
end
|
|
103
471
|
|
|
104
|
-
|
|
105
|
-
|
|
472
|
+
# The top-level fields named by a +columns:+ option
|
|
473
|
+
#
|
|
474
|
+
# @param columns [Array<String, Symbol>, String, Symbol, nil] field names; nil selects all
|
|
475
|
+
# @return [Array<Schema::Field>] the fields, in the order requested
|
|
476
|
+
# @raise [ArgumentError] when a name is not a top-level field
|
|
106
477
|
def select_fields(columns)
|
|
107
478
|
return @schema.fields unless columns
|
|
108
479
|
Array(columns).map { |c| @schema.field(c) or raise ArgumentError, "No such column #{c.inspect}" }
|
|
109
480
|
end
|
|
110
481
|
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
|
|
114
|
-
|
|
115
|
-
|
|
116
|
-
|
|
117
|
-
|
|
482
|
+
# Hash keys for the given fields (frozen, deduplicated Strings, or Symbols)
|
|
483
|
+
#
|
|
484
|
+
# @param fields [Array<Schema::Field>] top-level fields being returned
|
|
485
|
+
# @param symbolize [Boolean] whether to key by Symbol instead of String
|
|
486
|
+
# @return [Array<String>, Array<Symbol>] one key per field
|
|
487
|
+
def row_keys(fields, symbolize)
|
|
488
|
+
fields.map { |f| symbolize ? f.name.to_sym : -f.name }
|
|
489
|
+
end
|
|
490
|
+
|
|
491
|
+
# Turns column-wise batch data into row Hashes
|
|
492
|
+
#
|
|
493
|
+
# @param names [Array<String>, Array<Symbol>] Hash key per output field
|
|
494
|
+
# @param data [Array<Array>] values per output field, each holding +k+ entries
|
|
495
|
+
# @param k [Integer] number of rows in the batch
|
|
496
|
+
# @return [Array<Hash>] +k+ row Hashes
|
|
497
|
+
def build_rows(names, data, k)
|
|
498
|
+
return Array.new(k) { {} } if names.empty?
|
|
499
|
+
return data.first.map { |v| {names.first => v} } if names.size == 1
|
|
500
|
+
nf = names.size
|
|
501
|
+
Array.new(k) do |i|
|
|
502
|
+
row = {}
|
|
503
|
+
j = 0
|
|
504
|
+
while j < nf
|
|
505
|
+
row[names[j]] = data[j][i]
|
|
506
|
+
j += 1
|
|
118
507
|
end
|
|
119
|
-
|
|
508
|
+
row
|
|
120
509
|
end
|
|
121
510
|
end
|
|
122
511
|
|
|
512
|
+
# The column's value converter, with the time zone applied to timestamps
|
|
513
|
+
#
|
|
514
|
+
# @param column [Schema::Column] leaf column being read
|
|
515
|
+
# @return [Proc, nil] physical value => Ruby value, or nil when values are used as decoded
|
|
516
|
+
def converter_for(column)
|
|
517
|
+
base = column.converter
|
|
518
|
+
zc = @zone_converter
|
|
519
|
+
return base unless zc && base && instant_column?(column)
|
|
520
|
+
->(v) { zc.call(base.call(v)) }
|
|
521
|
+
end
|
|
522
|
+
|
|
523
|
+
# Timestamps that denote an instant (UTC-adjusted TIMESTAMP, INT96). Local timestamps
|
|
524
|
+
# (isAdjustedToUTC = false) are wall-clock values and are left as they are.
|
|
525
|
+
#
|
|
526
|
+
# @param column [Schema::Column] leaf column to check
|
|
527
|
+
# @return [Boolean] true when the time zone applies to this column's values
|
|
528
|
+
def instant_column?(column)
|
|
529
|
+
return true if column.type == Format::Type::INT96
|
|
530
|
+
kind, _unit, utc = Types.logical_of(column.node)
|
|
531
|
+
kind == :timestamp && utc
|
|
532
|
+
end
|
|
533
|
+
|
|
534
|
+
# A UTC offset string: "+02", "+0200", "+02:00" or "+02:00:00" (sign required)
|
|
535
|
+
OFFSET_PATTERN = /\A([+-])(\d\d)(?::?(\d\d)(?::?(\d\d))?)?\z/
|
|
536
|
+
|
|
537
|
+
# A lambda turning a UTC Time into the zone, or nil for UTC
|
|
538
|
+
#
|
|
539
|
+
# @param zone [String, Integer, Object, nil] the +time_zone:+ option: a UTC offset (String or
|
|
540
|
+
# seconds), a zone name (ActiveSupport or TZInfo), an object responding to #at or
|
|
541
|
+
# #utc_to_local, or nil
|
|
542
|
+
# @return [Proc, nil] Time => Time (or ActiveSupport::TimeWithZone), nil for UTC
|
|
543
|
+
# @raise [ArgumentError] for an unknown, unsupported or out-of-range zone
|
|
544
|
+
def zone_converter(zone)
|
|
545
|
+
case zone
|
|
546
|
+
when nil then nil
|
|
547
|
+
when Integer
|
|
548
|
+
return nil if zone.zero?
|
|
549
|
+
Time.at(0).getlocal(zone) # validates the range
|
|
550
|
+
->(t) { t.getlocal(zone) }
|
|
551
|
+
when String
|
|
552
|
+
return nil if zone.match?(/\A(?:utc|z)\z/i)
|
|
553
|
+
if (m = OFFSET_PATTERN.match(zone))
|
|
554
|
+
# As seconds, since Ruby 3.0 only parses "+HH:MM" offset strings
|
|
555
|
+
return zone_converter(((m[1] == "-") ? -1 : 1) * (m[2].to_i * 3600 + m[3].to_i * 60 + m[4].to_i))
|
|
556
|
+
end
|
|
557
|
+
if defined?(::ActiveSupport::TimeZone) && (tz = ::ActiveSupport::TimeZone[zone])
|
|
558
|
+
return ->(t) { tz.at(t) }
|
|
559
|
+
end
|
|
560
|
+
if defined?(::TZInfo::Timezone)
|
|
561
|
+
tz = ::TZInfo::Timezone.get(zone)
|
|
562
|
+
return ->(t) { t.getlocal(tz) }
|
|
563
|
+
end
|
|
564
|
+
raise ArgumentError, "Unknown time zone #{zone.inspect}: use a UTC offset like \"+02:00\", " \
|
|
565
|
+
"a timezone object (e.g. TZInfo::Timezone.get(#{zone.inspect})) or an ActiveSupport::TimeZone"
|
|
566
|
+
else
|
|
567
|
+
return ->(t) { zone.at(t) } if zone.respond_to?(:at)
|
|
568
|
+
if zone.respond_to?(:utc_to_local)
|
|
569
|
+
Time.at(0).getlocal(zone)
|
|
570
|
+
return ->(t) { t.getlocal(zone) }
|
|
571
|
+
end
|
|
572
|
+
raise ArgumentError, "Unsupported time_zone: #{zone.inspect}"
|
|
573
|
+
end
|
|
574
|
+
rescue ArgumentError => e
|
|
575
|
+
raise if e.message.start_with?("Unknown time zone", "Unsupported time_zone", "Invalid time_zone")
|
|
576
|
+
raise ArgumentError, "Invalid time_zone #{zone.inspect}: #{e.message}"
|
|
577
|
+
end
|
|
578
|
+
|
|
579
|
+
# Reads and decodes the footer: the FileMetaData Thrift struct, its 4-byte little-endian
|
|
580
|
+
# length and the closing magic.
|
|
581
|
+
#
|
|
582
|
+
# @return [Format::FileMetaData] the decoded footer
|
|
583
|
+
# @raise [FormatError] when the file is too short, lacks the magic or the footer is corrupt
|
|
584
|
+
# @raise [UnsupportedError] for an encrypted file (+PARE+ magic)
|
|
123
585
|
def read_footer
|
|
124
586
|
@io.seek(0, IO::SEEK_END)
|
|
125
587
|
size = @io.pos
|
|
126
588
|
raise FormatError, "File too small to be Parquet (#{size} bytes)" if size < 12
|
|
127
589
|
@io.seek(size - 8)
|
|
128
590
|
tail = @io.read(8)
|
|
129
|
-
raise FormatError, "Missing PAR1 footer magic" unless tail.byteslice(4, 4) == MAGIC
|
|
130
591
|
raise UnsupportedError, "Encrypted Parquet files are not supported" if tail.byteslice(4, 4) == "PARE"
|
|
592
|
+
raise FormatError, "Missing PAR1 footer magic" unless tail.byteslice(4, 4) == MAGIC
|
|
131
593
|
footer_len = tail.unpack1("V")
|
|
132
594
|
raise FormatError, "Footer length #{footer_len} exceeds file size" if footer_len + 12 > size
|
|
133
595
|
@io.seek(size - 8 - footer_len)
|
|
@@ -137,199 +599,26 @@ module Herringbone
|
|
|
137
599
|
raise FormatError, "Corrupt file metadata: #{e.message}"
|
|
138
600
|
end
|
|
139
601
|
|
|
140
|
-
# Decodes all pages of one column chunk into levels and values
|
|
141
|
-
class ColumnChunkReader
|
|
142
|
-
T = Format::Type
|
|
143
|
-
E = Format::Encoding
|
|
144
|
-
|
|
145
|
-
def initialize(io, chunk, column)
|
|
146
|
-
@io = io
|
|
147
|
-
@chunk = chunk
|
|
148
|
-
@column = column
|
|
149
|
-
@meta = chunk.meta_data or raise UnsupportedError, "Column chunk without metadata (encrypted?)"
|
|
150
|
-
raise UnsupportedError, "Column chunks in external files are not supported" if chunk.file_path
|
|
151
|
-
@max_def = column.max_definition_level
|
|
152
|
-
@max_rep = column.max_repetition_level
|
|
153
|
-
@converter = column.converter
|
|
154
|
-
end
|
|
155
|
-
|
|
156
|
-
def read
|
|
157
|
-
buf = read_bytes
|
|
158
|
-
defs = @max_def.positive? ? [] : nil
|
|
159
|
-
reps = @max_rep.positive? ? [] : nil
|
|
160
|
-
values = []
|
|
161
|
-
pos = 0
|
|
162
|
-
total = @meta.num_values
|
|
163
|
-
seen = 0
|
|
164
|
-
while seen < total && pos < buf.bytesize
|
|
165
|
-
header, pos = decode_page_header(buf, pos)
|
|
166
|
-
size = header.compressed_page_size
|
|
167
|
-
# Some old writers under-report total_compressed_size; read on past the declared end
|
|
168
|
-
extend_buffer(buf, pos + size - buf.bytesize) if pos + size > buf.bytesize
|
|
169
|
-
raise FormatError, "Page overruns column chunk" if pos + size > buf.bytesize
|
|
170
|
-
body = buf.byteslice(pos, size)
|
|
171
|
-
pos += size
|
|
172
|
-
case header.type
|
|
173
|
-
when Format::PageType::DICTIONARY_PAGE
|
|
174
|
-
read_dictionary(header, body)
|
|
175
|
-
when Format::PageType::DATA_PAGE
|
|
176
|
-
seen += read_data_page_v1(header, body, defs, reps, values)
|
|
177
|
-
when Format::PageType::DATA_PAGE_V2
|
|
178
|
-
seen += read_data_page_v2(header, body, defs, reps, values)
|
|
179
|
-
end
|
|
180
|
-
end
|
|
181
|
-
raise FormatError, "Column #{@column.dotted_path}: read #{seen} of #{total} values" if seen < total
|
|
182
|
-
[defs, reps, values]
|
|
183
|
-
rescue Thrift::Error => e
|
|
184
|
-
raise FormatError, "Corrupt page header in #{@column.dotted_path}: #{e.message}"
|
|
185
|
-
end
|
|
186
|
-
|
|
187
|
-
private
|
|
188
|
-
|
|
189
|
-
def read_bytes
|
|
190
|
-
start = @meta.data_page_offset
|
|
191
|
-
dict = @meta.dictionary_page_offset
|
|
192
|
-
# Some writers store 0 when there is no dictionary page
|
|
193
|
-
start = dict if dict && dict.positive? && dict < start
|
|
194
|
-
@start = start
|
|
195
|
-
@io.seek(start)
|
|
196
|
-
len = @meta.total_compressed_size
|
|
197
|
-
(@io.read(len) || "".b).b
|
|
198
|
-
end
|
|
199
|
-
|
|
200
|
-
def extend_buffer(buf, nbytes)
|
|
201
|
-
@io.seek(@start + buf.bytesize)
|
|
202
|
-
more = @io.read(nbytes)
|
|
203
|
-
buf << more.b if more
|
|
204
|
-
end
|
|
205
|
-
|
|
206
|
-
def decode_page_header(buf, pos)
|
|
207
|
-
Format::PageHeader.decode(buf, pos)
|
|
208
|
-
rescue Thrift::Error
|
|
209
|
-
# The header may straddle the declared end of the chunk
|
|
210
|
-
before = buf.bytesize
|
|
211
|
-
extend_buffer(buf, 1024)
|
|
212
|
-
raise if buf.bytesize == before
|
|
213
|
-
Format::PageHeader.decode(buf, pos)
|
|
214
|
-
end
|
|
215
|
-
|
|
216
|
-
def decompress(body, size)
|
|
217
|
-
Compression.decompress(@meta.codec, body, size)
|
|
218
|
-
end
|
|
219
|
-
|
|
220
|
-
def read_dictionary(header, body)
|
|
221
|
-
dh = header.dictionary_page_header
|
|
222
|
-
data = decompress(body, header.uncompressed_page_size)
|
|
223
|
-
vals, = Encodings::Plain.decode(data, 0, dh.num_values, @column.type, @column.type_length)
|
|
224
|
-
vals.map!(&@converter) if @converter
|
|
225
|
-
@dictionary = vals
|
|
226
|
-
end
|
|
227
|
-
|
|
228
|
-
def read_data_page_v1(header, body, defs, reps, values)
|
|
229
|
-
dh = header.data_page_header
|
|
230
|
-
n = dh.num_values
|
|
231
|
-
data = decompress(body, header.uncompressed_page_size)
|
|
232
|
-
pos = 0
|
|
233
|
-
if @max_rep.positive?
|
|
234
|
-
levels, pos = read_levels(data, pos, dh.repetition_level_encoding, @max_rep, n)
|
|
235
|
-
reps.concat(levels)
|
|
236
|
-
end
|
|
237
|
-
non_null = n
|
|
238
|
-
if @max_def.positive?
|
|
239
|
-
levels, pos = read_levels(data, pos, dh.definition_level_encoding, @max_def, n)
|
|
240
|
-
non_null = levels.count(@max_def)
|
|
241
|
-
defs.concat(levels)
|
|
242
|
-
end
|
|
243
|
-
values.concat(decode_values(data, pos, non_null, dh.encoding))
|
|
244
|
-
n
|
|
245
|
-
end
|
|
246
|
-
|
|
247
|
-
def read_data_page_v2(header, body, defs, reps, values)
|
|
248
|
-
dh = header.data_page_header_v2
|
|
249
|
-
n = dh.num_values
|
|
250
|
-
rep_len = dh.repetition_levels_byte_length
|
|
251
|
-
def_len = dh.definition_levels_byte_length
|
|
252
|
-
if @max_rep.positive?
|
|
253
|
-
reps.concat(Encodings::RLE.decode_hybrid(body, 0, rep_len, RLE_WIDTH[@max_rep], n))
|
|
254
|
-
end
|
|
255
|
-
non_null = n
|
|
256
|
-
if @max_def.positive?
|
|
257
|
-
levels = Encodings::RLE.decode_hybrid(body, rep_len, rep_len + def_len, RLE_WIDTH[@max_def], n)
|
|
258
|
-
non_null = levels.count(@max_def)
|
|
259
|
-
defs.concat(levels)
|
|
260
|
-
end
|
|
261
|
-
data = body.byteslice(rep_len + def_len, body.bytesize - rep_len - def_len)
|
|
262
|
-
if dh.is_compressed != false
|
|
263
|
-
data = decompress(data, header.uncompressed_page_size - rep_len - def_len)
|
|
264
|
-
end
|
|
265
|
-
values.concat(decode_values(data, 0, non_null, dh.encoding))
|
|
266
|
-
n
|
|
267
|
-
end
|
|
268
|
-
|
|
269
|
-
RLE_WIDTH = Hash.new { |h, k| h[k] = k.bit_length }
|
|
270
|
-
|
|
271
|
-
def read_levels(data, pos, encoding, max, n)
|
|
272
|
-
width = RLE_WIDTH[max]
|
|
273
|
-
case encoding
|
|
274
|
-
when E::RLE
|
|
275
|
-
len = data.byteslice(pos, 4).unpack1("V")
|
|
276
|
-
start = pos + 4
|
|
277
|
-
[Encodings::RLE.decode_hybrid(data, start, start + len, width, n), start + len]
|
|
278
|
-
when E::BIT_PACKED
|
|
279
|
-
[Encodings::RLE.decode_legacy_bit_packed(data, pos, width, n), pos + (n * width + 7) / 8]
|
|
280
|
-
else
|
|
281
|
-
raise UnsupportedError, "Unsupported level encoding #{E::NAMES[encoding] || encoding}"
|
|
282
|
-
end
|
|
283
|
-
end
|
|
284
|
-
|
|
285
|
-
def decode_values(data, pos, count, encoding)
|
|
286
|
-
type = @column.type
|
|
287
|
-
vals = case encoding
|
|
288
|
-
when E::PLAIN
|
|
289
|
-
Encodings::Plain.decode(data, pos, count, type, @column.type_length).first
|
|
290
|
-
when E::PLAIN_DICTIONARY, E::RLE_DICTIONARY
|
|
291
|
-
raise FormatError, "Dictionary-encoded page without a dictionary in #{@column.dotted_path}" unless @dictionary
|
|
292
|
-
return [] if count.zero?
|
|
293
|
-
width = data.getbyte(pos)
|
|
294
|
-
indices = Encodings::RLE.decode_hybrid(data, pos + 1, data.bytesize, width, count)
|
|
295
|
-
dict = @dictionary
|
|
296
|
-
raise FormatError, "Dictionary index out of range in #{@column.dotted_path}" if indices.max >= dict.size
|
|
297
|
-
return indices.map! { |i| dict[i] }
|
|
298
|
-
when E::RLE
|
|
299
|
-
raise UnsupportedError, "RLE value encoding is only supported for BOOLEAN" unless type == T::BOOLEAN
|
|
300
|
-
len = data.byteslice(pos, 4).unpack1("V")
|
|
301
|
-
Encodings::RLE.decode_hybrid(data, pos + 4, pos + 4 + len, 1, count).map! { |v| v == 1 }
|
|
302
|
-
when E::DELTA_BINARY_PACKED
|
|
303
|
-
bits = type == T::INT32 ? 32 : 64
|
|
304
|
-
vals, = Encodings::Delta.decode_binary_packed(data, pos, bits, count)
|
|
305
|
-
raise FormatError, "DELTA_BINARY_PACKED page has #{vals.size} values, need #{count}" if vals.size < count
|
|
306
|
-
vals
|
|
307
|
-
when E::DELTA_LENGTH_BYTE_ARRAY
|
|
308
|
-
Encodings::Delta.decode_length_byte_array(data, pos, count).first
|
|
309
|
-
when E::DELTA_BYTE_ARRAY
|
|
310
|
-
Encodings::Delta.decode_byte_array(data, pos, count).first
|
|
311
|
-
when E::BYTE_STREAM_SPLIT
|
|
312
|
-
width = case type
|
|
313
|
-
when T::INT32, T::FLOAT then 4
|
|
314
|
-
when T::INT64, T::DOUBLE then 8
|
|
315
|
-
when T::FIXED_LEN_BYTE_ARRAY then @column.type_length
|
|
316
|
-
else raise UnsupportedError, "BYTE_STREAM_SPLIT is not valid for #{T::NAMES[type]}"
|
|
317
|
-
end
|
|
318
|
-
plain, = Encodings::ByteStreamSplit.decode(data, pos, count, width)
|
|
319
|
-
Encodings::Plain.decode(plain, 0, count, type, @column.type_length).first
|
|
320
|
-
else
|
|
321
|
-
raise UnsupportedError, "Unsupported encoding #{E::NAMES[encoding] || encoding}"
|
|
322
|
-
end
|
|
323
|
-
vals.map!(&@converter) if @converter
|
|
324
|
-
vals
|
|
325
|
-
end
|
|
326
|
-
end
|
|
327
|
-
|
|
328
602
|
# Rebuilds nested values of one top-level field from the levels of its leaf columns
|
|
329
603
|
# (the "record assembly" half of the Dremel algorithm).
|
|
330
604
|
class Assembler
|
|
331
|
-
|
|
605
|
+
# @param field [Schema::Field] top-level field to assemble
|
|
606
|
+
# @param symbolize [Boolean] whether struct Hashes are keyed by Symbol instead of String
|
|
607
|
+
def initialize(field, symbolize = false)
|
|
332
608
|
@field = field
|
|
609
|
+
@symbolize = symbolize
|
|
610
|
+
@keys = {} # struct child Field => Hash key
|
|
611
|
+
end
|
|
612
|
+
|
|
613
|
+
# Assembles +n+ values of the field from +chunks+ (leaf column index =>
|
|
614
|
+
# [defs, reps, values] holding exactly those rows)
|
|
615
|
+
#
|
|
616
|
+
# @param n [Integer] number of rows in +chunks+
|
|
617
|
+
# @param chunks [Hash{Integer => Array(Array<Integer>, Array<Integer>, Array)}] leaf column
|
|
618
|
+
# index => [definition levels, repetition levels, values]; levels may be nil
|
|
619
|
+
# @return [Array] +n+ assembled values (nil, scalars, Arrays, Hashes)
|
|
620
|
+
# @raise [FormatError] when the levels do not add up to +n+ rows
|
|
621
|
+
def read_rows(n, chunks)
|
|
333
622
|
@defs = {}
|
|
334
623
|
@reps = {}
|
|
335
624
|
@vals = {}
|
|
@@ -340,9 +629,7 @@ module Herringbone
|
|
|
340
629
|
end
|
|
341
630
|
@ei = Hash.new(0) # entry cursor per leaf column
|
|
342
631
|
@vi = Hash.new(0) # value cursor per leaf column
|
|
343
|
-
end
|
|
344
632
|
|
|
345
|
-
def read_rows(n)
|
|
346
633
|
f = @field
|
|
347
634
|
if f.leaf? && f.column.max_repetition_level.zero?
|
|
348
635
|
idx = f.column.index
|
|
@@ -352,7 +639,7 @@ module Herringbone
|
|
|
352
639
|
max = f.column.max_definition_level
|
|
353
640
|
return vals if vals.size == defs.size
|
|
354
641
|
vi = -1
|
|
355
|
-
return defs.map { |d| d == max ? vals[vi += 1] : nil }
|
|
642
|
+
return defs.map { |d| (d == max) ? vals[vi += 1] : nil }
|
|
356
643
|
end
|
|
357
644
|
return read_simple_list(f) if f.kind == :list && f.element.leaf? && f.element.column.max_repetition_level == 1
|
|
358
645
|
|
|
@@ -368,6 +655,9 @@ module Herringbone
|
|
|
368
655
|
private
|
|
369
656
|
|
|
370
657
|
# Fast path for a top-level list of primitives (the most common nested shape)
|
|
658
|
+
#
|
|
659
|
+
# @param field [Schema::Field] list field whose element is a leaf with repetition level 1
|
|
660
|
+
# @return [Array<Array, nil>] one list (or nil) per row
|
|
371
661
|
def read_simple_list(field)
|
|
372
662
|
col = field.element.column
|
|
373
663
|
defs = @defs[col.index]
|
|
@@ -407,6 +697,19 @@ module Herringbone
|
|
|
407
697
|
out
|
|
408
698
|
end
|
|
409
699
|
|
|
700
|
+
# Hash key for a struct member, memoized
|
|
701
|
+
#
|
|
702
|
+
# @param field [Schema::Field] struct child
|
|
703
|
+
# @return [String, Symbol] frozen name, or Symbol when symbolizing
|
|
704
|
+
def key_for(field)
|
|
705
|
+
@keys[field] ||= @symbolize ? field.name.to_sym : -field.name
|
|
706
|
+
end
|
|
707
|
+
|
|
708
|
+
# Assembles one value of +field+ at the current entry cursors, recursing into children
|
|
709
|
+
#
|
|
710
|
+
# @param field [Schema::Field] field to assemble
|
|
711
|
+
# @return [Object, nil] scalar, Hash (struct, map), Array (list) or nil
|
|
712
|
+
# @raise [FormatError] when the definition levels run out
|
|
410
713
|
def read(field)
|
|
411
714
|
c = field.first_leaf.index
|
|
412
715
|
kind = field.kind
|
|
@@ -427,7 +730,7 @@ module Herringbone
|
|
|
427
730
|
@vals[c][v]
|
|
428
731
|
when :struct
|
|
429
732
|
h = {}
|
|
430
|
-
field.children.each { |ch| h[ch
|
|
733
|
+
field.children.each { |ch| h[key_for(ch)] = read(ch) }
|
|
431
734
|
h
|
|
432
735
|
when :list
|
|
433
736
|
if d < field.item_def
|
|
@@ -461,9 +764,19 @@ module Herringbone
|
|
|
461
764
|
end
|
|
462
765
|
end
|
|
463
766
|
|
|
767
|
+
# Moves every leaf of +field+ past one entry (a null or empty value takes a single entry)
|
|
768
|
+
#
|
|
769
|
+
# @param field [Schema::Field] field whose leaves to advance
|
|
770
|
+
# @return [void]
|
|
464
771
|
def skip(field)
|
|
465
772
|
field.leaves.each { |col| @ei[col.index] += 1 }
|
|
466
773
|
end
|
|
467
774
|
end
|
|
468
775
|
end
|
|
469
776
|
end
|
|
777
|
+
|
|
778
|
+
require_relative "reader/page_stream"
|
|
779
|
+
require_relative "reader/column_chunk_reader"
|
|
780
|
+
require_relative "reader/column_cursor"
|
|
781
|
+
require_relative "reader/scan"
|
|
782
|
+
require_relative "reader/numo"
|