herringbone 0.1.0 → 0.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/README.md +146 -74
- data/bin/herringbone +38 -26
- data/lib/herringbone/active_record.rb +10 -13
- data/lib/herringbone/bloom_filter.rb +270 -0
- data/lib/herringbone/compression.rb +73 -16
- data/lib/herringbone/encodings/delta.rb +7 -7
- data/lib/herringbone/encodings/plain.rb +7 -7
- data/lib/herringbone/encodings/rle.rb +2 -2
- data/lib/herringbone/format.rb +53 -0
- data/lib/herringbone/inspector.rb +1388 -0
- data/lib/herringbone/reader/column_chunk_reader.rb +296 -0
- data/lib/herringbone/reader/column_cursor.rb +181 -0
- data/lib/herringbone/reader/page_stream.rb +339 -0
- data/lib/herringbone/reader/scan.rb +256 -0
- data/lib/herringbone/reader.rb +294 -272
- data/lib/herringbone/schema.rb +21 -76
- data/lib/herringbone/types.rb +4 -4
- data/lib/herringbone/version.rb +1 -1
- data/lib/herringbone/visualizer.rb +1096 -0
- data/lib/herringbone/writer.rb +258 -90
- data/lib/herringbone/xxhash.rb +319 -0
- data/lib/herringbone.rb +36 -9
- metadata +12 -32
data/lib/herringbone/reader.rb
CHANGED
|
@@ -1,125 +1,321 @@
|
|
|
1
1
|
# frozen_string_literal: true
|
|
2
2
|
|
|
3
3
|
module Herringbone
|
|
4
|
-
# Reads Parquet files.
|
|
4
|
+
# Reads Parquet files from a random-access IO. Herringbone never opens files by path: the
|
|
5
|
+
# caller opens (and closes) the IO.
|
|
5
6
|
#
|
|
6
|
-
#
|
|
7
|
-
# reader
|
|
8
|
-
# reader.
|
|
9
|
-
# reader.
|
|
7
|
+
# File.open("data.parquet", "rb") do |file|
|
|
8
|
+
# reader = Herringbone::Reader.new(file)
|
|
9
|
+
# reader.each_row { |row| p row } # rows as Hashes with String keys
|
|
10
|
+
# reader.each_batch(1000, as: :columns) { |batch| ... } # { "id" => [...], ... } per batch
|
|
11
|
+
# reader.read(columns: ["id"], where: { id: 1..10 }) # everything at once
|
|
10
12
|
# end
|
|
13
|
+
#
|
|
14
|
+
# Rows are read in batches: pages are read and decoded one at a time per column, so memory use
|
|
15
|
+
# depends on the batch size and the page size, not on the size of the row groups.
|
|
16
|
+
#
|
|
17
|
+
# Options:
|
|
18
|
+
# keys: :string (default) or :symbol, for row Hashes and Hashes built from structs.
|
|
19
|
+
# Map keys are always the stored values.
|
|
20
|
+
# time_zone: return timestamps in this zone instead of UTC. A UTC offset ("+02:00", or
|
|
21
|
+
# seconds as an Integer), a timezone object Time#getlocal accepts (e.g. a
|
|
22
|
+
# TZInfo::Timezone), or anything responding to #at such as an
|
|
23
|
+
# ActiveSupport::TimeZone (Time.zone), which yields ActiveSupport::TimeWithZone.
|
|
11
24
|
class Reader
|
|
12
|
-
include Enumerable
|
|
13
|
-
|
|
14
25
|
MAGIC = "PAR1"
|
|
15
|
-
|
|
16
|
-
|
|
17
|
-
|
|
18
|
-
|
|
19
|
-
|
|
20
|
-
|
|
21
|
-
|
|
22
|
-
|
|
23
|
-
|
|
24
|
-
|
|
25
|
-
|
|
26
|
-
|
|
27
|
-
end
|
|
28
|
-
|
|
29
|
-
# +source+ is an IO (opened in binary mode), a path, or a String of Parquet bytes when
|
|
30
|
-
# +string: true+ is given.
|
|
31
|
-
def initialize(source)
|
|
32
|
-
@io = case source
|
|
33
|
-
when String then File.open(source, "rb")
|
|
34
|
-
when Pathname then File.open(source.to_s, "rb")
|
|
35
|
-
else source
|
|
26
|
+
DEFAULT_BATCH_SIZE = 1024
|
|
27
|
+
KEY_MODES = %i[string symbol].freeze
|
|
28
|
+
|
|
29
|
+
# The schema, and the file's FileMetaData (the decoded Thrift footer)
|
|
30
|
+
attr_reader :schema, :file_metadata
|
|
31
|
+
|
|
32
|
+
# +io+ must support #seek and #read (a File opened with "rb", StringIO, Tempfile...)
|
|
33
|
+
def initialize(io, keys: :string, time_zone: nil)
|
|
34
|
+
unless io.respond_to?(:seek) && io.respond_to?(:read)
|
|
35
|
+
raise ArgumentError, "Herringbone::Reader expects an IO that supports #seek and #read " \
|
|
36
|
+
"(e.g. File.open(path, \"rb\")), got #{io.class == String ? "a String" : io.class}" \
|
|
37
|
+
"#{" (wrap Parquet bytes in a StringIO)" if io.is_a?(String)}"
|
|
36
38
|
end
|
|
37
|
-
|
|
38
|
-
|
|
39
|
+
keys = keys.to_sym if keys.is_a?(String)
|
|
40
|
+
raise ArgumentError, "keys: must be :string or :symbol, got #{keys.inspect}" unless KEY_MODES.include?(keys)
|
|
41
|
+
@symbolize = keys == :symbol
|
|
42
|
+
@zone_converter = zone_converter(time_zone)
|
|
43
|
+
@io = io
|
|
44
|
+
@file_metadata = read_footer
|
|
45
|
+
@schema = Schema.from_elements(@file_metadata.schema)
|
|
39
46
|
end
|
|
40
47
|
|
|
41
|
-
def
|
|
42
|
-
|
|
43
|
-
end
|
|
48
|
+
def num_rows = @file_metadata.num_rows
|
|
49
|
+
def row_groups = @file_metadata.row_groups
|
|
44
50
|
|
|
45
|
-
|
|
46
|
-
|
|
51
|
+
# The footer's key/value metadata as a Hash (what the writer's metadata: option stores)
|
|
52
|
+
def metadata
|
|
53
|
+
(@file_metadata.key_value_metadata || []).to_h { |kv| [kv.key, kv.value] }
|
|
47
54
|
end
|
|
48
55
|
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
|
|
54
|
-
|
|
55
|
-
|
|
56
|
-
|
|
56
|
+
# Yields batches of up to +size+ rows (all batches are full except the last one; batches span
|
|
57
|
+
# row groups). Only one page per column and the current batch are held in memory.
|
|
58
|
+
#
|
|
59
|
+
# as: :rows (default) yields an Array of row Hashes; as: :columns yields a Hash of top-level
|
|
60
|
+
# field name => Array of that field's values in the batch, which skips building a Hash per row
|
|
61
|
+
# and is noticeably faster when you process data column by column.
|
|
62
|
+
#
|
|
63
|
+
# where: only yields rows matching all conditions (see Reader::Filter). Row groups and pages
|
|
64
|
+
# that cannot match are skipped using statistics, bloom filters and the page index, and the
|
|
65
|
+
# remaining rows are checked one by one. Filtered columns need not be in +columns+.
|
|
66
|
+
# from: skips the first rows of the file (jumping over pages with the page index), and
|
|
67
|
+
# limit: stops after yielding that many rows.
|
|
68
|
+
def each_batch(size = DEFAULT_BATCH_SIZE, columns: nil, as: :rows, where: nil, from: nil, limit: nil)
|
|
69
|
+
return enum_for(:each_batch, size, columns: columns, as: as, where: where, from: from, limit: limit) unless block_given?
|
|
70
|
+
size = Integer(size)
|
|
71
|
+
raise ArgumentError, "Batch size must be positive, got #{size}" unless size.positive?
|
|
72
|
+
raise ArgumentError, "as: must be :rows or :columns, got #{as.inspect}" unless as == :rows || as == :columns
|
|
73
|
+
raise ArgumentError, "limit: must not be negative" if limit && limit.negative?
|
|
74
|
+
return self if limit&.zero?
|
|
75
|
+
columnar = as == :columns
|
|
76
|
+
symbolize = @symbolize
|
|
77
|
+
out_fields = select_fields(columns)
|
|
78
|
+
filter = where && !where.empty? ? Filter.new(@schema, where) : nil
|
|
79
|
+
fields = filter ? out_fields | filter.fields : out_fields
|
|
80
|
+
names = row_keys(out_fields, symbolize)
|
|
81
|
+
nout = out_fields.size
|
|
82
|
+
converters = @schema.columns.map { |col| converter_for(col) }
|
|
83
|
+
assemblers = fields.map { |f| Assembler.new(f, symbolize) }
|
|
84
|
+
left_to_yield = limit
|
|
85
|
+
pending = pending_rows = nil
|
|
86
|
+
emit = lambda do |data, k|
|
|
87
|
+
if columnar
|
|
88
|
+
if pending
|
|
89
|
+
pending.each_with_index { |col, j| col.concat(data[j]) }
|
|
90
|
+
else
|
|
91
|
+
pending = data
|
|
92
|
+
end
|
|
93
|
+
pending_rows = (pending_rows || 0) + k
|
|
94
|
+
if pending_rows >= size
|
|
95
|
+
yield names.zip(pending).to_h
|
|
96
|
+
pending = pending_rows = nil
|
|
97
|
+
end
|
|
98
|
+
else
|
|
99
|
+
rows = build_rows(names, data, k)
|
|
100
|
+
if pending
|
|
101
|
+
pending.concat(rows)
|
|
102
|
+
else
|
|
103
|
+
pending = rows
|
|
104
|
+
end
|
|
105
|
+
pending_rows = pending.size
|
|
106
|
+
if pending_rows >= size
|
|
107
|
+
yield pending
|
|
108
|
+
pending = pending_rows = nil
|
|
109
|
+
end
|
|
110
|
+
end
|
|
111
|
+
end
|
|
57
112
|
|
|
58
|
-
|
|
59
|
-
|
|
60
|
-
|
|
61
|
-
|
|
62
|
-
|
|
63
|
-
|
|
64
|
-
|
|
65
|
-
|
|
66
|
-
|
|
67
|
-
|
|
68
|
-
|
|
69
|
-
|
|
113
|
+
plan_rows(filter, from).each do |rg_index, ranges|
|
|
114
|
+
rg = row_groups[rg_index]
|
|
115
|
+
partial = ranges != [[0, rg.num_rows]]
|
|
116
|
+
cursors = fields.map do |f|
|
|
117
|
+
f.leaves.map do |col|
|
|
118
|
+
reader = ColumnChunkReader.new(@io, rg.columns.fetch(col.index), col, converter: converters[col.index], lazy: true)
|
|
119
|
+
reader.locations = page_index(rg_index, col)[1]&.page_locations if partial
|
|
120
|
+
[col.index, ColumnCursor.new(reader)]
|
|
121
|
+
end
|
|
122
|
+
end
|
|
123
|
+
ranges.each do |first, stop|
|
|
124
|
+
cursors.each { |cs| cs.each { |_, cursor| cursor.seek(first) } }
|
|
125
|
+
left = stop - first
|
|
126
|
+
while left.positive?
|
|
127
|
+
# Rows still missing from the current batch (pending_rows counts rows in both modes)
|
|
128
|
+
k = pending ? size - pending_rows : size
|
|
129
|
+
raise Error, "Internal error: batch needs #{k} rows" unless k.positive? # never loop without progress
|
|
130
|
+
k = left if left < k
|
|
131
|
+
data = assemblers.each_with_index.map do |asm, j|
|
|
132
|
+
asm.read_rows(k, cursors[j].to_h { |idx, cursor| [idx, cursor.take(k)] })
|
|
133
|
+
end
|
|
134
|
+
left -= k
|
|
135
|
+
kept = k
|
|
136
|
+
if filter
|
|
137
|
+
keep = filter.matching_rows(data, fields, k, symbolize)
|
|
138
|
+
kept = keep.size
|
|
139
|
+
next if kept.zero?
|
|
140
|
+
data = data.first(nout).map { |col| keep.map { |i| col[i] } } if kept < k
|
|
141
|
+
end
|
|
142
|
+
data = data.first(nout) if data.size > nout
|
|
143
|
+
if left_to_yield && kept >= left_to_yield
|
|
144
|
+
emit.call(data.map { |col| col.first(left_to_yield) }, left_to_yield)
|
|
145
|
+
yield(columnar ? names.zip(pending).to_h : pending) if pending
|
|
146
|
+
return self
|
|
147
|
+
end
|
|
148
|
+
left_to_yield -= kept if left_to_yield
|
|
149
|
+
emit.call(data, kept)
|
|
150
|
+
end
|
|
70
151
|
end
|
|
71
152
|
end
|
|
153
|
+
yield(columnar ? names.zip(pending).to_h : pending) if pending
|
|
72
154
|
self
|
|
73
155
|
end
|
|
74
|
-
alias_method :each, :each_row
|
|
75
156
|
|
|
76
|
-
#
|
|
77
|
-
|
|
157
|
+
# What a read with +where:+ / +from:+ would touch, without reading any data: an Array of
|
|
158
|
+
# { row_group:, rows:, ranges: [[first_row, end_row), ...] } for the row groups that are
|
|
159
|
+
# read. Row groups ruled out entirely are left out.
|
|
160
|
+
def scan_plan(where: nil, from: nil)
|
|
161
|
+
filter = where && !where.empty? ? Filter.new(@schema, where) : nil
|
|
162
|
+
plan_rows(filter, from).map do |rg_index, ranges|
|
|
163
|
+
{ row_group: rg_index, rows: ranges.sum { |s, e| e - s }, ranges: ranges }
|
|
164
|
+
end
|
|
165
|
+
end
|
|
78
166
|
|
|
79
|
-
#
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
|
|
167
|
+
# Yields each row as a Hash of top-level field name => value. Takes the options of each_batch
|
|
168
|
+
# except as:.
|
|
169
|
+
def each_row(columns: nil, where: nil, from: nil, limit: nil, &block)
|
|
170
|
+
return enum_for(:each_row, columns: columns, where: where, from: from, limit: limit) unless block
|
|
171
|
+
each_batch(columns: columns, where: where, from: from, limit: limit) { |rows| rows.each(&block) }
|
|
172
|
+
self
|
|
83
173
|
end
|
|
84
174
|
|
|
85
|
-
#
|
|
86
|
-
|
|
87
|
-
|
|
88
|
-
|
|
175
|
+
# Reads the whole file (or the selected rows) at once: an Array of row Hashes, or with
|
|
176
|
+
# as: :columns a Hash of top-level field name => Array of values. Takes the options of each_batch.
|
|
177
|
+
def read(columns: nil, as: :rows, where: nil, from: nil, limit: nil)
|
|
178
|
+
if as == :columns
|
|
179
|
+
out = row_keys(select_fields(columns), @symbolize).to_h { |name| [name, []] }
|
|
180
|
+
each_batch(65_536, columns: columns, as: :columns, where: where, from: from, limit: limit) do |batch|
|
|
181
|
+
batch.each { |name, values| out[name].concat(values) }
|
|
182
|
+
end
|
|
183
|
+
else
|
|
184
|
+
out = []
|
|
185
|
+
each_batch(columns: columns, as: as, where: where, from: from, limit: limit) { |rows| out.concat(rows) }
|
|
186
|
+
end
|
|
187
|
+
out
|
|
89
188
|
end
|
|
90
189
|
|
|
91
|
-
#
|
|
92
|
-
#
|
|
93
|
-
def
|
|
190
|
+
# Internal (used by reads with where:/from:): [ColumnIndex or nil, OffsetIndex or nil] of a
|
|
191
|
+
# leaf column in a row group
|
|
192
|
+
def page_index(row_group_index, column)
|
|
94
193
|
column = @schema.column(column) unless column.is_a?(Schema::Column)
|
|
95
194
|
raise ArgumentError, "No such leaf column" unless column
|
|
96
|
-
|
|
97
|
-
|
|
195
|
+
@page_indexes ||= {}
|
|
196
|
+
@page_indexes[[row_group_index, column.index]] ||= begin
|
|
197
|
+
chunk = row_groups.fetch(row_group_index).columns.fetch(column.index)
|
|
198
|
+
[read_struct(Format::ColumnIndex, chunk.column_index_offset, chunk.column_index_length),
|
|
199
|
+
read_struct(Format::OffsetIndex, chunk.offset_index_offset, chunk.offset_index_length)]
|
|
200
|
+
end
|
|
98
201
|
end
|
|
99
202
|
|
|
100
203
|
def inspect
|
|
101
|
-
"#<#{self.class.name} rows=#{num_rows} row_groups=#{
|
|
204
|
+
"#<#{self.class.name} rows=#{num_rows} row_groups=#{row_groups.size} created_by=#{@file_metadata.created_by.inspect}>"
|
|
102
205
|
end
|
|
103
206
|
|
|
104
207
|
private
|
|
105
208
|
|
|
209
|
+
# [[row_group_index, [[first_row, end_row), ...]], ...] to read, after ruling out row groups
|
|
210
|
+
# (statistics, bloom filters) and pages (page index), and skipping the first +from+ rows
|
|
211
|
+
def plan_rows(filter, from)
|
|
212
|
+
from = Integer(from || 0)
|
|
213
|
+
raise ArgumentError, "from: must not be negative" if from.negative?
|
|
214
|
+
offset = 0
|
|
215
|
+
plan = []
|
|
216
|
+
row_groups.each_with_index do |rg, i|
|
|
217
|
+
n = rg.num_rows
|
|
218
|
+
first = from - offset
|
|
219
|
+
offset += n
|
|
220
|
+
next if n.zero? || first >= n
|
|
221
|
+
ranges = [[first.positive? ? first : 0, n]]
|
|
222
|
+
if filter
|
|
223
|
+
next unless filter.row_group_may_match?(self, i)
|
|
224
|
+
ranges = Filter.intersect(ranges, filter.page_ranges(self, i))
|
|
225
|
+
end
|
|
226
|
+
plan << [i, ranges] unless ranges.empty?
|
|
227
|
+
end
|
|
228
|
+
plan
|
|
229
|
+
end
|
|
230
|
+
|
|
231
|
+
def read_struct(klass, offset, length)
|
|
232
|
+
return nil unless offset && length&.positive?
|
|
233
|
+
@io.seek(offset)
|
|
234
|
+
bytes = @io.read(length)
|
|
235
|
+
return nil unless bytes&.bytesize == length
|
|
236
|
+
klass.decode(bytes.b).first
|
|
237
|
+
rescue Thrift::Error
|
|
238
|
+
nil # a damaged index only means pages cannot be skipped
|
|
239
|
+
end
|
|
240
|
+
|
|
106
241
|
def select_fields(columns)
|
|
107
242
|
return @schema.fields unless columns
|
|
108
243
|
Array(columns).map { |c| @schema.field(c) or raise ArgumentError, "No such column #{c.inspect}" }
|
|
109
244
|
end
|
|
110
245
|
|
|
111
|
-
def
|
|
112
|
-
|
|
113
|
-
|
|
114
|
-
|
|
115
|
-
|
|
116
|
-
|
|
117
|
-
|
|
246
|
+
def row_keys(fields, symbolize)
|
|
247
|
+
fields.map { |f| symbolize ? f.name.to_sym : -f.name }
|
|
248
|
+
end
|
|
249
|
+
|
|
250
|
+
def build_rows(names, data, k)
|
|
251
|
+
return Array.new(k) { {} } if names.empty?
|
|
252
|
+
return data.first.map { |v| { names.first => v } } if names.size == 1
|
|
253
|
+
nf = names.size
|
|
254
|
+
Array.new(k) do |i|
|
|
255
|
+
row = {}
|
|
256
|
+
j = 0
|
|
257
|
+
while j < nf
|
|
258
|
+
row[names[j]] = data[j][i]
|
|
259
|
+
j += 1
|
|
118
260
|
end
|
|
119
|
-
|
|
261
|
+
row
|
|
120
262
|
end
|
|
121
263
|
end
|
|
122
264
|
|
|
265
|
+
# The column's value converter, with the time zone applied to timestamps
|
|
266
|
+
def converter_for(column)
|
|
267
|
+
base = column.converter
|
|
268
|
+
zc = @zone_converter
|
|
269
|
+
return base unless zc && base && instant_column?(column)
|
|
270
|
+
->(v) { zc.call(base.call(v)) }
|
|
271
|
+
end
|
|
272
|
+
|
|
273
|
+
# Timestamps that denote an instant (UTC-adjusted TIMESTAMP, INT96). Local timestamps
|
|
274
|
+
# (isAdjustedToUTC = false) are wall-clock values and are left as they are.
|
|
275
|
+
def instant_column?(column)
|
|
276
|
+
return true if column.type == Format::Type::INT96
|
|
277
|
+
kind, _unit, utc = Types.logical_of(column.node)
|
|
278
|
+
kind == :timestamp && utc
|
|
279
|
+
end
|
|
280
|
+
|
|
281
|
+
OFFSET_PATTERN = /\A([+-])(\d\d)(?::?(\d\d)(?::?(\d\d))?)?\z/
|
|
282
|
+
|
|
283
|
+
# A lambda turning a UTC Time into the zone, or nil for UTC
|
|
284
|
+
def zone_converter(zone)
|
|
285
|
+
case zone
|
|
286
|
+
when nil then nil
|
|
287
|
+
when Integer
|
|
288
|
+
return nil if zone.zero?
|
|
289
|
+
Time.at(0).getlocal(zone) # validates the range
|
|
290
|
+
->(t) { t.getlocal(zone) }
|
|
291
|
+
when String
|
|
292
|
+
return nil if zone.match?(/\A(?:utc|z)\z/i)
|
|
293
|
+
if (m = OFFSET_PATTERN.match(zone))
|
|
294
|
+
# As seconds, since Ruby 3.0 only parses "+HH:MM" offset strings
|
|
295
|
+
return zone_converter((m[1] == "-" ? -1 : 1) * (m[2].to_i * 3600 + m[3].to_i * 60 + m[4].to_i))
|
|
296
|
+
end
|
|
297
|
+
if defined?(::ActiveSupport::TimeZone) && (tz = ::ActiveSupport::TimeZone[zone])
|
|
298
|
+
return ->(t) { tz.at(t) }
|
|
299
|
+
end
|
|
300
|
+
if defined?(::TZInfo::Timezone)
|
|
301
|
+
tz = ::TZInfo::Timezone.get(zone)
|
|
302
|
+
return ->(t) { t.getlocal(tz) }
|
|
303
|
+
end
|
|
304
|
+
raise ArgumentError, "Unknown time zone #{zone.inspect}: use a UTC offset like \"+02:00\", " \
|
|
305
|
+
"a timezone object (e.g. TZInfo::Timezone.get(#{zone.inspect})) or an ActiveSupport::TimeZone"
|
|
306
|
+
else
|
|
307
|
+
return ->(t) { zone.at(t) } if zone.respond_to?(:at)
|
|
308
|
+
if zone.respond_to?(:utc_to_local)
|
|
309
|
+
Time.at(0).getlocal(zone)
|
|
310
|
+
return ->(t) { t.getlocal(zone) }
|
|
311
|
+
end
|
|
312
|
+
raise ArgumentError, "Unsupported time_zone: #{zone.inspect}"
|
|
313
|
+
end
|
|
314
|
+
rescue ArgumentError => e
|
|
315
|
+
raise if e.message.start_with?("Unknown time zone", "Unsupported time_zone", "Invalid time_zone")
|
|
316
|
+
raise ArgumentError, "Invalid time_zone #{zone.inspect}: #{e.message}"
|
|
317
|
+
end
|
|
318
|
+
|
|
123
319
|
def read_footer
|
|
124
320
|
@io.seek(0, IO::SEEK_END)
|
|
125
321
|
size = @io.pos
|
|
@@ -137,199 +333,18 @@ module Herringbone
|
|
|
137
333
|
raise FormatError, "Corrupt file metadata: #{e.message}"
|
|
138
334
|
end
|
|
139
335
|
|
|
140
|
-
# Decodes all pages of one column chunk into levels and values
|
|
141
|
-
class ColumnChunkReader
|
|
142
|
-
T = Format::Type
|
|
143
|
-
E = Format::Encoding
|
|
144
|
-
|
|
145
|
-
def initialize(io, chunk, column)
|
|
146
|
-
@io = io
|
|
147
|
-
@chunk = chunk
|
|
148
|
-
@column = column
|
|
149
|
-
@meta = chunk.meta_data or raise UnsupportedError, "Column chunk without metadata (encrypted?)"
|
|
150
|
-
raise UnsupportedError, "Column chunks in external files are not supported" if chunk.file_path
|
|
151
|
-
@max_def = column.max_definition_level
|
|
152
|
-
@max_rep = column.max_repetition_level
|
|
153
|
-
@converter = column.converter
|
|
154
|
-
end
|
|
155
|
-
|
|
156
|
-
def read
|
|
157
|
-
buf = read_bytes
|
|
158
|
-
defs = @max_def.positive? ? [] : nil
|
|
159
|
-
reps = @max_rep.positive? ? [] : nil
|
|
160
|
-
values = []
|
|
161
|
-
pos = 0
|
|
162
|
-
total = @meta.num_values
|
|
163
|
-
seen = 0
|
|
164
|
-
while seen < total && pos < buf.bytesize
|
|
165
|
-
header, pos = decode_page_header(buf, pos)
|
|
166
|
-
size = header.compressed_page_size
|
|
167
|
-
# Some old writers under-report total_compressed_size; read on past the declared end
|
|
168
|
-
extend_buffer(buf, pos + size - buf.bytesize) if pos + size > buf.bytesize
|
|
169
|
-
raise FormatError, "Page overruns column chunk" if pos + size > buf.bytesize
|
|
170
|
-
body = buf.byteslice(pos, size)
|
|
171
|
-
pos += size
|
|
172
|
-
case header.type
|
|
173
|
-
when Format::PageType::DICTIONARY_PAGE
|
|
174
|
-
read_dictionary(header, body)
|
|
175
|
-
when Format::PageType::DATA_PAGE
|
|
176
|
-
seen += read_data_page_v1(header, body, defs, reps, values)
|
|
177
|
-
when Format::PageType::DATA_PAGE_V2
|
|
178
|
-
seen += read_data_page_v2(header, body, defs, reps, values)
|
|
179
|
-
end
|
|
180
|
-
end
|
|
181
|
-
raise FormatError, "Column #{@column.dotted_path}: read #{seen} of #{total} values" if seen < total
|
|
182
|
-
[defs, reps, values]
|
|
183
|
-
rescue Thrift::Error => e
|
|
184
|
-
raise FormatError, "Corrupt page header in #{@column.dotted_path}: #{e.message}"
|
|
185
|
-
end
|
|
186
|
-
|
|
187
|
-
private
|
|
188
|
-
|
|
189
|
-
def read_bytes
|
|
190
|
-
start = @meta.data_page_offset
|
|
191
|
-
dict = @meta.dictionary_page_offset
|
|
192
|
-
# Some writers store 0 when there is no dictionary page
|
|
193
|
-
start = dict if dict && dict.positive? && dict < start
|
|
194
|
-
@start = start
|
|
195
|
-
@io.seek(start)
|
|
196
|
-
len = @meta.total_compressed_size
|
|
197
|
-
(@io.read(len) || "".b).b
|
|
198
|
-
end
|
|
199
|
-
|
|
200
|
-
def extend_buffer(buf, nbytes)
|
|
201
|
-
@io.seek(@start + buf.bytesize)
|
|
202
|
-
more = @io.read(nbytes)
|
|
203
|
-
buf << more.b if more
|
|
204
|
-
end
|
|
205
|
-
|
|
206
|
-
def decode_page_header(buf, pos)
|
|
207
|
-
Format::PageHeader.decode(buf, pos)
|
|
208
|
-
rescue Thrift::Error
|
|
209
|
-
# The header may straddle the declared end of the chunk
|
|
210
|
-
before = buf.bytesize
|
|
211
|
-
extend_buffer(buf, 1024)
|
|
212
|
-
raise if buf.bytesize == before
|
|
213
|
-
Format::PageHeader.decode(buf, pos)
|
|
214
|
-
end
|
|
215
|
-
|
|
216
|
-
def decompress(body, size)
|
|
217
|
-
Compression.decompress(@meta.codec, body, size)
|
|
218
|
-
end
|
|
219
|
-
|
|
220
|
-
def read_dictionary(header, body)
|
|
221
|
-
dh = header.dictionary_page_header
|
|
222
|
-
data = decompress(body, header.uncompressed_page_size)
|
|
223
|
-
vals, = Encodings::Plain.decode(data, 0, dh.num_values, @column.type, @column.type_length)
|
|
224
|
-
vals.map!(&@converter) if @converter
|
|
225
|
-
@dictionary = vals
|
|
226
|
-
end
|
|
227
|
-
|
|
228
|
-
def read_data_page_v1(header, body, defs, reps, values)
|
|
229
|
-
dh = header.data_page_header
|
|
230
|
-
n = dh.num_values
|
|
231
|
-
data = decompress(body, header.uncompressed_page_size)
|
|
232
|
-
pos = 0
|
|
233
|
-
if @max_rep.positive?
|
|
234
|
-
levels, pos = read_levels(data, pos, dh.repetition_level_encoding, @max_rep, n)
|
|
235
|
-
reps.concat(levels)
|
|
236
|
-
end
|
|
237
|
-
non_null = n
|
|
238
|
-
if @max_def.positive?
|
|
239
|
-
levels, pos = read_levels(data, pos, dh.definition_level_encoding, @max_def, n)
|
|
240
|
-
non_null = levels.count(@max_def)
|
|
241
|
-
defs.concat(levels)
|
|
242
|
-
end
|
|
243
|
-
values.concat(decode_values(data, pos, non_null, dh.encoding))
|
|
244
|
-
n
|
|
245
|
-
end
|
|
246
|
-
|
|
247
|
-
def read_data_page_v2(header, body, defs, reps, values)
|
|
248
|
-
dh = header.data_page_header_v2
|
|
249
|
-
n = dh.num_values
|
|
250
|
-
rep_len = dh.repetition_levels_byte_length
|
|
251
|
-
def_len = dh.definition_levels_byte_length
|
|
252
|
-
if @max_rep.positive?
|
|
253
|
-
reps.concat(Encodings::RLE.decode_hybrid(body, 0, rep_len, RLE_WIDTH[@max_rep], n))
|
|
254
|
-
end
|
|
255
|
-
non_null = n
|
|
256
|
-
if @max_def.positive?
|
|
257
|
-
levels = Encodings::RLE.decode_hybrid(body, rep_len, rep_len + def_len, RLE_WIDTH[@max_def], n)
|
|
258
|
-
non_null = levels.count(@max_def)
|
|
259
|
-
defs.concat(levels)
|
|
260
|
-
end
|
|
261
|
-
data = body.byteslice(rep_len + def_len, body.bytesize - rep_len - def_len)
|
|
262
|
-
if dh.is_compressed != false
|
|
263
|
-
data = decompress(data, header.uncompressed_page_size - rep_len - def_len)
|
|
264
|
-
end
|
|
265
|
-
values.concat(decode_values(data, 0, non_null, dh.encoding))
|
|
266
|
-
n
|
|
267
|
-
end
|
|
268
|
-
|
|
269
|
-
RLE_WIDTH = Hash.new { |h, k| h[k] = k.bit_length }
|
|
270
|
-
|
|
271
|
-
def read_levels(data, pos, encoding, max, n)
|
|
272
|
-
width = RLE_WIDTH[max]
|
|
273
|
-
case encoding
|
|
274
|
-
when E::RLE
|
|
275
|
-
len = data.byteslice(pos, 4).unpack1("V")
|
|
276
|
-
start = pos + 4
|
|
277
|
-
[Encodings::RLE.decode_hybrid(data, start, start + len, width, n), start + len]
|
|
278
|
-
when E::BIT_PACKED
|
|
279
|
-
[Encodings::RLE.decode_legacy_bit_packed(data, pos, width, n), pos + (n * width + 7) / 8]
|
|
280
|
-
else
|
|
281
|
-
raise UnsupportedError, "Unsupported level encoding #{E::NAMES[encoding] || encoding}"
|
|
282
|
-
end
|
|
283
|
-
end
|
|
284
|
-
|
|
285
|
-
def decode_values(data, pos, count, encoding)
|
|
286
|
-
type = @column.type
|
|
287
|
-
vals = case encoding
|
|
288
|
-
when E::PLAIN
|
|
289
|
-
Encodings::Plain.decode(data, pos, count, type, @column.type_length).first
|
|
290
|
-
when E::PLAIN_DICTIONARY, E::RLE_DICTIONARY
|
|
291
|
-
raise FormatError, "Dictionary-encoded page without a dictionary in #{@column.dotted_path}" unless @dictionary
|
|
292
|
-
return [] if count.zero?
|
|
293
|
-
width = data.getbyte(pos)
|
|
294
|
-
indices = Encodings::RLE.decode_hybrid(data, pos + 1, data.bytesize, width, count)
|
|
295
|
-
dict = @dictionary
|
|
296
|
-
raise FormatError, "Dictionary index out of range in #{@column.dotted_path}" if indices.max >= dict.size
|
|
297
|
-
return indices.map! { |i| dict[i] }
|
|
298
|
-
when E::RLE
|
|
299
|
-
raise UnsupportedError, "RLE value encoding is only supported for BOOLEAN" unless type == T::BOOLEAN
|
|
300
|
-
len = data.byteslice(pos, 4).unpack1("V")
|
|
301
|
-
Encodings::RLE.decode_hybrid(data, pos + 4, pos + 4 + len, 1, count).map! { |v| v == 1 }
|
|
302
|
-
when E::DELTA_BINARY_PACKED
|
|
303
|
-
bits = type == T::INT32 ? 32 : 64
|
|
304
|
-
vals, = Encodings::Delta.decode_binary_packed(data, pos, bits, count)
|
|
305
|
-
raise FormatError, "DELTA_BINARY_PACKED page has #{vals.size} values, need #{count}" if vals.size < count
|
|
306
|
-
vals
|
|
307
|
-
when E::DELTA_LENGTH_BYTE_ARRAY
|
|
308
|
-
Encodings::Delta.decode_length_byte_array(data, pos, count).first
|
|
309
|
-
when E::DELTA_BYTE_ARRAY
|
|
310
|
-
Encodings::Delta.decode_byte_array(data, pos, count).first
|
|
311
|
-
when E::BYTE_STREAM_SPLIT
|
|
312
|
-
width = case type
|
|
313
|
-
when T::INT32, T::FLOAT then 4
|
|
314
|
-
when T::INT64, T::DOUBLE then 8
|
|
315
|
-
when T::FIXED_LEN_BYTE_ARRAY then @column.type_length
|
|
316
|
-
else raise UnsupportedError, "BYTE_STREAM_SPLIT is not valid for #{T::NAMES[type]}"
|
|
317
|
-
end
|
|
318
|
-
plain, = Encodings::ByteStreamSplit.decode(data, pos, count, width)
|
|
319
|
-
Encodings::Plain.decode(plain, 0, count, type, @column.type_length).first
|
|
320
|
-
else
|
|
321
|
-
raise UnsupportedError, "Unsupported encoding #{E::NAMES[encoding] || encoding}"
|
|
322
|
-
end
|
|
323
|
-
vals.map!(&@converter) if @converter
|
|
324
|
-
vals
|
|
325
|
-
end
|
|
326
|
-
end
|
|
327
|
-
|
|
328
336
|
# Rebuilds nested values of one top-level field from the levels of its leaf columns
|
|
329
337
|
# (the "record assembly" half of the Dremel algorithm).
|
|
330
338
|
class Assembler
|
|
331
|
-
def initialize(field,
|
|
339
|
+
def initialize(field, symbolize = false)
|
|
332
340
|
@field = field
|
|
341
|
+
@symbolize = symbolize
|
|
342
|
+
@keys = {} # struct child Field => Hash key
|
|
343
|
+
end
|
|
344
|
+
|
|
345
|
+
# Assembles +n+ values of the field from +chunks+ (leaf column index =>
|
|
346
|
+
# [defs, reps, values] holding exactly those rows)
|
|
347
|
+
def read_rows(n, chunks)
|
|
333
348
|
@defs = {}
|
|
334
349
|
@reps = {}
|
|
335
350
|
@vals = {}
|
|
@@ -340,9 +355,7 @@ module Herringbone
|
|
|
340
355
|
end
|
|
341
356
|
@ei = Hash.new(0) # entry cursor per leaf column
|
|
342
357
|
@vi = Hash.new(0) # value cursor per leaf column
|
|
343
|
-
end
|
|
344
358
|
|
|
345
|
-
def read_rows(n)
|
|
346
359
|
f = @field
|
|
347
360
|
if f.leaf? && f.column.max_repetition_level.zero?
|
|
348
361
|
idx = f.column.index
|
|
@@ -407,6 +420,10 @@ module Herringbone
|
|
|
407
420
|
out
|
|
408
421
|
end
|
|
409
422
|
|
|
423
|
+
def key_for(field)
|
|
424
|
+
@keys[field] ||= @symbolize ? field.name.to_sym : -field.name
|
|
425
|
+
end
|
|
426
|
+
|
|
410
427
|
def read(field)
|
|
411
428
|
c = field.first_leaf.index
|
|
412
429
|
kind = field.kind
|
|
@@ -427,7 +444,7 @@ module Herringbone
|
|
|
427
444
|
@vals[c][v]
|
|
428
445
|
when :struct
|
|
429
446
|
h = {}
|
|
430
|
-
field.children.each { |ch| h[ch
|
|
447
|
+
field.children.each { |ch| h[key_for(ch)] = read(ch) }
|
|
431
448
|
h
|
|
432
449
|
when :list
|
|
433
450
|
if d < field.item_def
|
|
@@ -467,3 +484,8 @@ module Herringbone
|
|
|
467
484
|
end
|
|
468
485
|
end
|
|
469
486
|
end
|
|
487
|
+
|
|
488
|
+
require_relative "reader/page_stream"
|
|
489
|
+
require_relative "reader/column_chunk_reader"
|
|
490
|
+
require_relative "reader/column_cursor"
|
|
491
|
+
require_relative "reader/scan"
|