herringbone 0.1.0 → 0.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,125 +1,321 @@
1
1
  # frozen_string_literal: true
2
2
 
3
3
  module Herringbone
4
- # Reads Parquet files.
4
+ # Reads Parquet files from a random-access IO. Herringbone never opens files by path: the
5
+ # caller opens (and closes) the IO.
5
6
  #
6
- # Herringbone::Reader.open("data.parquet") do |reader|
7
- # reader.each_row { |row| p row } # rows as Hashes with String keys
8
- # reader.column("name") # all values of a top-level field
9
- # reader.each_row(columns: ["id"]) { ... } # projection
7
+ # File.open("data.parquet", "rb") do |file|
8
+ # reader = Herringbone::Reader.new(file)
9
+ # reader.each_row { |row| p row } # rows as Hashes with String keys
10
+ # reader.each_batch(1000, as: :columns) { |batch| ... } # { "id" => [...], ... } per batch
11
+ # reader.read(columns: ["id"], where: { id: 1..10 }) # everything at once
10
12
  # end
13
+ #
14
+ # Rows are read in batches: pages are read and decoded one at a time per column, so memory use
15
+ # depends on the batch size and the page size, not on the size of the row groups.
16
+ #
17
+ # Options:
18
+ # keys: :string (default) or :symbol, for row Hashes and Hashes built from structs.
19
+ # Map keys are always the stored values.
20
+ # time_zone: return timestamps in this zone instead of UTC. A UTC offset ("+02:00", or
21
+ # seconds as an Integer), a timezone object Time#getlocal accepts (e.g. a
22
+ # TZInfo::Timezone), or anything responding to #at such as an
23
+ # ActiveSupport::TimeZone (Time.zone), which yields ActiveSupport::TimeWithZone.
11
24
  class Reader
12
- include Enumerable
13
-
14
25
  MAGIC = "PAR1"
15
-
16
- attr_reader :metadata, :schema
17
-
18
- def self.open(path)
19
- io = File.open(path, "rb")
20
- reader = new(io)
21
- return reader unless block_given?
22
- begin
23
- yield reader
24
- ensure
25
- io.close
26
- end
27
- end
28
-
29
- # +source+ is an IO (opened in binary mode), a path, or a String of Parquet bytes when
30
- # +string: true+ is given.
31
- def initialize(source)
32
- @io = case source
33
- when String then File.open(source, "rb")
34
- when Pathname then File.open(source.to_s, "rb")
35
- else source
26
+ DEFAULT_BATCH_SIZE = 1024
27
+ KEY_MODES = %i[string symbol].freeze
28
+
29
+ # The schema, and the file's FileMetaData (the decoded Thrift footer)
30
+ attr_reader :schema, :file_metadata
31
+
32
+ # +io+ must support #seek and #read (a File opened with "rb", StringIO, Tempfile...)
33
+ def initialize(io, keys: :string, time_zone: nil)
34
+ unless io.respond_to?(:seek) && io.respond_to?(:read)
35
+ raise ArgumentError, "Herringbone::Reader expects an IO that supports #seek and #read " \
36
+ "(e.g. File.open(path, \"rb\")), got #{io.class == String ? "a String" : io.class}" \
37
+ "#{" (wrap Parquet bytes in a StringIO)" if io.is_a?(String)}"
36
38
  end
37
- @metadata = read_footer
38
- @schema = Schema.from_elements(@metadata.schema)
39
+ keys = keys.to_sym if keys.is_a?(String)
40
+ raise ArgumentError, "keys: must be :string or :symbol, got #{keys.inspect}" unless KEY_MODES.include?(keys)
41
+ @symbolize = keys == :symbol
42
+ @zone_converter = zone_converter(time_zone)
43
+ @io = io
44
+ @file_metadata = read_footer
45
+ @schema = Schema.from_elements(@file_metadata.schema)
39
46
  end
40
47
 
41
- def self.from_string(bytes)
42
- new(StringIO.new(bytes.b))
43
- end
48
+ def num_rows = @file_metadata.num_rows
49
+ def row_groups = @file_metadata.row_groups
44
50
 
45
- def close
46
- @io.close
51
+ # The footer's key/value metadata as a Hash (what the writer's metadata: option stores)
52
+ def metadata
53
+ (@file_metadata.key_value_metadata || []).to_h { |kv| [kv.key, kv.value] }
47
54
  end
48
55
 
49
- def num_rows = @metadata.num_rows
50
- def row_groups = @metadata.row_groups
51
- def num_row_groups = @metadata.row_groups.size
52
- def created_by = @metadata.created_by
53
-
54
- def key_value_metadata
55
- (@metadata.key_value_metadata || []).to_h { |kv| [kv.key, kv.value] }
56
- end
56
+ # Yields batches of up to +size+ rows (all batches are full except the last one; batches span
57
+ # row groups). Only one page per column and the current batch are held in memory.
58
+ #
59
+ # as: :rows (default) yields an Array of row Hashes; as: :columns yields a Hash of top-level
60
+ # field name => Array of that field's values in the batch, which skips building a Hash per row
61
+ # and is noticeably faster when you process data column by column.
62
+ #
63
+ # where: only yields rows matching all conditions (see Reader::Filter). Row groups and pages
64
+ # that cannot match are skipped using statistics, bloom filters and the page index, and the
65
+ # remaining rows are checked one by one. Filtered columns need not be in +columns+.
66
+ # from: skips the first rows of the file (jumping over pages with the page index), and
67
+ # limit: stops after yielding that many rows.
68
+ def each_batch(size = DEFAULT_BATCH_SIZE, columns: nil, as: :rows, where: nil, from: nil, limit: nil)
69
+ return enum_for(:each_batch, size, columns: columns, as: as, where: where, from: from, limit: limit) unless block_given?
70
+ size = Integer(size)
71
+ raise ArgumentError, "Batch size must be positive, got #{size}" unless size.positive?
72
+ raise ArgumentError, "as: must be :rows or :columns, got #{as.inspect}" unless as == :rows || as == :columns
73
+ raise ArgumentError, "limit: must not be negative" if limit && limit.negative?
74
+ return self if limit&.zero?
75
+ columnar = as == :columns
76
+ symbolize = @symbolize
77
+ out_fields = select_fields(columns)
78
+ filter = where && !where.empty? ? Filter.new(@schema, where) : nil
79
+ fields = filter ? out_fields | filter.fields : out_fields
80
+ names = row_keys(out_fields, symbolize)
81
+ nout = out_fields.size
82
+ converters = @schema.columns.map { |col| converter_for(col) }
83
+ assemblers = fields.map { |f| Assembler.new(f, symbolize) }
84
+ left_to_yield = limit
85
+ pending = pending_rows = nil
86
+ emit = lambda do |data, k|
87
+ if columnar
88
+ if pending
89
+ pending.each_with_index { |col, j| col.concat(data[j]) }
90
+ else
91
+ pending = data
92
+ end
93
+ pending_rows = (pending_rows || 0) + k
94
+ if pending_rows >= size
95
+ yield names.zip(pending).to_h
96
+ pending = pending_rows = nil
97
+ end
98
+ else
99
+ rows = build_rows(names, data, k)
100
+ if pending
101
+ pending.concat(rows)
102
+ else
103
+ pending = rows
104
+ end
105
+ pending_rows = pending.size
106
+ if pending_rows >= size
107
+ yield pending
108
+ pending = pending_rows = nil
109
+ end
110
+ end
111
+ end
57
112
 
58
- # Yields each row as a Hash of top-level field name => value
59
- def each_row(columns: nil, &block)
60
- return enum_for(:each_row, columns: columns) unless block
61
- fields = select_fields(columns)
62
- row_groups.each_index do |rg|
63
- data = read_row_group_fields(rg, fields)
64
- n = row_groups[rg].num_rows
65
- names = fields.map(&:name)
66
- n.times do |i|
67
- row = {}
68
- names.each_with_index { |name, j| row[name] = data[j][i] }
69
- yield row
113
+ plan_rows(filter, from).each do |rg_index, ranges|
114
+ rg = row_groups[rg_index]
115
+ partial = ranges != [[0, rg.num_rows]]
116
+ cursors = fields.map do |f|
117
+ f.leaves.map do |col|
118
+ reader = ColumnChunkReader.new(@io, rg.columns.fetch(col.index), col, converter: converters[col.index], lazy: true)
119
+ reader.locations = page_index(rg_index, col)[1]&.page_locations if partial
120
+ [col.index, ColumnCursor.new(reader)]
121
+ end
122
+ end
123
+ ranges.each do |first, stop|
124
+ cursors.each { |cs| cs.each { |_, cursor| cursor.seek(first) } }
125
+ left = stop - first
126
+ while left.positive?
127
+ # Rows still missing from the current batch (pending_rows counts rows in both modes)
128
+ k = pending ? size - pending_rows : size
129
+ raise Error, "Internal error: batch needs #{k} rows" unless k.positive? # never loop without progress
130
+ k = left if left < k
131
+ data = assemblers.each_with_index.map do |asm, j|
132
+ asm.read_rows(k, cursors[j].to_h { |idx, cursor| [idx, cursor.take(k)] })
133
+ end
134
+ left -= k
135
+ kept = k
136
+ if filter
137
+ keep = filter.matching_rows(data, fields, k, symbolize)
138
+ kept = keep.size
139
+ next if kept.zero?
140
+ data = data.first(nout).map { |col| keep.map { |i| col[i] } } if kept < k
141
+ end
142
+ data = data.first(nout) if data.size > nout
143
+ if left_to_yield && kept >= left_to_yield
144
+ emit.call(data.map { |col| col.first(left_to_yield) }, left_to_yield)
145
+ yield(columnar ? names.zip(pending).to_h : pending) if pending
146
+ return self
147
+ end
148
+ left_to_yield -= kept if left_to_yield
149
+ emit.call(data, kept)
150
+ end
70
151
  end
71
152
  end
153
+ yield(columnar ? names.zip(pending).to_h : pending) if pending
72
154
  self
73
155
  end
74
- alias_method :each, :each_row
75
156
 
76
- # Rows as an Array of Hashes
77
- def rows(columns: nil) = each_row(columns: columns).to_a
157
+ # What a read with +where:+ / +from:+ would touch, without reading any data: an Array of
158
+ # { row_group:, rows:, ranges: [[first_row, end_row), ...] } for the row groups that are
159
+ # read. Row groups ruled out entirely are left out.
160
+ def scan_plan(where: nil, from: nil)
161
+ filter = where && !where.empty? ? Filter.new(@schema, where) : nil
162
+ plan_rows(filter, from).map do |rg_index, ranges|
163
+ { row_group: rg_index, rows: ranges.sum { |s, e| e - s }, ranges: ranges }
164
+ end
165
+ end
78
166
 
79
- # All values of a single top-level field, across all row groups
80
- def column(name)
81
- field = @schema.field(name) or raise ArgumentError, "No such column #{name.inspect}"
82
- row_groups.each_index.flat_map { |rg| read_row_group_fields(rg, [field]).first }
167
+ # Yields each row as a Hash of top-level field name => value. Takes the options of each_batch
168
+ # except as:.
169
+ def each_row(columns: nil, where: nil, from: nil, limit: nil, &block)
170
+ return enum_for(:each_row, columns: columns, where: where, from: from, limit: limit) unless block
171
+ each_batch(columns: columns, where: where, from: from, limit: limit) { |rows| rows.each(&block) }
172
+ self
83
173
  end
84
174
 
85
- # Hash of field name => Array of values, for the given row group
86
- def read_row_group(index, columns: nil)
87
- fields = select_fields(columns)
88
- fields.map(&:name).zip(read_row_group_fields(index, fields)).to_h
175
+ # Reads the whole file (or the selected rows) at once: an Array of row Hashes, or with
176
+ # as: :columns a Hash of top-level field name => Array of values. Takes the options of each_batch.
177
+ def read(columns: nil, as: :rows, where: nil, from: nil, limit: nil)
178
+ if as == :columns
179
+ out = row_keys(select_fields(columns), @symbolize).to_h { |name| [name, []] }
180
+ each_batch(65_536, columns: columns, as: :columns, where: where, from: from, limit: limit) do |batch|
181
+ batch.each { |name, values| out[name].concat(values) }
182
+ end
183
+ else
184
+ out = []
185
+ each_batch(columns: columns, as: as, where: where, from: from, limit: limit) { |rows| out.concat(rows) }
186
+ end
187
+ out
89
188
  end
90
189
 
91
- # Raw column data for a leaf column: [definition_levels, repetition_levels, values].
92
- # Levels are nil when the column's max level is 0.
93
- def read_column_chunk(row_group_index, column)
190
+ # Internal (used by reads with where:/from:): [ColumnIndex or nil, OffsetIndex or nil] of a
191
+ # leaf column in a row group
192
+ def page_index(row_group_index, column)
94
193
  column = @schema.column(column) unless column.is_a?(Schema::Column)
95
194
  raise ArgumentError, "No such leaf column" unless column
96
- chunk = row_groups.fetch(row_group_index).columns.fetch(column.index)
97
- ColumnChunkReader.new(@io, chunk, column).read
195
+ @page_indexes ||= {}
196
+ @page_indexes[[row_group_index, column.index]] ||= begin
197
+ chunk = row_groups.fetch(row_group_index).columns.fetch(column.index)
198
+ [read_struct(Format::ColumnIndex, chunk.column_index_offset, chunk.column_index_length),
199
+ read_struct(Format::OffsetIndex, chunk.offset_index_offset, chunk.offset_index_length)]
200
+ end
98
201
  end
99
202
 
100
203
  def inspect
101
- "#<#{self.class.name} rows=#{num_rows} row_groups=#{num_row_groups} created_by=#{created_by.inspect}>"
204
+ "#<#{self.class.name} rows=#{num_rows} row_groups=#{row_groups.size} created_by=#{@file_metadata.created_by.inspect}>"
102
205
  end
103
206
 
104
207
  private
105
208
 
209
+ # [[row_group_index, [[first_row, end_row), ...]], ...] to read, after ruling out row groups
210
+ # (statistics, bloom filters) and pages (page index), and skipping the first +from+ rows
211
+ def plan_rows(filter, from)
212
+ from = Integer(from || 0)
213
+ raise ArgumentError, "from: must not be negative" if from.negative?
214
+ offset = 0
215
+ plan = []
216
+ row_groups.each_with_index do |rg, i|
217
+ n = rg.num_rows
218
+ first = from - offset
219
+ offset += n
220
+ next if n.zero? || first >= n
221
+ ranges = [[first.positive? ? first : 0, n]]
222
+ if filter
223
+ next unless filter.row_group_may_match?(self, i)
224
+ ranges = Filter.intersect(ranges, filter.page_ranges(self, i))
225
+ end
226
+ plan << [i, ranges] unless ranges.empty?
227
+ end
228
+ plan
229
+ end
230
+
231
+ def read_struct(klass, offset, length)
232
+ return nil unless offset && length&.positive?
233
+ @io.seek(offset)
234
+ bytes = @io.read(length)
235
+ return nil unless bytes&.bytesize == length
236
+ klass.decode(bytes.b).first
237
+ rescue Thrift::Error
238
+ nil # a damaged index only means pages cannot be skipped
239
+ end
240
+
106
241
  def select_fields(columns)
107
242
  return @schema.fields unless columns
108
243
  Array(columns).map { |c| @schema.field(c) or raise ArgumentError, "No such column #{c.inspect}" }
109
244
  end
110
245
 
111
- def read_row_group_fields(rg, fields)
112
- rg_meta = row_groups.fetch(rg)
113
- n = rg_meta.num_rows
114
- fields.map do |field|
115
- chunks = {}
116
- field.leaves.each do |col|
117
- chunks[col.index] = ColumnChunkReader.new(@io, rg_meta.columns.fetch(col.index), col).read
246
+ def row_keys(fields, symbolize)
247
+ fields.map { |f| symbolize ? f.name.to_sym : -f.name }
248
+ end
249
+
250
+ def build_rows(names, data, k)
251
+ return Array.new(k) { {} } if names.empty?
252
+ return data.first.map { |v| { names.first => v } } if names.size == 1
253
+ nf = names.size
254
+ Array.new(k) do |i|
255
+ row = {}
256
+ j = 0
257
+ while j < nf
258
+ row[names[j]] = data[j][i]
259
+ j += 1
118
260
  end
119
- Assembler.new(field, chunks).read_rows(n)
261
+ row
120
262
  end
121
263
  end
122
264
 
265
+ # The column's value converter, with the time zone applied to timestamps
266
+ def converter_for(column)
267
+ base = column.converter
268
+ zc = @zone_converter
269
+ return base unless zc && base && instant_column?(column)
270
+ ->(v) { zc.call(base.call(v)) }
271
+ end
272
+
273
+ # Timestamps that denote an instant (UTC-adjusted TIMESTAMP, INT96). Local timestamps
274
+ # (isAdjustedToUTC = false) are wall-clock values and are left as they are.
275
+ def instant_column?(column)
276
+ return true if column.type == Format::Type::INT96
277
+ kind, _unit, utc = Types.logical_of(column.node)
278
+ kind == :timestamp && utc
279
+ end
280
+
281
+ OFFSET_PATTERN = /\A([+-])(\d\d)(?::?(\d\d)(?::?(\d\d))?)?\z/
282
+
283
+ # A lambda turning a UTC Time into the zone, or nil for UTC
284
+ def zone_converter(zone)
285
+ case zone
286
+ when nil then nil
287
+ when Integer
288
+ return nil if zone.zero?
289
+ Time.at(0).getlocal(zone) # validates the range
290
+ ->(t) { t.getlocal(zone) }
291
+ when String
292
+ return nil if zone.match?(/\A(?:utc|z)\z/i)
293
+ if (m = OFFSET_PATTERN.match(zone))
294
+ # As seconds, since Ruby 3.0 only parses "+HH:MM" offset strings
295
+ return zone_converter((m[1] == "-" ? -1 : 1) * (m[2].to_i * 3600 + m[3].to_i * 60 + m[4].to_i))
296
+ end
297
+ if defined?(::ActiveSupport::TimeZone) && (tz = ::ActiveSupport::TimeZone[zone])
298
+ return ->(t) { tz.at(t) }
299
+ end
300
+ if defined?(::TZInfo::Timezone)
301
+ tz = ::TZInfo::Timezone.get(zone)
302
+ return ->(t) { t.getlocal(tz) }
303
+ end
304
+ raise ArgumentError, "Unknown time zone #{zone.inspect}: use a UTC offset like \"+02:00\", " \
305
+ "a timezone object (e.g. TZInfo::Timezone.get(#{zone.inspect})) or an ActiveSupport::TimeZone"
306
+ else
307
+ return ->(t) { zone.at(t) } if zone.respond_to?(:at)
308
+ if zone.respond_to?(:utc_to_local)
309
+ Time.at(0).getlocal(zone)
310
+ return ->(t) { t.getlocal(zone) }
311
+ end
312
+ raise ArgumentError, "Unsupported time_zone: #{zone.inspect}"
313
+ end
314
+ rescue ArgumentError => e
315
+ raise if e.message.start_with?("Unknown time zone", "Unsupported time_zone", "Invalid time_zone")
316
+ raise ArgumentError, "Invalid time_zone #{zone.inspect}: #{e.message}"
317
+ end
318
+
123
319
  def read_footer
124
320
  @io.seek(0, IO::SEEK_END)
125
321
  size = @io.pos
@@ -137,199 +333,18 @@ module Herringbone
137
333
  raise FormatError, "Corrupt file metadata: #{e.message}"
138
334
  end
139
335
 
140
- # Decodes all pages of one column chunk into levels and values
141
- class ColumnChunkReader
142
- T = Format::Type
143
- E = Format::Encoding
144
-
145
- def initialize(io, chunk, column)
146
- @io = io
147
- @chunk = chunk
148
- @column = column
149
- @meta = chunk.meta_data or raise UnsupportedError, "Column chunk without metadata (encrypted?)"
150
- raise UnsupportedError, "Column chunks in external files are not supported" if chunk.file_path
151
- @max_def = column.max_definition_level
152
- @max_rep = column.max_repetition_level
153
- @converter = column.converter
154
- end
155
-
156
- def read
157
- buf = read_bytes
158
- defs = @max_def.positive? ? [] : nil
159
- reps = @max_rep.positive? ? [] : nil
160
- values = []
161
- pos = 0
162
- total = @meta.num_values
163
- seen = 0
164
- while seen < total && pos < buf.bytesize
165
- header, pos = decode_page_header(buf, pos)
166
- size = header.compressed_page_size
167
- # Some old writers under-report total_compressed_size; read on past the declared end
168
- extend_buffer(buf, pos + size - buf.bytesize) if pos + size > buf.bytesize
169
- raise FormatError, "Page overruns column chunk" if pos + size > buf.bytesize
170
- body = buf.byteslice(pos, size)
171
- pos += size
172
- case header.type
173
- when Format::PageType::DICTIONARY_PAGE
174
- read_dictionary(header, body)
175
- when Format::PageType::DATA_PAGE
176
- seen += read_data_page_v1(header, body, defs, reps, values)
177
- when Format::PageType::DATA_PAGE_V2
178
- seen += read_data_page_v2(header, body, defs, reps, values)
179
- end
180
- end
181
- raise FormatError, "Column #{@column.dotted_path}: read #{seen} of #{total} values" if seen < total
182
- [defs, reps, values]
183
- rescue Thrift::Error => e
184
- raise FormatError, "Corrupt page header in #{@column.dotted_path}: #{e.message}"
185
- end
186
-
187
- private
188
-
189
- def read_bytes
190
- start = @meta.data_page_offset
191
- dict = @meta.dictionary_page_offset
192
- # Some writers store 0 when there is no dictionary page
193
- start = dict if dict && dict.positive? && dict < start
194
- @start = start
195
- @io.seek(start)
196
- len = @meta.total_compressed_size
197
- (@io.read(len) || "".b).b
198
- end
199
-
200
- def extend_buffer(buf, nbytes)
201
- @io.seek(@start + buf.bytesize)
202
- more = @io.read(nbytes)
203
- buf << more.b if more
204
- end
205
-
206
- def decode_page_header(buf, pos)
207
- Format::PageHeader.decode(buf, pos)
208
- rescue Thrift::Error
209
- # The header may straddle the declared end of the chunk
210
- before = buf.bytesize
211
- extend_buffer(buf, 1024)
212
- raise if buf.bytesize == before
213
- Format::PageHeader.decode(buf, pos)
214
- end
215
-
216
- def decompress(body, size)
217
- Compression.decompress(@meta.codec, body, size)
218
- end
219
-
220
- def read_dictionary(header, body)
221
- dh = header.dictionary_page_header
222
- data = decompress(body, header.uncompressed_page_size)
223
- vals, = Encodings::Plain.decode(data, 0, dh.num_values, @column.type, @column.type_length)
224
- vals.map!(&@converter) if @converter
225
- @dictionary = vals
226
- end
227
-
228
- def read_data_page_v1(header, body, defs, reps, values)
229
- dh = header.data_page_header
230
- n = dh.num_values
231
- data = decompress(body, header.uncompressed_page_size)
232
- pos = 0
233
- if @max_rep.positive?
234
- levels, pos = read_levels(data, pos, dh.repetition_level_encoding, @max_rep, n)
235
- reps.concat(levels)
236
- end
237
- non_null = n
238
- if @max_def.positive?
239
- levels, pos = read_levels(data, pos, dh.definition_level_encoding, @max_def, n)
240
- non_null = levels.count(@max_def)
241
- defs.concat(levels)
242
- end
243
- values.concat(decode_values(data, pos, non_null, dh.encoding))
244
- n
245
- end
246
-
247
- def read_data_page_v2(header, body, defs, reps, values)
248
- dh = header.data_page_header_v2
249
- n = dh.num_values
250
- rep_len = dh.repetition_levels_byte_length
251
- def_len = dh.definition_levels_byte_length
252
- if @max_rep.positive?
253
- reps.concat(Encodings::RLE.decode_hybrid(body, 0, rep_len, RLE_WIDTH[@max_rep], n))
254
- end
255
- non_null = n
256
- if @max_def.positive?
257
- levels = Encodings::RLE.decode_hybrid(body, rep_len, rep_len + def_len, RLE_WIDTH[@max_def], n)
258
- non_null = levels.count(@max_def)
259
- defs.concat(levels)
260
- end
261
- data = body.byteslice(rep_len + def_len, body.bytesize - rep_len - def_len)
262
- if dh.is_compressed != false
263
- data = decompress(data, header.uncompressed_page_size - rep_len - def_len)
264
- end
265
- values.concat(decode_values(data, 0, non_null, dh.encoding))
266
- n
267
- end
268
-
269
- RLE_WIDTH = Hash.new { |h, k| h[k] = k.bit_length }
270
-
271
- def read_levels(data, pos, encoding, max, n)
272
- width = RLE_WIDTH[max]
273
- case encoding
274
- when E::RLE
275
- len = data.byteslice(pos, 4).unpack1("V")
276
- start = pos + 4
277
- [Encodings::RLE.decode_hybrid(data, start, start + len, width, n), start + len]
278
- when E::BIT_PACKED
279
- [Encodings::RLE.decode_legacy_bit_packed(data, pos, width, n), pos + (n * width + 7) / 8]
280
- else
281
- raise UnsupportedError, "Unsupported level encoding #{E::NAMES[encoding] || encoding}"
282
- end
283
- end
284
-
285
- def decode_values(data, pos, count, encoding)
286
- type = @column.type
287
- vals = case encoding
288
- when E::PLAIN
289
- Encodings::Plain.decode(data, pos, count, type, @column.type_length).first
290
- when E::PLAIN_DICTIONARY, E::RLE_DICTIONARY
291
- raise FormatError, "Dictionary-encoded page without a dictionary in #{@column.dotted_path}" unless @dictionary
292
- return [] if count.zero?
293
- width = data.getbyte(pos)
294
- indices = Encodings::RLE.decode_hybrid(data, pos + 1, data.bytesize, width, count)
295
- dict = @dictionary
296
- raise FormatError, "Dictionary index out of range in #{@column.dotted_path}" if indices.max >= dict.size
297
- return indices.map! { |i| dict[i] }
298
- when E::RLE
299
- raise UnsupportedError, "RLE value encoding is only supported for BOOLEAN" unless type == T::BOOLEAN
300
- len = data.byteslice(pos, 4).unpack1("V")
301
- Encodings::RLE.decode_hybrid(data, pos + 4, pos + 4 + len, 1, count).map! { |v| v == 1 }
302
- when E::DELTA_BINARY_PACKED
303
- bits = type == T::INT32 ? 32 : 64
304
- vals, = Encodings::Delta.decode_binary_packed(data, pos, bits, count)
305
- raise FormatError, "DELTA_BINARY_PACKED page has #{vals.size} values, need #{count}" if vals.size < count
306
- vals
307
- when E::DELTA_LENGTH_BYTE_ARRAY
308
- Encodings::Delta.decode_length_byte_array(data, pos, count).first
309
- when E::DELTA_BYTE_ARRAY
310
- Encodings::Delta.decode_byte_array(data, pos, count).first
311
- when E::BYTE_STREAM_SPLIT
312
- width = case type
313
- when T::INT32, T::FLOAT then 4
314
- when T::INT64, T::DOUBLE then 8
315
- when T::FIXED_LEN_BYTE_ARRAY then @column.type_length
316
- else raise UnsupportedError, "BYTE_STREAM_SPLIT is not valid for #{T::NAMES[type]}"
317
- end
318
- plain, = Encodings::ByteStreamSplit.decode(data, pos, count, width)
319
- Encodings::Plain.decode(plain, 0, count, type, @column.type_length).first
320
- else
321
- raise UnsupportedError, "Unsupported encoding #{E::NAMES[encoding] || encoding}"
322
- end
323
- vals.map!(&@converter) if @converter
324
- vals
325
- end
326
- end
327
-
328
336
  # Rebuilds nested values of one top-level field from the levels of its leaf columns
329
337
  # (the "record assembly" half of the Dremel algorithm).
330
338
  class Assembler
331
- def initialize(field, chunks)
339
+ def initialize(field, symbolize = false)
332
340
  @field = field
341
+ @symbolize = symbolize
342
+ @keys = {} # struct child Field => Hash key
343
+ end
344
+
345
+ # Assembles +n+ values of the field from +chunks+ (leaf column index =>
346
+ # [defs, reps, values] holding exactly those rows)
347
+ def read_rows(n, chunks)
333
348
  @defs = {}
334
349
  @reps = {}
335
350
  @vals = {}
@@ -340,9 +355,7 @@ module Herringbone
340
355
  end
341
356
  @ei = Hash.new(0) # entry cursor per leaf column
342
357
  @vi = Hash.new(0) # value cursor per leaf column
343
- end
344
358
 
345
- def read_rows(n)
346
359
  f = @field
347
360
  if f.leaf? && f.column.max_repetition_level.zero?
348
361
  idx = f.column.index
@@ -407,6 +420,10 @@ module Herringbone
407
420
  out
408
421
  end
409
422
 
423
+ def key_for(field)
424
+ @keys[field] ||= @symbolize ? field.name.to_sym : -field.name
425
+ end
426
+
410
427
  def read(field)
411
428
  c = field.first_leaf.index
412
429
  kind = field.kind
@@ -427,7 +444,7 @@ module Herringbone
427
444
  @vals[c][v]
428
445
  when :struct
429
446
  h = {}
430
- field.children.each { |ch| h[ch.name] = read(ch) }
447
+ field.children.each { |ch| h[key_for(ch)] = read(ch) }
431
448
  h
432
449
  when :list
433
450
  if d < field.item_def
@@ -467,3 +484,8 @@ module Herringbone
467
484
  end
468
485
  end
469
486
  end
487
+
488
+ require_relative "reader/page_stream"
489
+ require_relative "reader/column_chunk_reader"
490
+ require_relative "reader/column_cursor"
491
+ require_relative "reader/scan"