herringbone 0.1.0 → 0.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -5,6 +5,8 @@ module Herringbone
5
5
  # It still carries an "experimental" warning, which is silenced once here: the gem only uses
6
6
  # new/for/copy/get_string/free, and falls back to String operations where it is missing.
7
7
  module IOBufferSupport
8
+ # true when IO::Buffer has the methods the decompressors use and HERRINGBONE_NO_IO_BUFFER is
9
+ # not set in the environment (which forces the String fallback)
8
10
  AVAILABLE = begin
9
11
  if defined?(IO::Buffer) && IO::Buffer.method_defined?(:copy) && IO::Buffer.method_defined?(:get_string)
10
12
  previous = Warning[:experimental]
@@ -18,7 +20,7 @@ module Herringbone
18
20
  else
19
21
  false
20
22
  end
21
- rescue StandardError
23
+ rescue
22
24
  false
23
25
  end
24
26
  end
@@ -0,0 +1,388 @@
1
+ # frozen_string_literal: true
2
+
3
+ module Herringbone
4
+ class Reader
5
+ # Decodes the pages of one column chunk, one page at a time. Pages are read from the IO as
6
+ # they are needed (a page header, then its body), so memory use is bounded by the page size
7
+ # rather than the size of the chunk.
8
+ #
9
+ # reader = ColumnChunkReader.new(io, chunk, column)
10
+ # while (page = reader.next_page)
11
+ # defs, reps, values = page # levels are nil when the column's max level is 0
12
+ # end
13
+ #
14
+ # The declared total_compressed_size of the chunk is not relied on (some old writers
15
+ # under-report it): pages are read until the chunk's num_values have been seen.
16
+ class ColumnChunkReader
17
+ # Shorthand for the physical type constants
18
+ T = Format::Type
19
+ # Shorthand for the encoding constants
20
+ E = Format::Encoding
21
+
22
+ # Bytes read past a page body, so that the next page header usually needs no extra read
23
+ READ_AHEAD = 64 * 1024
24
+ # Bytes read for a page header whose size is not known; grown 4x while decoding fails
25
+ HEADER_GUESS = 1024
26
+ # Largest header read attempted before a page header is declared corrupt
27
+ MAX_HEADER = 64 * 1024 * 1024
28
+
29
+ # @return [Schema::Column] the leaf column this chunk belongs to
30
+ attr_reader :column
31
+
32
+ # With +lazy+, values of data pages that are not dictionary-encoded are returned as
33
+ # physical values, and #page_converter is what still has to be applied to them. This keeps
34
+ # a decoded page small (Integers instead of Time or BigDecimal objects) when only a slice
35
+ # of it is needed at a time.
36
+ #
37
+ # @return [Proc, nil] converter of the last page returned by #next_page, nil when none
38
+ attr_reader :page_converter
39
+
40
+ # @param io [IO, StringIO] the file, read with #seek and #read
41
+ # @param chunk [Format::ColumnChunk] the chunk's footer entry
42
+ # @param column [Schema::Column] the leaf column the chunk stores
43
+ # @param converter [Proc, nil] physical value => Ruby value, applied to values and dictionaries
44
+ # @param lazy [Boolean] leave non-dictionary values physical in #next_page (see #page_converter)
45
+ # @raise [UnsupportedError] for chunks without metadata (encrypted) or stored in another file
46
+ def initialize(io, chunk, column, converter: column.converter, lazy: false)
47
+ @io = io
48
+ @chunk = chunk
49
+ @column = column
50
+ @meta = chunk.meta_data or raise UnsupportedError, "Column chunk without metadata (encrypted?)"
51
+ raise UnsupportedError, "Column chunks in external files are not supported" if chunk.file_path
52
+ @max_def = column.max_definition_level
53
+ @max_rep = column.max_repetition_level
54
+ @converter = converter
55
+ @lazy = lazy
56
+ start = @meta.data_page_offset
57
+ dict = @meta.dictionary_page_offset
58
+ # Some writers store 0 when there is no dictionary page
59
+ start = dict if dict&.positive? && dict < start
60
+ @pos = start
61
+ @start = start
62
+ @locations = nil # OffsetIndex page locations, when jumping between pages
63
+ @page_number = nil
64
+ @total = @meta.num_values
65
+ @seen = 0
66
+ @buf = "".b
67
+ @buf_pos = start
68
+ end
69
+
70
+ # All pages concatenated: [definition_levels, repetition_levels, values]
71
+ #
72
+ # @return [Array(Array<Integer>, Array<Integer>, Array)] levels are nil when the column's
73
+ # max level is 0
74
+ # @raise [FormatError] when a page is corrupt or overruns the file
75
+ def read
76
+ defs = @max_def.positive? ? [] : nil
77
+ reps = @max_rep.positive? ? [] : nil
78
+ values = []
79
+ while (page = next_page)
80
+ d, r, v = page
81
+ defs&.concat(d)
82
+ reps&.concat(r)
83
+ values.concat(v)
84
+ end
85
+ [defs, reps, values]
86
+ end
87
+
88
+ # The next data page as [defs, reps, values], or nil after the last one
89
+ #
90
+ # @return [Array(Array<Integer>, Array<Integer>, Array), nil] levels are nil when the
91
+ # column's max level is 0; values hold the non-null entries
92
+ # @raise [FormatError] when a page is corrupt or overruns the file
93
+ def next_page
94
+ page = next_stream or return nil
95
+ n = page.remaining
96
+ defs, reps = page.read_levels(n)
97
+ non_null = defs ? defs.count(@max_def) : n
98
+ values = page.read_values(non_null)
99
+ if (conv = page.converter)
100
+ if @lazy
101
+ @page_converter = conv
102
+ else
103
+ values.map!(&conv)
104
+ end
105
+ else
106
+ @page_converter = nil
107
+ end
108
+ [defs, reps, values]
109
+ end
110
+
111
+ # The next data page as a PageStream::Page that decodes its levels and values on demand,
112
+ # or nil after the last one. Values come out physical; apply Page#converter to them.
113
+ # Dictionary pages are read on the way.
114
+ #
115
+ # @return [PageStream::Page, nil] the next data page
116
+ # @raise [FormatError] when a page header is corrupt or a page overruns the file
117
+ # @raise [UnsupportedError] for an unsupported encoding or compression codec
118
+ def next_stream
119
+ while more_pages?
120
+ header, body = read_page
121
+ case header.type
122
+ when Format::PageType::DICTIONARY_PAGE
123
+ read_dictionary(header, body)
124
+ when Format::PageType::DATA_PAGE
125
+ page = data_page_v1(header, body)
126
+ @seen += header.data_page_header.num_values
127
+ @page_number += 1 if @page_number
128
+ return page
129
+ when Format::PageType::DATA_PAGE_V2
130
+ page = data_page_v2(header, body)
131
+ @seen += header.data_page_header_v2.num_values
132
+ @page_number += 1 if @page_number
133
+ return page
134
+ end
135
+ end
136
+ nil
137
+ rescue Thrift::Error => e
138
+ raise FormatError, "Corrupt page header in #{@column.dotted_path}: #{e.message}"
139
+ end
140
+
141
+ # @return [Array<Format::PageLocation>, nil] the chunk's data page locations from its
142
+ # OffsetIndex (enables #jump_to_page)
143
+ attr_accessor :locations
144
+
145
+ # Continues reading at data page +index+ of the OffsetIndex. The dictionary page (which
146
+ # the OffsetIndex does not list) is read first if it has not been yet.
147
+ #
148
+ # @param index [Integer] position of the page in #locations
149
+ # @return [void]
150
+ # @raise [ArgumentError] when no #locations are set
151
+ # @raise [IndexError] when +index+ is outside #locations
152
+ def jump_to_page(index)
153
+ raise ArgumentError, "No OffsetIndex for #{@column.dotted_path}" unless @locations
154
+ load_dictionary
155
+ @pos = @locations.fetch(index).offset
156
+ @page_number = index
157
+ end
158
+
159
+ # Whether all of the chunk's values have been returned
160
+ #
161
+ # @return [Boolean] true once num_values entries have been seen
162
+ def done? = @seen >= @total
163
+
164
+ # @return [Integer] entries (levels, nulls included) of the data pages returned so far
165
+ attr_reader :seen
166
+
167
+ # @return [Integer] the chunk's num_values from its ColumnMetaData
168
+ attr_reader :total
169
+
170
+ private
171
+
172
+ # Whether another data page is due: up to the last OffsetIndex location after a jump,
173
+ # otherwise until num_values entries have been seen
174
+ #
175
+ # @return [Boolean] true when #next_stream should read on
176
+ def more_pages?
177
+ @page_number ? @page_number < @locations.size : @seen < @total
178
+ end
179
+
180
+ # Reads the dictionary page at the start of the chunk, if there is one
181
+ #
182
+ # @return [void]
183
+ def load_dictionary
184
+ return if @dictionary || @dictionary_checked
185
+ @dictionary_checked = true
186
+ first_data = @locations.first&.offset
187
+ return if first_data.nil? || @start >= first_data
188
+ saved = @pos
189
+ @pos = @start
190
+ header, body = read_page
191
+ read_dictionary(header, body) if header.type == Format::PageType::DICTIONARY_PAGE
192
+ @pos = saved
193
+ end
194
+
195
+ # Reads the page header at @pos and the page body after it
196
+ #
197
+ # @return [Array(Format::PageHeader, String)] the header and the (still compressed) body
198
+ # @raise [FormatError] for a negative page size or a page that overruns the file
199
+ def read_page
200
+ # With an OffsetIndex the page's size (header included) is known, so read exactly that
201
+ loc = @page_number && @locations[@page_number]
202
+ exact = loc && loc.offset == @pos && loc.compressed_page_size.positive?
203
+ want = exact ? loc.compressed_page_size : HEADER_GUESS
204
+ begin
205
+ buf, off = window(@pos, want)
206
+ header, hend = Format::PageHeader.decode(buf, off)
207
+ rescue Thrift::Error
208
+ # The header may be larger than guessed (long statistics) or be cut off by EOF
209
+ raise if buf.bytesize - off < want || want >= MAX_HEADER
210
+ want *= 4
211
+ retry
212
+ end
213
+ hlen = hend - off
214
+ size = header.compressed_page_size
215
+ raise FormatError, "Negative page size in #{@column.dotted_path}" if size.nil? || size.negative?
216
+ buf, off = window(@pos + hlen, size + (exact ? 0 : READ_AHEAD), size)
217
+ if buf.bytesize - off < size
218
+ raise FormatError, "Column #{@column.dotted_path}: page overruns the file (read #{@seen} of #{@total} values)"
219
+ end
220
+ body = (off.zero? && buf.bytesize == size) ? buf : buf.byteslice(off, size)
221
+ @pos += hlen + size
222
+ [header, body]
223
+ end
224
+
225
+ # Returns [buffer, offset] where buffer[offset..] holds at least +need+ bytes from file
226
+ # position +pos+ (fewer only at EOF), reading +len+ bytes when the buffer does not cover it.
227
+ #
228
+ # @param pos [Integer] file offset wanted
229
+ # @param len [Integer] bytes to read when the buffer has to be refilled
230
+ # @param need [Integer] bytes that must be available from +pos+ to reuse the buffer
231
+ # @return [Array(String, Integer)] binary buffer and the offset of +pos+ in it
232
+ def window(pos, len, need = len)
233
+ off = pos - @buf_pos
234
+ return [@buf, off] if off >= 0 && off + need <= @buf.bytesize
235
+ @io.seek(pos)
236
+ @buf = @io.read(len) || "".b
237
+ @buf.force_encoding(Encoding::BINARY)
238
+ @buf_pos = pos
239
+ [@buf, 0]
240
+ end
241
+
242
+ # Decompresses a page body with the chunk's codec
243
+ #
244
+ # @param body [String] compressed bytes
245
+ # @param size [Integer] uncompressed size from the page header
246
+ # @return [String] uncompressed bytes
247
+ # @raise [UnsupportedError] for an unsupported codec, naming the column
248
+ def decompress(body, size)
249
+ Compression.decompress(@meta.codec, body, size)
250
+ rescue UnsupportedError => e
251
+ raise e, "#{e.message} (column #{@column.dotted_path})"
252
+ end
253
+
254
+ # Decodes a dictionary page (always PLAIN) and keeps its converted values
255
+ #
256
+ # @param header [Format::PageHeader] the dictionary page header
257
+ # @param body [String] the compressed page body
258
+ # @return [Array] the dictionary values
259
+ def read_dictionary(header, body)
260
+ dh = header.dictionary_page_header
261
+ data = decompress(body, header.uncompressed_page_size)
262
+ vals, = Encodings::Plain.decode(data, 0, dh.num_values, @column.type, @column.type_length)
263
+ vals.map!(&@converter) if @converter
264
+ @dictionary = vals
265
+ end
266
+
267
+ # A DATA_PAGE: the whole body is compressed, levels come first with their own length prefix
268
+ #
269
+ # @param header [Format::PageHeader] the data page header
270
+ # @param body [String] the compressed page body
271
+ # @return [PageStream::Page] the page, decoded on demand
272
+ def data_page_v1(header, body)
273
+ dh = header.data_page_header
274
+ n = dh.num_values
275
+ data = decompress(body, header.uncompressed_page_size)
276
+ pos = 0
277
+ reps = defs = nil
278
+ reps, pos = level_decoder(data, pos, dh.repetition_level_encoding, @max_rep, n) if @max_rep.positive?
279
+ defs, pos = level_decoder(data, pos, dh.definition_level_encoding, @max_def, n) if @max_def.positive?
280
+ values, conv = value_decoder(data, pos, dh.encoding)
281
+ PageStream::Page.new(n, defs, reps, values, conv)
282
+ end
283
+
284
+ # A DATA_PAGE_V2: levels are stored uncompressed before the (optionally compressed) values
285
+ #
286
+ # @param header [Format::PageHeader] the data page header
287
+ # @param body [String] the page body
288
+ # @return [PageStream::Page] the page, decoded on demand
289
+ def data_page_v2(header, body)
290
+ dh = header.data_page_header_v2
291
+ n = dh.num_values
292
+ rep_len = dh.repetition_levels_byte_length
293
+ def_len = dh.definition_levels_byte_length
294
+ reps = defs = nil
295
+ reps = PageStream::HybridDecoder.new(body, 0, rep_len, RLE_WIDTH[@max_rep]) if @max_rep.positive?
296
+ defs = PageStream::HybridDecoder.new(body, rep_len, rep_len + def_len, RLE_WIDTH[@max_def]) if @max_def.positive?
297
+ data = body.byteslice(rep_len + def_len, body.bytesize - rep_len - def_len)
298
+ if dh.is_compressed != false
299
+ data = decompress(data, header.uncompressed_page_size - rep_len - def_len)
300
+ end
301
+ values, conv = value_decoder(data, 0, dh.encoding)
302
+ PageStream::Page.new(n, defs, reps, values, conv)
303
+ end
304
+
305
+ # Bit width of levels up to a max level (memoized Integer#bit_length)
306
+ RLE_WIDTH = Hash.new { |h, k| h[k] = k.bit_length }
307
+
308
+ # [decoder, position after the levels]
309
+ #
310
+ # @param data [String] the uncompressed page
311
+ # @param pos [Integer] offset of the levels in +data+
312
+ # @param encoding [Integer] Format::Encoding of the levels (RLE or BIT_PACKED)
313
+ # @param max [Integer] max level of the column, which sets the bit width
314
+ # @param n [Integer] number of entries in the page
315
+ # @return [Array(PageStream::HybridDecoder, Integer), Array(PageStream::ArrayDecoder, Integer)]
316
+ # the decoder and the offset just past the levels
317
+ # @raise [FormatError] when the RLE length prefix is cut off
318
+ # @raise [UnsupportedError] for any other level encoding
319
+ def level_decoder(data, pos, encoding, max, n)
320
+ width = RLE_WIDTH[max]
321
+ case encoding
322
+ when E::RLE
323
+ len = data.byteslice(pos, 4)&.unpack1("V") or raise FormatError, "Truncated levels"
324
+ start = pos + 4
325
+ [PageStream::HybridDecoder.new(data, start, start + len, width), start + len]
326
+ when E::BIT_PACKED
327
+ levels = Encodings::RLE.decode_legacy_bit_packed(data, pos, width, n)
328
+ [PageStream::ArrayDecoder.new(levels), pos + (n * width + 7) / 8]
329
+ else
330
+ raise UnsupportedError, "Unsupported level encoding #{E::NAMES[encoding] || encoding}"
331
+ end
332
+ end
333
+
334
+ # Physical type => [String#unpack directive, byte width] for PLAIN fixed-width values
335
+ FIXED_FORMATS = {
336
+ T::INT32 => ["l<", 4], T::INT64 => ["q<", 8], T::FLOAT => ["e", 4], T::DOUBLE => ["E", 8]
337
+ }.freeze
338
+
339
+ # [value decoder, converter still to apply to its values (nil for dictionary pages)]
340
+ #
341
+ # @param data [String] the uncompressed values section
342
+ # @param pos [Integer] offset of the values in +data+
343
+ # @param encoding [Integer] Format::Encoding of the values
344
+ # @return [Array(Object, Proc)] a PageStream decoder (responds to #read) and the converter,
345
+ # which may be nil
346
+ # @raise [FormatError] for a dictionary-encoded page in a chunk without a dictionary page
347
+ # @raise [UnsupportedError] for an encoding not valid for the column's type, or unknown
348
+ def value_decoder(data, pos, encoding)
349
+ type = @column.type
350
+ decoder = case encoding
351
+ when E::PLAIN
352
+ case type
353
+ when T::BOOLEAN then PageStream::BooleanDecoder.new(data, pos)
354
+ when T::INT96 then PageStream::Int96Decoder.new(data, pos)
355
+ when T::BYTE_ARRAY then PageStream::ByteArrayDecoder.new(data, pos)
356
+ when T::FIXED_LEN_BYTE_ARRAY then PageStream::FixedBytesDecoder.new(data, pos, @column.type_length)
357
+ else PageStream::FixedDecoder.new(data, pos, *FIXED_FORMATS.fetch(type))
358
+ end
359
+ when E::PLAIN_DICTIONARY, E::RLE_DICTIONARY
360
+ raise FormatError, "Dictionary-encoded page without a dictionary in #{@column.dotted_path}" unless @dictionary
361
+ # Dictionary values are converted once, when the dictionary page is read
362
+ return [PageStream::DictionaryDecoder.new(data, pos, @dictionary, @column.dotted_path), nil]
363
+ when E::RLE
364
+ raise UnsupportedError, "RLE value encoding is only supported for BOOLEAN" unless type == T::BOOLEAN
365
+ PageStream::RleBooleanDecoder.new(data, pos)
366
+ when E::DELTA_BINARY_PACKED
367
+ bits = (type == T::INT32) ? 32 : 64
368
+ PageStream::ArrayDecoder.new(Encodings::Delta.decode_binary_packed(data, pos, bits).first)
369
+ when E::DELTA_LENGTH_BYTE_ARRAY
370
+ PageStream::DeltaLengthDecoder.new(data, pos)
371
+ when E::DELTA_BYTE_ARRAY
372
+ PageStream::DeltaByteArrayDecoder.new(data, pos)
373
+ when E::BYTE_STREAM_SPLIT
374
+ width = case type
375
+ when T::INT32, T::FLOAT then 4
376
+ when T::INT64, T::DOUBLE then 8
377
+ when T::FIXED_LEN_BYTE_ARRAY then @column.type_length
378
+ else raise UnsupportedError, "BYTE_STREAM_SPLIT is not valid for #{T::NAMES[type]}"
379
+ end
380
+ PageStream::ByteStreamSplitDecoder.new(data, pos, width, type, @column.type_length)
381
+ else
382
+ raise UnsupportedError, "Unsupported encoding #{E::NAMES[encoding] || encoding}"
383
+ end
384
+ [decoder, @converter]
385
+ end
386
+ end
387
+ end
388
+ end
@@ -0,0 +1,225 @@
1
+ # frozen_string_literal: true
2
+
3
+ module Herringbone
4
+ class Reader
5
+ # Walks the pages of one leaf column chunk and hands out the entries (levels and values)
6
+ # of the next +k+ rows. Pages are decoded incrementally (see PageStream), so only the
7
+ # entries of the requested rows (plus a small lookahead of levels for repeated columns)
8
+ # become Ruby objects. A row starts at an entry with repetition level 0 and may continue
9
+ # over any number of following pages.
10
+ class ColumnCursor
11
+ # Levels decoded ahead at a time when looking for row starts in repeated columns
12
+ LOOKAHEAD = 4096
13
+
14
+ # @param chunk_reader [ColumnChunkReader] reader of the chunk, positioned at its start
15
+ def initialize(chunk_reader)
16
+ @src = chunk_reader
17
+ col = chunk_reader.column
18
+ @path = col.dotted_path
19
+ @max_def = col.max_definition_level
20
+ @repeated = col.max_repetition_level.positive?
21
+ @page = nil
22
+ # Repeated columns: levels decoded from the current page but not handed out yet
23
+ @bd = @br = nil
24
+ @bi = 0
25
+ @started = false
26
+ @row = 0 # rows handed out or skipped so far
27
+ @page_idx = -1 # index of the current data page within the chunk
28
+ end
29
+
30
+ # @return [Integer] rows handed out or skipped so far (the chunk row the cursor is at)
31
+ attr_reader :row
32
+
33
+ # Moves forward to row +target+ of the chunk (0-based). With an OffsetIndex, pages before
34
+ # the one holding +target+ are not read at all; otherwise rows are skipped page by page,
35
+ # decoding levels but not building values.
36
+ #
37
+ # @param target [Integer] row of the chunk to stop at
38
+ # @return [void]
39
+ # @raise [ArgumentError] when +target+ is before the current row
40
+ # @raise [FormatError] when the chunk runs out of pages before +target+
41
+ def seek(target)
42
+ raise ArgumentError, "Cannot seek backwards (at row #{@row}, asked for #{target})" if target < @row
43
+ return if target == @row
44
+ locs = @src.locations
45
+ if locs
46
+ j = locs.bsearch_index { |loc| loc.first_row_index > target }
47
+ j = (j || locs.size) - 1
48
+ if j > @page_idx
49
+ @src.jump_to_page(j)
50
+ @page = nil
51
+ @bd = @br = nil
52
+ @bi = 0
53
+ @page_idx = j - 1
54
+ @row = locs[j].first_row_index
55
+ @started = true # pages listed in an OffsetIndex start at row boundaries
56
+ end
57
+ end
58
+ skip(target - @row)
59
+ end
60
+
61
+ # Moves past the next +k+ rows without building their values
62
+ #
63
+ # @param k [Integer] number of rows to skip; zero or less does nothing
64
+ # @return [void]
65
+ # @raise [FormatError] when the chunk runs out of pages
66
+ def skip(k)
67
+ return if k <= 0
68
+ @repeated ? take_repeated(k, false) : take_flat(k, false)
69
+ @row += k
70
+ nil
71
+ end
72
+
73
+ # [definition_levels, repetition_levels, values] of the next +k+ rows. Levels are nil
74
+ # when the column's max level is 0.
75
+ #
76
+ # @param k [Integer] number of rows to hand out
77
+ # @return [Array(Array<Integer>, Array<Integer>, Array)] definition levels, repetition
78
+ # levels and the (converted) values of the non-null entries
79
+ # @raise [FormatError] when the chunk runs out of pages or does not start at a row boundary
80
+ def take(k)
81
+ pieces = @repeated ? take_repeated(k) : take_flat(k)
82
+ @row += k
83
+ return pieces.first if pieces.size == 1
84
+ defs = @max_def.positive? ? [] : nil
85
+ reps = @repeated ? [] : nil
86
+ vals = []
87
+ pieces.each do |d, r, v|
88
+ defs&.concat(d)
89
+ reps&.concat(r)
90
+ vals.concat(v)
91
+ end
92
+ [defs, reps, vals]
93
+ end
94
+
95
+ private
96
+
97
+ # Non-repeated columns, where every entry is a row: reads page by page until +k+ entries
98
+ #
99
+ # @param k [Integer] number of rows
100
+ # @param keep [Boolean] false to skip the values instead of decoding them
101
+ # @return [Array<Array(Array<Integer>, nil, Array)>] [defs, nil, values] per page touched;
102
+ # empty when +keep+ is false
103
+ # @raise [FormatError] when the chunk runs out of pages
104
+ def take_flat(k, keep = true)
105
+ pieces = []
106
+ while k > 0
107
+ load_page! while @page.nil? || @page.remaining.zero?
108
+ t = @page.remaining
109
+ t = k if k < t
110
+ defs, = @page.read_levels(t)
111
+ nv = defs ? defs.count(@max_def) : t
112
+ if keep
113
+ pieces << [defs, nil, values(nv)]
114
+ else
115
+ @page.skip_values(nv)
116
+ end
117
+ k -= t
118
+ end
119
+ pieces
120
+ end
121
+
122
+ # Collects entries until +k+ rows have started and the next row start (or the end of the
123
+ # column) is reached. Values are read from a page before moving on to the next one.
124
+ #
125
+ # @param k [Integer] number of rows
126
+ # @param keep [Boolean] false to skip the values instead of decoding them
127
+ # @return [Array<Array(Array<Integer>, Array<Integer>, Array)>] [defs, reps, values] per
128
+ # page touched (defs nil without definition levels); empty when +keep+ is false
129
+ # @raise [FormatError] when the first page does not start at a row boundary
130
+ def take_repeated(k, keep = true)
131
+ pieces = []
132
+ rows = 0
133
+ defs = @max_def.positive? ? [] : nil
134
+ reps = []
135
+ while true
136
+ if @bi >= @br.to_a.size
137
+ if @page&.remaining&.positive?
138
+ @bd, @br = @page.read_levels(LOOKAHEAD)
139
+ @bi = 0
140
+ else
141
+ flush(pieces, defs, reps, keep)
142
+ defs = @max_def.positive? ? [] : nil
143
+ reps = []
144
+ break unless load_page
145
+ next
146
+ end
147
+ end
148
+ unless @started
149
+ raise FormatError, "Column #{@path}: first page does not start at a row boundary" unless @br[@bi].zero?
150
+ @started = true
151
+ end
152
+ br = @br
153
+ i = @bi
154
+ n = br.size
155
+ done = false
156
+ while i < n
157
+ if br[i].zero?
158
+ if rows == k
159
+ done = true
160
+ break
161
+ end
162
+ rows += 1
163
+ end
164
+ i += 1
165
+ end
166
+ if i > @bi
167
+ reps.concat(br[@bi, i - @bi])
168
+ defs&.concat(@bd[@bi, i - @bi])
169
+ @bi = i
170
+ end
171
+ break if done
172
+ end
173
+ flush(pieces, defs, reps, keep)
174
+ pieces
175
+ end
176
+
177
+ # Reads (or skips) the values belonging to the collected entries of the current page
178
+ #
179
+ # @param pieces [Array<Array>] output list a [defs, reps, values] piece is appended to
180
+ # @param defs [Array<Integer>, nil] collected definition levels (nil without them)
181
+ # @param reps [Array<Integer>] collected repetition levels; nothing happens when empty
182
+ # @param keep [Boolean] false to skip the values instead of appending a piece
183
+ # @return [void]
184
+ def flush(pieces, defs, reps, keep)
185
+ return if reps.empty?
186
+ nv = defs ? defs.count(@max_def) : reps.size
187
+ if keep
188
+ pieces << [defs, reps, values(nv)]
189
+ else
190
+ @page.skip_values(nv)
191
+ end
192
+ end
193
+
194
+ # The next +n+ values of the current page, converted when the page has a converter
195
+ #
196
+ # @param n [Integer] number of values (non-null entries)
197
+ # @return [Array] Ruby values
198
+ def values(n)
199
+ vals = @page.read_values(n)
200
+ conv = @page.converter
201
+ conv ? vals.map!(&conv) : vals
202
+ end
203
+
204
+ # Like #load_page, but running out of pages is an error
205
+ #
206
+ # @return [void]
207
+ # @raise [FormatError] when the chunk has no more data pages
208
+ def load_page!
209
+ return if load_page
210
+ raise FormatError, "Column #{@path}: ran out of pages after #{@src.seen} of #{@src.total} values"
211
+ end
212
+
213
+ # Moves to the next data page; false at the end of the chunk
214
+ #
215
+ # @return [Boolean] whether a page was loaded
216
+ def load_page
217
+ @page = @src.next_stream or return false
218
+ @page_idx += 1
219
+ @bd = @br = nil
220
+ @bi = 0
221
+ true
222
+ end
223
+ end
224
+ end
225
+ end