herringbone 0.1.0 → 0.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,615 @@
1
+ # frozen_string_literal: true
2
+
3
+ module Herringbone
4
+ class Reader
5
+ # Incremental decoders for the contents of one data page. The page bytes are decoded as they
6
+ # are asked for, so a caller that takes a few hundred entries at a time never holds a whole
7
+ # page's worth of levels or values as Ruby objects.
8
+ module PageStream
9
+ # Values unpacked from a bit-packed run at a time (a multiple of 8)
10
+ CHUNK = 1024
11
+
12
+ # The levels and values of one data page
13
+ class Page
14
+ # @return [Integer] entries (levels) of the page not read yet
15
+ attr_reader :remaining
16
+
17
+ # @return [Proc, nil] physical value => Ruby value, still to be applied to #read_values
18
+ attr_reader :converter
19
+
20
+ # @param entries [Integer] number of entries (num_values of the page header, nulls included)
21
+ # @param defs [HybridDecoder, ArrayDecoder, nil] definition levels; nil when max level is 0
22
+ # @param reps [HybridDecoder, ArrayDecoder, nil] repetition levels; nil when max level is 0
23
+ # @param values [Object] value decoder responding to #read(n) (one of the decoders here)
24
+ # @param converter [Proc, nil] converter for the decoded values, nil when none is needed
25
+ def initialize(entries, defs, reps, values, converter)
26
+ @remaining = entries
27
+ @defs = defs
28
+ @reps = reps
29
+ @values = values
30
+ @converter = converter
31
+ end
32
+
33
+ # [definition_levels, repetition_levels] of the next +n+ entries (nil when a column has no
34
+ # levels of that kind)
35
+ #
36
+ # @param n [Integer] entries wanted; capped at #remaining
37
+ # @return [Array(Array<Integer>, Array<Integer>)] definition and repetition levels
38
+ def read_levels(n)
39
+ n = @remaining if n > @remaining
40
+ @remaining -= n
41
+ [@defs&.read(n), @reps&.read(n)]
42
+ end
43
+
44
+ # The next +n+ values, as physical values (apply #converter for Ruby values)
45
+ #
46
+ # @param n [Integer] number of non-null values
47
+ # @return [Array] decoded values
48
+ # @raise [FormatError] when the page holds fewer values
49
+ def read_values(n)
50
+ n.zero? ? [] : @values.read(n)
51
+ end
52
+
53
+ # Moves past the next +n+ values without returning them
54
+ #
55
+ # @param n [Integer] number of non-null values
56
+ # @return [void]
57
+ # @raise [FormatError] when the page holds fewer values
58
+ def skip_values(n)
59
+ return if n.zero?
60
+ @values.respond_to?(:skip) ? @values.skip(n) : @values.read(n)
61
+ end
62
+
63
+ # The value decoder (for read(as: :numo), which asks it for bytes or Numo arrays)
64
+ #
65
+ # @return [Object] one of the decoders in PageStream
66
+ def value_decoder = @values
67
+
68
+ # Which of the next +n+ entries are defined (definition level == +max_def+), as a
69
+ # Numo::Bit, or nil when the column has no definition levels. For columns without
70
+ # repetition levels; used by read(as: :numo).
71
+ #
72
+ # @param n [Integer] entries wanted; capped at #remaining
73
+ # @param max_def [Integer] the column's max definition level
74
+ # @return [Numo::Bit, nil] 1 where the entry holds a value
75
+ def read_validity_numo(n, max_def)
76
+ n = @remaining if n > @remaining
77
+ @remaining -= n
78
+ defs = @defs or return nil
79
+ return defs.read_flags(n) if max_def == 1 && defs.respond_to?(:read_flags)
80
+ levels = defs.respond_to?(:read_numo) ? defs.read_numo(n, Numo::UInt8) : Numo::UInt8.cast(defs.read(n))
81
+ levels.eq(max_def)
82
+ end
83
+ end
84
+
85
+ # A decoder over an Array that is already decoded (legacy encodings, booleans, deltas)
86
+ class ArrayDecoder
87
+ # @param values [Array] all of the page's decoded levels or values
88
+ def initialize(values)
89
+ @values = values
90
+ @i = 0
91
+ end
92
+
93
+ # @param n [Integer] number of entries to hand out
94
+ # @return [Array] the next +n+ entries
95
+ # @raise [FormatError] when fewer than +n+ are left
96
+ def read(n)
97
+ raise FormatError, "Page has fewer values than its levels require" if @i + n > @values.size
98
+ out = @values[@i, n]
99
+ @i += n
100
+ out
101
+ end
102
+ end
103
+
104
+ # The RLE / bit-packed hybrid, decoded run by run
105
+ class HybridDecoder
106
+ # @param data [String] binary page data
107
+ # @param pos [Integer] offset of the first run header
108
+ # @param limit [Integer] offset just past the encoded runs
109
+ # @param width [Integer] bit width of each value
110
+ def initialize(data, pos, limit, width)
111
+ @data = data
112
+ @pos = pos
113
+ @limit = limit
114
+ @width = width
115
+ @value_bytes = (width + 7) / 8
116
+ @left = 0 # values left in the current run
117
+ @rle = true
118
+ @value = 0
119
+ @buf = nil
120
+ @bi = 0
121
+ @groups = 0 # bit-packed groups of 8 not unpacked yet
122
+ end
123
+
124
+ # Bit-packed runs are unpacked up to CHUNK values at a time
125
+ #
126
+ # @param n [Integer] number of values to decode
127
+ # @return [Array<Integer>] the next +n+ values
128
+ # @raise [FormatError] when the runs end before +n+ values
129
+ def read(n)
130
+ out = []
131
+ while n > 0
132
+ next_run if @left.zero?
133
+ t = (n < @left) ? n : @left
134
+ if @rle
135
+ out.fill(@value, out.size, t)
136
+ else
137
+ if @buf.nil? || @bi >= @buf.size
138
+ g = (@groups < CHUNK / 8) ? @groups : CHUNK / 8
139
+ @buf = Encodings::RLE.unpack_bits(@data, @pos, g * 8, @width)
140
+ @pos += g * @width
141
+ @groups -= g
142
+ @bi = 0
143
+ end
144
+ avail = @buf.size - @bi
145
+ t = avail if t > avail
146
+ out.concat(@buf[@bi, t])
147
+ @bi += t
148
+ end
149
+ @left -= t
150
+ n -= t
151
+ end
152
+ out
153
+ end
154
+
155
+ # For a bit width of 1: the next +n+ values as a Numo::Bit (read(as: :numo)). Runs are
156
+ # collected as "0"/"1" characters, which costs less per run than Numo calls do (levels of
157
+ # columns with scattered nulls come in many short runs).
158
+ #
159
+ # @param n [Integer] number of values to decode
160
+ # @return [Numo::Bit] the next +n+ values
161
+ # @raise [ArgumentError] when the bit width is not 1
162
+ # @raise [FormatError] when the runs end before +n+ values
163
+ def read_flags(n)
164
+ raise ArgumentError, "read_flags needs a bit width of 1" unless @width == 1
165
+ out = String.new(capacity: n, encoding: Encoding::BINARY)
166
+ while n > 0
167
+ next_run if @left.zero?
168
+ t = (n < @left) ? n : @left
169
+ if @rle
170
+ out << ((@value == 1) ? "1" : "0") * t
171
+ elsif @buf && @bi < @buf.size
172
+ avail = @buf.size - @bi
173
+ t = avail if t > avail
174
+ out << @buf[@bi, t].join
175
+ @bi += t
176
+ else
177
+ groups = (t + 7) / 8
178
+ bits = @data.byteslice(@pos, groups)&.unpack1("b*") || +""
179
+ bits << "0" * (groups * 8 - bits.bytesize) if bits.bytesize < groups * 8 # truncated last run
180
+ @pos += groups
181
+ @groups -= groups
182
+ if groups * 8 > t
183
+ out << bits.byteslice(0, t)
184
+ @buf = bits.byteslice(t, groups * 8 - t).bytes.map! { |c| c - 48 }
185
+ else
186
+ out << bits
187
+ @buf = nil
188
+ end
189
+ @bi = 0
190
+ end
191
+ @left -= t
192
+ n -= t
193
+ end
194
+ Numo::UInt8.from_binary(out).eq(49)
195
+ end
196
+
197
+ # The next +n+ values as a Numo array of +klass+ (read(as: :numo)): RLE runs are
198
+ # filled and bit-packed runs unpacked by Numo, with no Ruby object per value. Leaves the
199
+ # decoder in a state #read can continue from.
200
+ #
201
+ # @param n [Integer] number of values to decode
202
+ # @param klass [Class] Numo integer class to fill, e.g. Numo::UInt8 or Numo::Int32
203
+ # @return [Numo::NArray] the next +n+ values as a +klass+ array
204
+ # @raise [FormatError] when the runs end before +n+ values
205
+ def read_numo(n, klass)
206
+ out = klass.zeros(n)
207
+ i = 0
208
+ while n > 0
209
+ next_run if @left.zero?
210
+ t = (n < @left) ? n : @left
211
+ if @rle
212
+ out[i...i + t] = @value
213
+ elsif @buf && @bi < @buf.size
214
+ # Values of this run that #read (or a previous call) already unpacked
215
+ avail = @buf.size - @bi
216
+ t = avail if t > avail
217
+ out[i...i + t] = @buf[@bi, t]
218
+ @bi += t
219
+ else
220
+ groups = (t + 7) / 8 # @left == @groups * 8 here, so these groups exist
221
+ vals = NumoColumns.unpack_bits(@data, @pos, groups * 8, @width)
222
+ @pos += groups * @width
223
+ @groups -= groups
224
+ # A partly used group goes to the buffer #read and this method take values from
225
+ @buf = (groups * 8 > t) ? vals[t..].to_a : nil
226
+ @bi = 0
227
+ out[i...i + t] = (groups * 8 > t) ? vals[0...t] : vals
228
+ end
229
+ @left -= t
230
+ n -= t
231
+ i += t
232
+ end
233
+ out
234
+ end
235
+
236
+ private
237
+
238
+ # Reads run headers until a non-empty run starts: an RLE run (count and one value) or
239
+ # a bit-packed run (groups of 8 values)
240
+ #
241
+ # @return [void]
242
+ # @raise [FormatError] when there are no more runs before +limit+
243
+ def next_run
244
+ while @left.zero?
245
+ raise FormatError, "RLE data exhausted" if @pos >= @limit
246
+ header, @pos = Encodings::RLE.read_uleb(@data, @pos)
247
+ if header & 1 == 1
248
+ @rle = false
249
+ @groups = header >> 1
250
+ @left = @groups * 8
251
+ @buf = nil
252
+ else
253
+ @rle = true
254
+ @left = header >> 1
255
+ v = 0
256
+ @value_bytes.times { |k| v |= @data.getbyte(@pos + k).to_i << (8 * k) }
257
+ @value = v
258
+ @pos += @value_bytes
259
+ end
260
+ end
261
+ end
262
+ end
263
+
264
+ # PLAIN INT32 / INT64 / FLOAT / DOUBLE / INT96
265
+ class FixedDecoder
266
+ # @param data [String] binary page data
267
+ # @param pos [Integer] offset of the first value
268
+ # @param format [String] String#unpack directive of one value, e.g. "l<"
269
+ # @param width [Integer] bytes per value
270
+ def initialize(data, pos, format, width)
271
+ @data = data
272
+ @pos = pos
273
+ @format = format
274
+ @width = width
275
+ end
276
+
277
+ # @param n [Integer] number of values to decode
278
+ # @return [Array<Integer>, Array<Float>] the next +n+ values
279
+ # @raise [FormatError] when the page holds fewer values
280
+ def read(n)
281
+ bytes = n * @width
282
+ raise FormatError, "Truncated PLAIN data" if @pos + bytes > @data.bytesize
283
+ out = @data.byteslice(@pos, bytes).unpack("#{@format}#{n}")
284
+ @pos += bytes
285
+ out
286
+ end
287
+
288
+ # @param n [Integer] number of values to move past
289
+ # @return [void]
290
+ # @raise [FormatError] when the page holds fewer values
291
+ def skip(n)
292
+ raise FormatError, "Truncated PLAIN data" if @pos + n * @width > @data.bytesize
293
+ @pos += n * @width
294
+ end
295
+
296
+ # The next +n+ values as their PLAIN (little-endian) bytes, for read(as: :numo)
297
+ #
298
+ # @param n [Integer] number of values
299
+ # @return [String] +n+ * width bytes
300
+ # @raise [FormatError] when the page holds fewer values
301
+ def read_bytes(n)
302
+ bytes = n * @width
303
+ raise FormatError, "Truncated PLAIN data" if @pos + bytes > @data.bytesize
304
+ out = @data.byteslice(@pos, bytes)
305
+ @pos += bytes
306
+ out
307
+ end
308
+ end
309
+
310
+ # PLAIN INT96 (legacy Impala/Spark timestamps): 8 bytes of nanoseconds, 4 of Julian day
311
+ class Int96Decoder < FixedDecoder
312
+ # @param data [String] binary page data
313
+ # @param pos [Integer] offset of the first value
314
+ def initialize(data, pos)
315
+ super(data, pos, "Q<L<", 12)
316
+ end
317
+
318
+ # @param n [Integer] number of values to decode
319
+ # @return [Array<Array(Integer, Integer)>] [nanoseconds of the day, Julian day] per value
320
+ # @raise [FormatError] when the page holds fewer values
321
+ def read(n)
322
+ bytes = n * 12
323
+ raise FormatError, "Truncated INT96 data" if @pos + bytes > @data.bytesize
324
+ out = @data.byteslice(@pos, bytes).unpack("Q<L<" * n).each_slice(2).to_a
325
+ @pos += bytes
326
+ out
327
+ end
328
+ end
329
+
330
+ # PLAIN FIXED_LEN_BYTE_ARRAY
331
+ class FixedBytesDecoder
332
+ # @param data [String] binary page data
333
+ # @param pos [Integer] offset of the first value
334
+ # @param width [Integer] the column's type_length
335
+ def initialize(data, pos, width)
336
+ @data = data
337
+ @pos = pos
338
+ @width = width
339
+ end
340
+
341
+ # @param n [Integer] number of values to decode
342
+ # @return [Array<String>] the next +n+ values as binary Strings
343
+ # @raise [FormatError] when the page holds fewer values
344
+ def read(n)
345
+ raise FormatError, "Truncated FIXED_LEN_BYTE_ARRAY data" if @pos + n * @width > @data.bytesize
346
+ out = Array.new(n) { |i| @data.byteslice(@pos + i * @width, @width) }
347
+ @pos += n * @width
348
+ out
349
+ end
350
+
351
+ # @param n [Integer] number of values to move past
352
+ # @return [void]
353
+ # @raise [FormatError] when the page holds fewer values
354
+ def skip(n)
355
+ raise FormatError, "Truncated FIXED_LEN_BYTE_ARRAY data" if @pos + n * @width > @data.bytesize
356
+ @pos += n * @width
357
+ end
358
+ end
359
+
360
+ # PLAIN BYTE_ARRAY: 4-byte length, then the bytes
361
+ class ByteArrayDecoder
362
+ # @param data [String] binary page data
363
+ # @param pos [Integer] offset of the first length prefix
364
+ def initialize(data, pos)
365
+ @data = data
366
+ @pos = pos
367
+ end
368
+
369
+ # @param n [Integer] number of values to decode
370
+ # @return [Array<String>] the next +n+ values as binary Strings
371
+ # @raise [FormatError] when a length prefix or value runs past the page
372
+ def read(n)
373
+ data = @data
374
+ size = data.bytesize
375
+ pos = @pos
376
+ out = Array.new(n)
377
+ i = 0
378
+ while i < n
379
+ raise FormatError, "Truncated BYTE_ARRAY data" if pos + 4 > size
380
+ len = data.getbyte(pos) | (data.getbyte(pos + 1) << 8) | (data.getbyte(pos + 2) << 16) | (data.getbyte(pos + 3) << 24)
381
+ pos += 4
382
+ raise FormatError, "BYTE_ARRAY value overruns page" if pos + len > size
383
+ out[i] = data.byteslice(pos, len)
384
+ pos += len
385
+ i += 1
386
+ end
387
+ @pos = pos
388
+ out
389
+ end
390
+
391
+ # Walks the length prefixes without creating Strings
392
+ #
393
+ # @param n [Integer] number of values to move past
394
+ # @return [void]
395
+ # @raise [FormatError] when a length prefix or value runs past the page
396
+ def skip(n)
397
+ data = @data
398
+ size = data.bytesize
399
+ pos = @pos
400
+ n.times do
401
+ raise FormatError, "Truncated BYTE_ARRAY data" if pos + 4 > size
402
+ pos += 4 + (data.getbyte(pos) | (data.getbyte(pos + 1) << 8) | (data.getbyte(pos + 2) << 16) | (data.getbyte(pos + 3) << 24))
403
+ end
404
+ raise FormatError, "BYTE_ARRAY value overruns page" if pos > size
405
+ @pos = pos
406
+ end
407
+ end
408
+
409
+ # PLAIN BOOLEAN: one bit per value, LSB first
410
+ class BooleanDecoder
411
+ # @param data [String] binary page data
412
+ # @param pos [Integer] offset of the first value; the rest of +data+ is unpacked at once
413
+ def initialize(data, pos)
414
+ @bits = data.byteslice(pos, data.bytesize - pos).unpack1("b*")
415
+ @i = 0
416
+ end
417
+
418
+ # @param n [Integer] number of values to decode
419
+ # @return [Array<Boolean>] the next +n+ values
420
+ # @raise [FormatError] when the page holds fewer values
421
+ def read(n)
422
+ raise FormatError, "Truncated BOOLEAN data" if @i + n > @bits.bytesize
423
+ out = Array.new(n) { |k| @bits.getbyte(@i + k) == 49 }
424
+ @i += n
425
+ out
426
+ end
427
+
428
+ # The next +n+ values as a Numo::Bit (read(as: :numo))
429
+ #
430
+ # @param n [Integer] number of values to decode
431
+ # @return [Numo::Bit] 1 for true
432
+ # @raise [FormatError] when the page holds fewer values
433
+ def read_numo(n)
434
+ raise FormatError, "Truncated BOOLEAN data" if @i + n > @bits.bytesize
435
+ out = Numo::UInt8.from_binary(@bits.byteslice(@i, n)).eq(49) # "0"/"1" characters
436
+ @i += n
437
+ out
438
+ end
439
+ end
440
+
441
+ # RLE_DICTIONARY / PLAIN_DICTIONARY indices mapped to the (already converted) dictionary
442
+ class DictionaryDecoder
443
+ # @return [Array] the chunk's dictionary values (converted, shared by all its pages)
444
+ attr_reader :dictionary
445
+
446
+ # @param data [String] binary page data
447
+ # @param pos [Integer] offset of the bit width byte that precedes the RLE/bit-packed indices
448
+ # @param dictionary [Array] values the indices point into, already converted
449
+ # @param path [String] dotted column path, for error messages
450
+ def initialize(data, pos, dictionary, path)
451
+ @indices = HybridDecoder.new(data, pos + 1, data.bytesize, data.getbyte(pos).to_i)
452
+ @dictionary = dictionary
453
+ @path = path
454
+ end
455
+
456
+ # Indices are decoded but not looked up
457
+ #
458
+ # @param n [Integer] number of values to move past
459
+ # @return [void]
460
+ # @raise [FormatError] when the indices run out
461
+ def skip(n)
462
+ @indices.read(n)
463
+ end
464
+
465
+ # @param n [Integer] number of values to decode; must be positive
466
+ # @return [Array] the dictionary values at the next +n+ indices
467
+ # @raise [FormatError] when an index is out of range or the indices run out
468
+ def read(n)
469
+ indices = @indices.read(n)
470
+ dict = @dictionary
471
+ raise FormatError, "Dictionary index out of range in #{@path}" if indices.max >= dict.size
472
+ indices.map! { |i| dict[i] }
473
+ end
474
+
475
+ # The next +n+ indices as a Numo::Int32, bounds-checked (read(as: :numo))
476
+ #
477
+ # @param n [Integer] number of indices to decode
478
+ # @return [Numo::Int32] indices into #dictionary
479
+ # @raise [FormatError] when an index is out of range or the indices run out
480
+ def read_indices_numo(n)
481
+ indices = @indices.read_numo(n, Numo::Int32)
482
+ raise FormatError, "Dictionary index out of range in #{@path}" if n.positive? && indices.max >= @dictionary.size
483
+ indices
484
+ end
485
+ end
486
+
487
+ # RLE-encoded BOOLEAN values
488
+ class RleBooleanDecoder
489
+ # @param data [String] binary page data
490
+ # @param pos [Integer] offset of the 4-byte length prefix of the RLE data
491
+ def initialize(data, pos)
492
+ len = data.byteslice(pos, 4).unpack1("V")
493
+ @bits = HybridDecoder.new(data, pos + 4, pos + 4 + len, 1)
494
+ end
495
+
496
+ # @param n [Integer] number of values to decode
497
+ # @return [Array<Boolean>] the next +n+ values
498
+ # @raise [FormatError] when the runs end before +n+ values
499
+ def read(n)
500
+ @bits.read(n).map! { |v| v == 1 }
501
+ end
502
+
503
+ # The next +n+ values as a Numo::Bit (read(as: :numo))
504
+ #
505
+ # @param n [Integer] number of values to decode
506
+ # @return [Numo::Bit] 1 for true
507
+ # @raise [FormatError] when the runs end before +n+ values
508
+ def read_numo(n)
509
+ @bits.read_flags(n)
510
+ end
511
+ end
512
+
513
+ # DELTA_LENGTH_BYTE_ARRAY: all lengths first (Integers), then the bytes sliced as needed
514
+ class DeltaLengthDecoder
515
+ # @param data [String] binary page data
516
+ # @param pos [Integer] offset of the DELTA_BINARY_PACKED lengths
517
+ # @raise [FormatError] when the lengths are malformed
518
+ def initialize(data, pos)
519
+ @lengths, @pos = Encodings::Delta.decode_binary_packed(data, pos, 32)
520
+ @data = data
521
+ @i = 0
522
+ end
523
+
524
+ # @param n [Integer] number of values to decode
525
+ # @return [Array<String>] the next +n+ values as binary Strings
526
+ # @raise [FormatError] when there are too few lengths or a value runs past the page
527
+ def read(n)
528
+ raise FormatError, "DELTA_LENGTH_BYTE_ARRAY has too few values" if @i + n > @lengths.size
529
+ data = @data
530
+ out = Array.new(n) do |k|
531
+ len = @lengths[@i + k]
532
+ raise FormatError, "DELTA_LENGTH_BYTE_ARRAY value overruns page" if len.negative? || @pos + len > data.bytesize
533
+ s = data.byteslice(@pos, len)
534
+ @pos += len
535
+ s
536
+ end
537
+ @i += n
538
+ out
539
+ end
540
+ end
541
+
542
+ # DELTA_BYTE_ARRAY: prefix lengths and suffixes, each value built from the previous one
543
+ class DeltaByteArrayDecoder
544
+ # @param data [String] binary page data
545
+ # @param pos [Integer] offset of the DELTA_BINARY_PACKED prefix lengths
546
+ # @raise [FormatError] when the prefix or suffix lengths are malformed
547
+ def initialize(data, pos)
548
+ @prefixes, pos = Encodings::Delta.decode_binary_packed(data, pos, 32)
549
+ @suffixes = DeltaLengthDecoder.new(data, pos)
550
+ @prev = "".b
551
+ @i = 0
552
+ end
553
+
554
+ # @param n [Integer] number of values to decode
555
+ # @return [Array<String>] the next +n+ values as binary Strings
556
+ # @raise [FormatError] when there are too few values or a prefix is negative or longer than
557
+ # the previous value
558
+ def read(n)
559
+ raise FormatError, "DELTA_BYTE_ARRAY has too few values" if @i + n > @prefixes.size
560
+ suffixes = @suffixes.read(n)
561
+ prev = @prev
562
+ out = Array.new(n) do |k|
563
+ prefix = @prefixes[@i + k]
564
+ raise FormatError, "DELTA_BYTE_ARRAY prefix length #{prefix} out of range" if prefix > prev.bytesize || prefix.negative?
565
+ prev = prefix.zero? ? suffixes[k] : prev.byteslice(0, prefix) + suffixes[k]
566
+ end
567
+ @prev = prev
568
+ @i += n
569
+ out
570
+ end
571
+ end
572
+
573
+ # BYTE_STREAM_SPLIT: byte k of every value lives in stream k; values are gathered per call
574
+ class ByteStreamSplitDecoder
575
+ # @param data [String] binary page data; the streams run to its end
576
+ # @param pos [Integer] offset of the first stream
577
+ # @param width [Integer] bytes per value (and number of streams)
578
+ # @param type [Integer] Format::Type of the column
579
+ # @param type_length [Integer, nil] value size in bytes for FIXED_LEN_BYTE_ARRAY, else unused
580
+ def initialize(data, pos, width, type, type_length)
581
+ @data = data
582
+ @pos = pos
583
+ @width = width
584
+ @count = (data.bytesize - pos) / width
585
+ @type = type
586
+ @type_length = type_length
587
+ @i = 0
588
+ end
589
+
590
+ # @param n [Integer] number of values to decode
591
+ # @return [Array<Integer>, Array<Float>, Array<String>] the next +n+ values, decoded as PLAIN
592
+ # @raise [FormatError] when the streams hold fewer values
593
+ def read(n)
594
+ raise FormatError, "Truncated BYTE_STREAM_SPLIT data" if @i + n > @count
595
+ streams = Array.new(@width) { |k| @data.byteslice(@pos + k * @count + @i, n) }.join
596
+ @i += n
597
+ plain, = Encodings::ByteStreamSplit.decode(streams, 0, n, @width)
598
+ Encodings::Plain.decode(plain, 0, n, @type, @type_length).first
599
+ end
600
+
601
+ # The next +n+ values re-interleaved into PLAIN bytes by Numo (read(as: :numo))
602
+ #
603
+ # @param n [Integer] number of values
604
+ # @return [String] +n+ * width bytes in PLAIN layout
605
+ # @raise [FormatError] when the streams hold fewer values
606
+ def read_bytes(n)
607
+ raise FormatError, "Truncated BYTE_STREAM_SPLIT data" if @i + n > @count
608
+ streams = Array.new(@width) { |k| @data.byteslice(@pos + k * @count + @i, n) }.join
609
+ @i += n
610
+ Numo::UInt8.from_binary(streams, [@width, n]).transpose.to_binary.b
611
+ end
612
+ end
613
+ end
614
+ end
615
+ end