herringbone 0.2.0 → 0.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -11,8 +11,17 @@ module Herringbone
11
11
 
12
12
  # The levels and values of one data page
13
13
  class Page
14
- attr_reader :remaining, :converter
14
+ # @return [Integer] entries (levels) of the page not read yet
15
+ attr_reader :remaining
15
16
 
17
+ # @return [Proc, nil] physical value => Ruby value, still to be applied to #read_values
18
+ attr_reader :converter
19
+
20
+ # @param entries [Integer] number of entries (num_values of the page header, nulls included)
21
+ # @param defs [HybridDecoder, ArrayDecoder, nil] definition levels; nil when max level is 0
22
+ # @param reps [HybridDecoder, ArrayDecoder, nil] repetition levels; nil when max level is 0
23
+ # @param values [Object] value decoder responding to #read(n) (one of the decoders here)
24
+ # @param converter [Proc, nil] converter for the decoded values, nil when none is needed
16
25
  def initialize(entries, defs, reps, values, converter)
17
26
  @remaining = entries
18
27
  @defs = defs
@@ -23,6 +32,9 @@ module Herringbone
23
32
 
24
33
  # [definition_levels, repetition_levels] of the next +n+ entries (nil when a column has no
25
34
  # levels of that kind)
35
+ #
36
+ # @param n [Integer] entries wanted; capped at #remaining
37
+ # @return [Array(Array<Integer>, Array<Integer>)] definition and repetition levels
26
38
  def read_levels(n)
27
39
  n = @remaining if n > @remaining
28
40
  @remaining -= n
@@ -30,24 +42,57 @@ module Herringbone
30
42
  end
31
43
 
32
44
  # The next +n+ values, as physical values (apply #converter for Ruby values)
45
+ #
46
+ # @param n [Integer] number of non-null values
47
+ # @return [Array] decoded values
48
+ # @raise [FormatError] when the page holds fewer values
33
49
  def read_values(n)
34
50
  n.zero? ? [] : @values.read(n)
35
51
  end
36
52
 
37
53
  # Moves past the next +n+ values without returning them
54
+ #
55
+ # @param n [Integer] number of non-null values
56
+ # @return [void]
57
+ # @raise [FormatError] when the page holds fewer values
38
58
  def skip_values(n)
39
59
  return if n.zero?
40
60
  @values.respond_to?(:skip) ? @values.skip(n) : @values.read(n)
41
61
  end
62
+
63
+ # The value decoder (for read(as: :numo), which asks it for bytes or Numo arrays)
64
+ #
65
+ # @return [Object] one of the decoders in PageStream
66
+ def value_decoder = @values
67
+
68
+ # Which of the next +n+ entries are defined (definition level == +max_def+), as a
69
+ # Numo::Bit, or nil when the column has no definition levels. For columns without
70
+ # repetition levels; used by read(as: :numo).
71
+ #
72
+ # @param n [Integer] entries wanted; capped at #remaining
73
+ # @param max_def [Integer] the column's max definition level
74
+ # @return [Numo::Bit, nil] 1 where the entry holds a value
75
+ def read_validity_numo(n, max_def)
76
+ n = @remaining if n > @remaining
77
+ @remaining -= n
78
+ defs = @defs or return nil
79
+ return defs.read_flags(n) if max_def == 1 && defs.respond_to?(:read_flags)
80
+ levels = defs.respond_to?(:read_numo) ? defs.read_numo(n, Numo::UInt8) : Numo::UInt8.cast(defs.read(n))
81
+ levels.eq(max_def)
82
+ end
42
83
  end
43
84
 
44
85
  # A decoder over an Array that is already decoded (legacy encodings, booleans, deltas)
45
86
  class ArrayDecoder
87
+ # @param values [Array] all of the page's decoded levels or values
46
88
  def initialize(values)
47
89
  @values = values
48
90
  @i = 0
49
91
  end
50
92
 
93
+ # @param n [Integer] number of entries to hand out
94
+ # @return [Array] the next +n+ entries
95
+ # @raise [FormatError] when fewer than +n+ are left
51
96
  def read(n)
52
97
  raise FormatError, "Page has fewer values than its levels require" if @i + n > @values.size
53
98
  out = @values[@i, n]
@@ -58,6 +103,10 @@ module Herringbone
58
103
 
59
104
  # The RLE / bit-packed hybrid, decoded run by run
60
105
  class HybridDecoder
106
+ # @param data [String] binary page data
107
+ # @param pos [Integer] offset of the first run header
108
+ # @param limit [Integer] offset just past the encoded runs
109
+ # @param width [Integer] bit width of each value
61
110
  def initialize(data, pos, limit, width)
62
111
  @data = data
63
112
  @pos = pos
@@ -72,16 +121,21 @@ module Herringbone
72
121
  @groups = 0 # bit-packed groups of 8 not unpacked yet
73
122
  end
74
123
 
124
+ # Bit-packed runs are unpacked up to CHUNK values at a time
125
+ #
126
+ # @param n [Integer] number of values to decode
127
+ # @return [Array<Integer>] the next +n+ values
128
+ # @raise [FormatError] when the runs end before +n+ values
75
129
  def read(n)
76
130
  out = []
77
131
  while n > 0
78
132
  next_run if @left.zero?
79
- t = n < @left ? n : @left
133
+ t = (n < @left) ? n : @left
80
134
  if @rle
81
135
  out.fill(@value, out.size, t)
82
136
  else
83
137
  if @buf.nil? || @bi >= @buf.size
84
- g = @groups < CHUNK / 8 ? @groups : CHUNK / 8
138
+ g = (@groups < CHUNK / 8) ? @groups : CHUNK / 8
85
139
  @buf = Encodings::RLE.unpack_bits(@data, @pos, g * 8, @width)
86
140
  @pos += g * @width
87
141
  @groups -= g
@@ -98,8 +152,94 @@ module Herringbone
98
152
  out
99
153
  end
100
154
 
155
+ # For a bit width of 1: the next +n+ values as a Numo::Bit (read(as: :numo)). Runs are
156
+ # collected as "0"/"1" characters, which costs less per run than Numo calls do (levels of
157
+ # columns with scattered nulls come in many short runs).
158
+ #
159
+ # @param n [Integer] number of values to decode
160
+ # @return [Numo::Bit] the next +n+ values
161
+ # @raise [ArgumentError] when the bit width is not 1
162
+ # @raise [FormatError] when the runs end before +n+ values
163
+ def read_flags(n)
164
+ raise ArgumentError, "read_flags needs a bit width of 1" unless @width == 1
165
+ out = String.new(capacity: n, encoding: Encoding::BINARY)
166
+ while n > 0
167
+ next_run if @left.zero?
168
+ t = (n < @left) ? n : @left
169
+ if @rle
170
+ out << ((@value == 1) ? "1" : "0") * t
171
+ elsif @buf && @bi < @buf.size
172
+ avail = @buf.size - @bi
173
+ t = avail if t > avail
174
+ out << @buf[@bi, t].join
175
+ @bi += t
176
+ else
177
+ groups = (t + 7) / 8
178
+ bits = @data.byteslice(@pos, groups)&.unpack1("b*") || +""
179
+ bits << "0" * (groups * 8 - bits.bytesize) if bits.bytesize < groups * 8 # truncated last run
180
+ @pos += groups
181
+ @groups -= groups
182
+ if groups * 8 > t
183
+ out << bits.byteslice(0, t)
184
+ @buf = bits.byteslice(t, groups * 8 - t).bytes.map! { |c| c - 48 }
185
+ else
186
+ out << bits
187
+ @buf = nil
188
+ end
189
+ @bi = 0
190
+ end
191
+ @left -= t
192
+ n -= t
193
+ end
194
+ Numo::UInt8.from_binary(out).eq(49)
195
+ end
196
+
197
+ # The next +n+ values as a Numo array of +klass+ (read(as: :numo)): RLE runs are
198
+ # filled and bit-packed runs unpacked by Numo, with no Ruby object per value. Leaves the
199
+ # decoder in a state #read can continue from.
200
+ #
201
+ # @param n [Integer] number of values to decode
202
+ # @param klass [Class] Numo integer class to fill, e.g. Numo::UInt8 or Numo::Int32
203
+ # @return [Numo::NArray] the next +n+ values as a +klass+ array
204
+ # @raise [FormatError] when the runs end before +n+ values
205
+ def read_numo(n, klass)
206
+ out = klass.zeros(n)
207
+ i = 0
208
+ while n > 0
209
+ next_run if @left.zero?
210
+ t = (n < @left) ? n : @left
211
+ if @rle
212
+ out[i...i + t] = @value
213
+ elsif @buf && @bi < @buf.size
214
+ # Values of this run that #read (or a previous call) already unpacked
215
+ avail = @buf.size - @bi
216
+ t = avail if t > avail
217
+ out[i...i + t] = @buf[@bi, t]
218
+ @bi += t
219
+ else
220
+ groups = (t + 7) / 8 # @left == @groups * 8 here, so these groups exist
221
+ vals = NumoColumns.unpack_bits(@data, @pos, groups * 8, @width)
222
+ @pos += groups * @width
223
+ @groups -= groups
224
+ # A partly used group goes to the buffer #read and this method take values from
225
+ @buf = (groups * 8 > t) ? vals[t..].to_a : nil
226
+ @bi = 0
227
+ out[i...i + t] = (groups * 8 > t) ? vals[0...t] : vals
228
+ end
229
+ @left -= t
230
+ n -= t
231
+ i += t
232
+ end
233
+ out
234
+ end
235
+
101
236
  private
102
237
 
238
+ # Reads run headers until a non-empty run starts: an RLE run (count and one value) or
239
+ # a bit-packed run (groups of 8 values)
240
+ #
241
+ # @return [void]
242
+ # @raise [FormatError] when there are no more runs before +limit+
103
243
  def next_run
104
244
  while @left.zero?
105
245
  raise FormatError, "RLE data exhausted" if @pos >= @limit
@@ -123,6 +263,10 @@ module Herringbone
123
263
 
124
264
  # PLAIN INT32 / INT64 / FLOAT / DOUBLE / INT96
125
265
  class FixedDecoder
266
+ # @param data [String] binary page data
267
+ # @param pos [Integer] offset of the first value
268
+ # @param format [String] String#unpack directive of one value, e.g. "l<"
269
+ # @param width [Integer] bytes per value
126
270
  def initialize(data, pos, format, width)
127
271
  @data = data
128
272
  @pos = pos
@@ -130,6 +274,9 @@ module Herringbone
130
274
  @width = width
131
275
  end
132
276
 
277
+ # @param n [Integer] number of values to decode
278
+ # @return [Array<Integer>, Array<Float>] the next +n+ values
279
+ # @raise [FormatError] when the page holds fewer values
133
280
  def read(n)
134
281
  bytes = n * @width
135
282
  raise FormatError, "Truncated PLAIN data" if @pos + bytes > @data.bytesize
@@ -138,17 +285,39 @@ module Herringbone
138
285
  out
139
286
  end
140
287
 
288
+ # @param n [Integer] number of values to move past
289
+ # @return [void]
290
+ # @raise [FormatError] when the page holds fewer values
141
291
  def skip(n)
142
292
  raise FormatError, "Truncated PLAIN data" if @pos + n * @width > @data.bytesize
143
293
  @pos += n * @width
144
294
  end
295
+
296
+ # The next +n+ values as their PLAIN (little-endian) bytes, for read(as: :numo)
297
+ #
298
+ # @param n [Integer] number of values
299
+ # @return [String] +n+ * width bytes
300
+ # @raise [FormatError] when the page holds fewer values
301
+ def read_bytes(n)
302
+ bytes = n * @width
303
+ raise FormatError, "Truncated PLAIN data" if @pos + bytes > @data.bytesize
304
+ out = @data.byteslice(@pos, bytes)
305
+ @pos += bytes
306
+ out
307
+ end
145
308
  end
146
309
 
310
+ # PLAIN INT96 (legacy Impala/Spark timestamps): 8 bytes of nanoseconds, 4 of Julian day
147
311
  class Int96Decoder < FixedDecoder
312
+ # @param data [String] binary page data
313
+ # @param pos [Integer] offset of the first value
148
314
  def initialize(data, pos)
149
315
  super(data, pos, "Q<L<", 12)
150
316
  end
151
317
 
318
+ # @param n [Integer] number of values to decode
319
+ # @return [Array<Array(Integer, Integer)>] [nanoseconds of the day, Julian day] per value
320
+ # @raise [FormatError] when the page holds fewer values
152
321
  def read(n)
153
322
  bytes = n * 12
154
323
  raise FormatError, "Truncated INT96 data" if @pos + bytes > @data.bytesize
@@ -160,12 +329,18 @@ module Herringbone
160
329
 
161
330
  # PLAIN FIXED_LEN_BYTE_ARRAY
162
331
  class FixedBytesDecoder
332
+ # @param data [String] binary page data
333
+ # @param pos [Integer] offset of the first value
334
+ # @param width [Integer] the column's type_length
163
335
  def initialize(data, pos, width)
164
336
  @data = data
165
337
  @pos = pos
166
338
  @width = width
167
339
  end
168
340
 
341
+ # @param n [Integer] number of values to decode
342
+ # @return [Array<String>] the next +n+ values as binary Strings
343
+ # @raise [FormatError] when the page holds fewer values
169
344
  def read(n)
170
345
  raise FormatError, "Truncated FIXED_LEN_BYTE_ARRAY data" if @pos + n * @width > @data.bytesize
171
346
  out = Array.new(n) { |i| @data.byteslice(@pos + i * @width, @width) }
@@ -173,6 +348,9 @@ module Herringbone
173
348
  out
174
349
  end
175
350
 
351
+ # @param n [Integer] number of values to move past
352
+ # @return [void]
353
+ # @raise [FormatError] when the page holds fewer values
176
354
  def skip(n)
177
355
  raise FormatError, "Truncated FIXED_LEN_BYTE_ARRAY data" if @pos + n * @width > @data.bytesize
178
356
  @pos += n * @width
@@ -181,11 +359,16 @@ module Herringbone
181
359
 
182
360
  # PLAIN BYTE_ARRAY: 4-byte length, then the bytes
183
361
  class ByteArrayDecoder
362
+ # @param data [String] binary page data
363
+ # @param pos [Integer] offset of the first length prefix
184
364
  def initialize(data, pos)
185
365
  @data = data
186
366
  @pos = pos
187
367
  end
188
368
 
369
+ # @param n [Integer] number of values to decode
370
+ # @return [Array<String>] the next +n+ values as binary Strings
371
+ # @raise [FormatError] when a length prefix or value runs past the page
189
372
  def read(n)
190
373
  data = @data
191
374
  size = data.bytesize
@@ -206,6 +389,10 @@ module Herringbone
206
389
  end
207
390
 
208
391
  # Walks the length prefixes without creating Strings
392
+ #
393
+ # @param n [Integer] number of values to move past
394
+ # @return [void]
395
+ # @raise [FormatError] when a length prefix or value runs past the page
209
396
  def skip(n)
210
397
  data = @data
211
398
  size = data.bytesize
@@ -221,21 +408,45 @@ module Herringbone
221
408
 
222
409
  # PLAIN BOOLEAN: one bit per value, LSB first
223
410
  class BooleanDecoder
411
+ # @param data [String] binary page data
412
+ # @param pos [Integer] offset of the first value; the rest of +data+ is unpacked at once
224
413
  def initialize(data, pos)
225
414
  @bits = data.byteslice(pos, data.bytesize - pos).unpack1("b*")
226
415
  @i = 0
227
416
  end
228
417
 
418
+ # @param n [Integer] number of values to decode
419
+ # @return [Array<Boolean>] the next +n+ values
420
+ # @raise [FormatError] when the page holds fewer values
229
421
  def read(n)
230
422
  raise FormatError, "Truncated BOOLEAN data" if @i + n > @bits.bytesize
231
423
  out = Array.new(n) { |k| @bits.getbyte(@i + k) == 49 }
232
424
  @i += n
233
425
  out
234
426
  end
427
+
428
+ # The next +n+ values as a Numo::Bit (read(as: :numo))
429
+ #
430
+ # @param n [Integer] number of values to decode
431
+ # @return [Numo::Bit] 1 for true
432
+ # @raise [FormatError] when the page holds fewer values
433
+ def read_numo(n)
434
+ raise FormatError, "Truncated BOOLEAN data" if @i + n > @bits.bytesize
435
+ out = Numo::UInt8.from_binary(@bits.byteslice(@i, n)).eq(49) # "0"/"1" characters
436
+ @i += n
437
+ out
438
+ end
235
439
  end
236
440
 
237
441
  # RLE_DICTIONARY / PLAIN_DICTIONARY indices mapped to the (already converted) dictionary
238
442
  class DictionaryDecoder
443
+ # @return [Array] the chunk's dictionary values (converted, shared by all its pages)
444
+ attr_reader :dictionary
445
+
446
+ # @param data [String] binary page data
447
+ # @param pos [Integer] offset of the bit width byte that precedes the RLE/bit-packed indices
448
+ # @param dictionary [Array] values the indices point into, already converted
449
+ # @param path [String] dotted column path, for error messages
239
450
  def initialize(data, pos, dictionary, path)
240
451
  @indices = HybridDecoder.new(data, pos + 1, data.bytesize, data.getbyte(pos).to_i)
241
452
  @dictionary = dictionary
@@ -243,38 +454,76 @@ module Herringbone
243
454
  end
244
455
 
245
456
  # Indices are decoded but not looked up
457
+ #
458
+ # @param n [Integer] number of values to move past
459
+ # @return [void]
460
+ # @raise [FormatError] when the indices run out
246
461
  def skip(n)
247
462
  @indices.read(n)
248
463
  end
249
464
 
465
+ # @param n [Integer] number of values to decode; must be positive
466
+ # @return [Array] the dictionary values at the next +n+ indices
467
+ # @raise [FormatError] when an index is out of range or the indices run out
250
468
  def read(n)
251
469
  indices = @indices.read(n)
252
470
  dict = @dictionary
253
471
  raise FormatError, "Dictionary index out of range in #{@path}" if indices.max >= dict.size
254
472
  indices.map! { |i| dict[i] }
255
473
  end
474
+
475
+ # The next +n+ indices as a Numo::Int32, bounds-checked (read(as: :numo))
476
+ #
477
+ # @param n [Integer] number of indices to decode
478
+ # @return [Numo::Int32] indices into #dictionary
479
+ # @raise [FormatError] when an index is out of range or the indices run out
480
+ def read_indices_numo(n)
481
+ indices = @indices.read_numo(n, Numo::Int32)
482
+ raise FormatError, "Dictionary index out of range in #{@path}" if n.positive? && indices.max >= @dictionary.size
483
+ indices
484
+ end
256
485
  end
257
486
 
258
487
  # RLE-encoded BOOLEAN values
259
488
  class RleBooleanDecoder
489
+ # @param data [String] binary page data
490
+ # @param pos [Integer] offset of the 4-byte length prefix of the RLE data
260
491
  def initialize(data, pos)
261
492
  len = data.byteslice(pos, 4).unpack1("V")
262
493
  @bits = HybridDecoder.new(data, pos + 4, pos + 4 + len, 1)
263
494
  end
264
495
 
496
+ # @param n [Integer] number of values to decode
497
+ # @return [Array<Boolean>] the next +n+ values
498
+ # @raise [FormatError] when the runs end before +n+ values
265
499
  def read(n)
266
500
  @bits.read(n).map! { |v| v == 1 }
267
501
  end
502
+
503
+ # The next +n+ values as a Numo::Bit (read(as: :numo))
504
+ #
505
+ # @param n [Integer] number of values to decode
506
+ # @return [Numo::Bit] 1 for true
507
+ # @raise [FormatError] when the runs end before +n+ values
508
+ def read_numo(n)
509
+ @bits.read_flags(n)
510
+ end
268
511
  end
269
512
 
270
513
  # DELTA_LENGTH_BYTE_ARRAY: all lengths first (Integers), then the bytes sliced as needed
271
514
  class DeltaLengthDecoder
515
+ # @param data [String] binary page data
516
+ # @param pos [Integer] offset of the DELTA_BINARY_PACKED lengths
517
+ # @raise [FormatError] when the lengths are malformed
272
518
  def initialize(data, pos)
273
519
  @lengths, @pos = Encodings::Delta.decode_binary_packed(data, pos, 32)
274
520
  @data = data
275
521
  @i = 0
276
522
  end
277
523
 
524
+ # @param n [Integer] number of values to decode
525
+ # @return [Array<String>] the next +n+ values as binary Strings
526
+ # @raise [FormatError] when there are too few lengths or a value runs past the page
278
527
  def read(n)
279
528
  raise FormatError, "DELTA_LENGTH_BYTE_ARRAY has too few values" if @i + n > @lengths.size
280
529
  data = @data
@@ -292,6 +541,9 @@ module Herringbone
292
541
 
293
542
  # DELTA_BYTE_ARRAY: prefix lengths and suffixes, each value built from the previous one
294
543
  class DeltaByteArrayDecoder
544
+ # @param data [String] binary page data
545
+ # @param pos [Integer] offset of the DELTA_BINARY_PACKED prefix lengths
546
+ # @raise [FormatError] when the prefix or suffix lengths are malformed
295
547
  def initialize(data, pos)
296
548
  @prefixes, pos = Encodings::Delta.decode_binary_packed(data, pos, 32)
297
549
  @suffixes = DeltaLengthDecoder.new(data, pos)
@@ -299,13 +551,17 @@ module Herringbone
299
551
  @i = 0
300
552
  end
301
553
 
554
+ # @param n [Integer] number of values to decode
555
+ # @return [Array<String>] the next +n+ values as binary Strings
556
+ # @raise [FormatError] when there are too few values or a prefix is negative or longer than
557
+ # the previous value
302
558
  def read(n)
303
559
  raise FormatError, "DELTA_BYTE_ARRAY has too few values" if @i + n > @prefixes.size
304
560
  suffixes = @suffixes.read(n)
305
561
  prev = @prev
306
562
  out = Array.new(n) do |k|
307
563
  prefix = @prefixes[@i + k]
308
- raise FormatError, "DELTA_BYTE_ARRAY prefix longer than previous value" if prefix > prev.bytesize
564
+ raise FormatError, "DELTA_BYTE_ARRAY prefix length #{prefix} out of range" if prefix > prev.bytesize || prefix.negative?
309
565
  prev = prefix.zero? ? suffixes[k] : prev.byteslice(0, prefix) + suffixes[k]
310
566
  end
311
567
  @prev = prev
@@ -316,6 +572,11 @@ module Herringbone
316
572
 
317
573
  # BYTE_STREAM_SPLIT: byte k of every value lives in stream k; values are gathered per call
318
574
  class ByteStreamSplitDecoder
575
+ # @param data [String] binary page data; the streams run to its end
576
+ # @param pos [Integer] offset of the first stream
577
+ # @param width [Integer] bytes per value (and number of streams)
578
+ # @param type [Integer] Format::Type of the column
579
+ # @param type_length [Integer, nil] value size in bytes for FIXED_LEN_BYTE_ARRAY, else unused
319
580
  def initialize(data, pos, width, type, type_length)
320
581
  @data = data
321
582
  @pos = pos
@@ -326,6 +587,9 @@ module Herringbone
326
587
  @i = 0
327
588
  end
328
589
 
590
+ # @param n [Integer] number of values to decode
591
+ # @return [Array<Integer>, Array<Float>, Array<String>] the next +n+ values, decoded as PLAIN
592
+ # @raise [FormatError] when the streams hold fewer values
329
593
  def read(n)
330
594
  raise FormatError, "Truncated BYTE_STREAM_SPLIT data" if @i + n > @count
331
595
  streams = Array.new(@width) { |k| @data.byteslice(@pos + k * @count + @i, n) }.join
@@ -333,6 +597,18 @@ module Herringbone
333
597
  plain, = Encodings::ByteStreamSplit.decode(streams, 0, n, @width)
334
598
  Encodings::Plain.decode(plain, 0, n, @type, @type_length).first
335
599
  end
600
+
601
+ # The next +n+ values re-interleaved into PLAIN bytes by Numo (read(as: :numo))
602
+ #
603
+ # @param n [Integer] number of values
604
+ # @return [String] +n+ * width bytes in PLAIN layout
605
+ # @raise [FormatError] when the streams hold fewer values
606
+ def read_bytes(n)
607
+ raise FormatError, "Truncated BYTE_STREAM_SPLIT data" if @i + n > @count
608
+ streams = Array.new(@width) { |k| @data.byteslice(@pos + k * @count + @i, n) }.join
609
+ @i += n
610
+ Numo::UInt8.from_binary(streams, [@width, n]).transpose.to_binary.b
611
+ end
336
612
  end
337
613
  end
338
614
  end