herringbone 0.2.0 → 0.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/README.md +56 -4
- data/lib/herringbone/active_record.rb +44 -12
- data/lib/herringbone/bloom_filter.rb +112 -12
- data/lib/herringbone/byte_values.rb +29 -0
- data/lib/herringbone/codecs/lz4.rb +66 -3
- data/lib/herringbone/codecs/snappy.rb +132 -3
- data/lib/herringbone/compression.rb +48 -0
- data/lib/herringbone/encodings/delta.rb +70 -13
- data/lib/herringbone/encodings/plain.rb +22 -0
- data/lib/herringbone/encodings/rle.rb +53 -1
- data/lib/herringbone/format.rb +150 -0
- data/lib/herringbone/inspector.rb +477 -81
- data/lib/herringbone/io_buffer_support.rb +3 -1
- data/lib/herringbone/reader/column_chunk_reader.rb +98 -6
- data/lib/herringbone/reader/column_cursor.rb +46 -2
- data/lib/herringbone/reader/numo.rb +365 -0
- data/lib/herringbone/reader/page_stream.rb +280 -4
- data/lib/herringbone/reader/scan.rb +103 -9
- data/lib/herringbone/reader.rb +308 -17
- data/lib/herringbone/schema.rb +306 -15
- data/lib/herringbone/thrift.rb +155 -4
- data/lib/herringbone/types.rb +176 -41
- data/lib/herringbone/version.rb +2 -1
- data/lib/herringbone/visualizer.rb +51 -11
- data/lib/herringbone/writer.rb +308 -31
- data/lib/herringbone/xxhash.rb +108 -2
- data/lib/herringbone.rb +28 -0
- metadata +3 -2
|
@@ -11,8 +11,17 @@ module Herringbone
|
|
|
11
11
|
|
|
12
12
|
# The levels and values of one data page
|
|
13
13
|
class Page
|
|
14
|
-
|
|
14
|
+
# @return [Integer] entries (levels) of the page not read yet
|
|
15
|
+
attr_reader :remaining
|
|
15
16
|
|
|
17
|
+
# @return [Proc, nil] physical value => Ruby value, still to be applied to #read_values
|
|
18
|
+
attr_reader :converter
|
|
19
|
+
|
|
20
|
+
# @param entries [Integer] number of entries (num_values of the page header, nulls included)
|
|
21
|
+
# @param defs [HybridDecoder, ArrayDecoder, nil] definition levels; nil when max level is 0
|
|
22
|
+
# @param reps [HybridDecoder, ArrayDecoder, nil] repetition levels; nil when max level is 0
|
|
23
|
+
# @param values [Object] value decoder responding to #read(n) (one of the decoders here)
|
|
24
|
+
# @param converter [Proc, nil] converter for the decoded values, nil when none is needed
|
|
16
25
|
def initialize(entries, defs, reps, values, converter)
|
|
17
26
|
@remaining = entries
|
|
18
27
|
@defs = defs
|
|
@@ -23,6 +32,9 @@ module Herringbone
|
|
|
23
32
|
|
|
24
33
|
# [definition_levels, repetition_levels] of the next +n+ entries (nil when a column has no
|
|
25
34
|
# levels of that kind)
|
|
35
|
+
#
|
|
36
|
+
# @param n [Integer] entries wanted; capped at #remaining
|
|
37
|
+
# @return [Array(Array<Integer>, Array<Integer>)] definition and repetition levels
|
|
26
38
|
def read_levels(n)
|
|
27
39
|
n = @remaining if n > @remaining
|
|
28
40
|
@remaining -= n
|
|
@@ -30,24 +42,57 @@ module Herringbone
|
|
|
30
42
|
end
|
|
31
43
|
|
|
32
44
|
# The next +n+ values, as physical values (apply #converter for Ruby values)
|
|
45
|
+
#
|
|
46
|
+
# @param n [Integer] number of non-null values
|
|
47
|
+
# @return [Array] decoded values
|
|
48
|
+
# @raise [FormatError] when the page holds fewer values
|
|
33
49
|
def read_values(n)
|
|
34
50
|
n.zero? ? [] : @values.read(n)
|
|
35
51
|
end
|
|
36
52
|
|
|
37
53
|
# Moves past the next +n+ values without returning them
|
|
54
|
+
#
|
|
55
|
+
# @param n [Integer] number of non-null values
|
|
56
|
+
# @return [void]
|
|
57
|
+
# @raise [FormatError] when the page holds fewer values
|
|
38
58
|
def skip_values(n)
|
|
39
59
|
return if n.zero?
|
|
40
60
|
@values.respond_to?(:skip) ? @values.skip(n) : @values.read(n)
|
|
41
61
|
end
|
|
62
|
+
|
|
63
|
+
# The value decoder (for read(as: :numo), which asks it for bytes or Numo arrays)
|
|
64
|
+
#
|
|
65
|
+
# @return [Object] one of the decoders in PageStream
|
|
66
|
+
def value_decoder = @values
|
|
67
|
+
|
|
68
|
+
# Which of the next +n+ entries are defined (definition level == +max_def+), as a
|
|
69
|
+
# Numo::Bit, or nil when the column has no definition levels. For columns without
|
|
70
|
+
# repetition levels; used by read(as: :numo).
|
|
71
|
+
#
|
|
72
|
+
# @param n [Integer] entries wanted; capped at #remaining
|
|
73
|
+
# @param max_def [Integer] the column's max definition level
|
|
74
|
+
# @return [Numo::Bit, nil] 1 where the entry holds a value
|
|
75
|
+
def read_validity_numo(n, max_def)
|
|
76
|
+
n = @remaining if n > @remaining
|
|
77
|
+
@remaining -= n
|
|
78
|
+
defs = @defs or return nil
|
|
79
|
+
return defs.read_flags(n) if max_def == 1 && defs.respond_to?(:read_flags)
|
|
80
|
+
levels = defs.respond_to?(:read_numo) ? defs.read_numo(n, Numo::UInt8) : Numo::UInt8.cast(defs.read(n))
|
|
81
|
+
levels.eq(max_def)
|
|
82
|
+
end
|
|
42
83
|
end
|
|
43
84
|
|
|
44
85
|
# A decoder over an Array that is already decoded (legacy encodings, booleans, deltas)
|
|
45
86
|
class ArrayDecoder
|
|
87
|
+
# @param values [Array] all of the page's decoded levels or values
|
|
46
88
|
def initialize(values)
|
|
47
89
|
@values = values
|
|
48
90
|
@i = 0
|
|
49
91
|
end
|
|
50
92
|
|
|
93
|
+
# @param n [Integer] number of entries to hand out
|
|
94
|
+
# @return [Array] the next +n+ entries
|
|
95
|
+
# @raise [FormatError] when fewer than +n+ are left
|
|
51
96
|
def read(n)
|
|
52
97
|
raise FormatError, "Page has fewer values than its levels require" if @i + n > @values.size
|
|
53
98
|
out = @values[@i, n]
|
|
@@ -58,6 +103,10 @@ module Herringbone
|
|
|
58
103
|
|
|
59
104
|
# The RLE / bit-packed hybrid, decoded run by run
|
|
60
105
|
class HybridDecoder
|
|
106
|
+
# @param data [String] binary page data
|
|
107
|
+
# @param pos [Integer] offset of the first run header
|
|
108
|
+
# @param limit [Integer] offset just past the encoded runs
|
|
109
|
+
# @param width [Integer] bit width of each value
|
|
61
110
|
def initialize(data, pos, limit, width)
|
|
62
111
|
@data = data
|
|
63
112
|
@pos = pos
|
|
@@ -72,16 +121,21 @@ module Herringbone
|
|
|
72
121
|
@groups = 0 # bit-packed groups of 8 not unpacked yet
|
|
73
122
|
end
|
|
74
123
|
|
|
124
|
+
# Bit-packed runs are unpacked up to CHUNK values at a time
|
|
125
|
+
#
|
|
126
|
+
# @param n [Integer] number of values to decode
|
|
127
|
+
# @return [Array<Integer>] the next +n+ values
|
|
128
|
+
# @raise [FormatError] when the runs end before +n+ values
|
|
75
129
|
def read(n)
|
|
76
130
|
out = []
|
|
77
131
|
while n > 0
|
|
78
132
|
next_run if @left.zero?
|
|
79
|
-
t = n < @left ? n : @left
|
|
133
|
+
t = (n < @left) ? n : @left
|
|
80
134
|
if @rle
|
|
81
135
|
out.fill(@value, out.size, t)
|
|
82
136
|
else
|
|
83
137
|
if @buf.nil? || @bi >= @buf.size
|
|
84
|
-
g = @groups < CHUNK / 8 ? @groups : CHUNK / 8
|
|
138
|
+
g = (@groups < CHUNK / 8) ? @groups : CHUNK / 8
|
|
85
139
|
@buf = Encodings::RLE.unpack_bits(@data, @pos, g * 8, @width)
|
|
86
140
|
@pos += g * @width
|
|
87
141
|
@groups -= g
|
|
@@ -98,8 +152,94 @@ module Herringbone
|
|
|
98
152
|
out
|
|
99
153
|
end
|
|
100
154
|
|
|
155
|
+
# For a bit width of 1: the next +n+ values as a Numo::Bit (read(as: :numo)). Runs are
|
|
156
|
+
# collected as "0"/"1" characters, which costs less per run than Numo calls do (levels of
|
|
157
|
+
# columns with scattered nulls come in many short runs).
|
|
158
|
+
#
|
|
159
|
+
# @param n [Integer] number of values to decode
|
|
160
|
+
# @return [Numo::Bit] the next +n+ values
|
|
161
|
+
# @raise [ArgumentError] when the bit width is not 1
|
|
162
|
+
# @raise [FormatError] when the runs end before +n+ values
|
|
163
|
+
def read_flags(n)
|
|
164
|
+
raise ArgumentError, "read_flags needs a bit width of 1" unless @width == 1
|
|
165
|
+
out = String.new(capacity: n, encoding: Encoding::BINARY)
|
|
166
|
+
while n > 0
|
|
167
|
+
next_run if @left.zero?
|
|
168
|
+
t = (n < @left) ? n : @left
|
|
169
|
+
if @rle
|
|
170
|
+
out << ((@value == 1) ? "1" : "0") * t
|
|
171
|
+
elsif @buf && @bi < @buf.size
|
|
172
|
+
avail = @buf.size - @bi
|
|
173
|
+
t = avail if t > avail
|
|
174
|
+
out << @buf[@bi, t].join
|
|
175
|
+
@bi += t
|
|
176
|
+
else
|
|
177
|
+
groups = (t + 7) / 8
|
|
178
|
+
bits = @data.byteslice(@pos, groups)&.unpack1("b*") || +""
|
|
179
|
+
bits << "0" * (groups * 8 - bits.bytesize) if bits.bytesize < groups * 8 # truncated last run
|
|
180
|
+
@pos += groups
|
|
181
|
+
@groups -= groups
|
|
182
|
+
if groups * 8 > t
|
|
183
|
+
out << bits.byteslice(0, t)
|
|
184
|
+
@buf = bits.byteslice(t, groups * 8 - t).bytes.map! { |c| c - 48 }
|
|
185
|
+
else
|
|
186
|
+
out << bits
|
|
187
|
+
@buf = nil
|
|
188
|
+
end
|
|
189
|
+
@bi = 0
|
|
190
|
+
end
|
|
191
|
+
@left -= t
|
|
192
|
+
n -= t
|
|
193
|
+
end
|
|
194
|
+
Numo::UInt8.from_binary(out).eq(49)
|
|
195
|
+
end
|
|
196
|
+
|
|
197
|
+
# The next +n+ values as a Numo array of +klass+ (read(as: :numo)): RLE runs are
|
|
198
|
+
# filled and bit-packed runs unpacked by Numo, with no Ruby object per value. Leaves the
|
|
199
|
+
# decoder in a state #read can continue from.
|
|
200
|
+
#
|
|
201
|
+
# @param n [Integer] number of values to decode
|
|
202
|
+
# @param klass [Class] Numo integer class to fill, e.g. Numo::UInt8 or Numo::Int32
|
|
203
|
+
# @return [Numo::NArray] the next +n+ values as a +klass+ array
|
|
204
|
+
# @raise [FormatError] when the runs end before +n+ values
|
|
205
|
+
def read_numo(n, klass)
|
|
206
|
+
out = klass.zeros(n)
|
|
207
|
+
i = 0
|
|
208
|
+
while n > 0
|
|
209
|
+
next_run if @left.zero?
|
|
210
|
+
t = (n < @left) ? n : @left
|
|
211
|
+
if @rle
|
|
212
|
+
out[i...i + t] = @value
|
|
213
|
+
elsif @buf && @bi < @buf.size
|
|
214
|
+
# Values of this run that #read (or a previous call) already unpacked
|
|
215
|
+
avail = @buf.size - @bi
|
|
216
|
+
t = avail if t > avail
|
|
217
|
+
out[i...i + t] = @buf[@bi, t]
|
|
218
|
+
@bi += t
|
|
219
|
+
else
|
|
220
|
+
groups = (t + 7) / 8 # @left == @groups * 8 here, so these groups exist
|
|
221
|
+
vals = NumoColumns.unpack_bits(@data, @pos, groups * 8, @width)
|
|
222
|
+
@pos += groups * @width
|
|
223
|
+
@groups -= groups
|
|
224
|
+
# A partly used group goes to the buffer #read and this method take values from
|
|
225
|
+
@buf = (groups * 8 > t) ? vals[t..].to_a : nil
|
|
226
|
+
@bi = 0
|
|
227
|
+
out[i...i + t] = (groups * 8 > t) ? vals[0...t] : vals
|
|
228
|
+
end
|
|
229
|
+
@left -= t
|
|
230
|
+
n -= t
|
|
231
|
+
i += t
|
|
232
|
+
end
|
|
233
|
+
out
|
|
234
|
+
end
|
|
235
|
+
|
|
101
236
|
private
|
|
102
237
|
|
|
238
|
+
# Reads run headers until a non-empty run starts: an RLE run (count and one value) or
|
|
239
|
+
# a bit-packed run (groups of 8 values)
|
|
240
|
+
#
|
|
241
|
+
# @return [void]
|
|
242
|
+
# @raise [FormatError] when there are no more runs before +limit+
|
|
103
243
|
def next_run
|
|
104
244
|
while @left.zero?
|
|
105
245
|
raise FormatError, "RLE data exhausted" if @pos >= @limit
|
|
@@ -123,6 +263,10 @@ module Herringbone
|
|
|
123
263
|
|
|
124
264
|
# PLAIN INT32 / INT64 / FLOAT / DOUBLE / INT96
|
|
125
265
|
class FixedDecoder
|
|
266
|
+
# @param data [String] binary page data
|
|
267
|
+
# @param pos [Integer] offset of the first value
|
|
268
|
+
# @param format [String] String#unpack directive of one value, e.g. "l<"
|
|
269
|
+
# @param width [Integer] bytes per value
|
|
126
270
|
def initialize(data, pos, format, width)
|
|
127
271
|
@data = data
|
|
128
272
|
@pos = pos
|
|
@@ -130,6 +274,9 @@ module Herringbone
|
|
|
130
274
|
@width = width
|
|
131
275
|
end
|
|
132
276
|
|
|
277
|
+
# @param n [Integer] number of values to decode
|
|
278
|
+
# @return [Array<Integer>, Array<Float>] the next +n+ values
|
|
279
|
+
# @raise [FormatError] when the page holds fewer values
|
|
133
280
|
def read(n)
|
|
134
281
|
bytes = n * @width
|
|
135
282
|
raise FormatError, "Truncated PLAIN data" if @pos + bytes > @data.bytesize
|
|
@@ -138,17 +285,39 @@ module Herringbone
|
|
|
138
285
|
out
|
|
139
286
|
end
|
|
140
287
|
|
|
288
|
+
# @param n [Integer] number of values to move past
|
|
289
|
+
# @return [void]
|
|
290
|
+
# @raise [FormatError] when the page holds fewer values
|
|
141
291
|
def skip(n)
|
|
142
292
|
raise FormatError, "Truncated PLAIN data" if @pos + n * @width > @data.bytesize
|
|
143
293
|
@pos += n * @width
|
|
144
294
|
end
|
|
295
|
+
|
|
296
|
+
# The next +n+ values as their PLAIN (little-endian) bytes, for read(as: :numo)
|
|
297
|
+
#
|
|
298
|
+
# @param n [Integer] number of values
|
|
299
|
+
# @return [String] +n+ * width bytes
|
|
300
|
+
# @raise [FormatError] when the page holds fewer values
|
|
301
|
+
def read_bytes(n)
|
|
302
|
+
bytes = n * @width
|
|
303
|
+
raise FormatError, "Truncated PLAIN data" if @pos + bytes > @data.bytesize
|
|
304
|
+
out = @data.byteslice(@pos, bytes)
|
|
305
|
+
@pos += bytes
|
|
306
|
+
out
|
|
307
|
+
end
|
|
145
308
|
end
|
|
146
309
|
|
|
310
|
+
# PLAIN INT96 (legacy Impala/Spark timestamps): 8 bytes of nanoseconds, 4 of Julian day
|
|
147
311
|
class Int96Decoder < FixedDecoder
|
|
312
|
+
# @param data [String] binary page data
|
|
313
|
+
# @param pos [Integer] offset of the first value
|
|
148
314
|
def initialize(data, pos)
|
|
149
315
|
super(data, pos, "Q<L<", 12)
|
|
150
316
|
end
|
|
151
317
|
|
|
318
|
+
# @param n [Integer] number of values to decode
|
|
319
|
+
# @return [Array<Array(Integer, Integer)>] [nanoseconds of the day, Julian day] per value
|
|
320
|
+
# @raise [FormatError] when the page holds fewer values
|
|
152
321
|
def read(n)
|
|
153
322
|
bytes = n * 12
|
|
154
323
|
raise FormatError, "Truncated INT96 data" if @pos + bytes > @data.bytesize
|
|
@@ -160,12 +329,18 @@ module Herringbone
|
|
|
160
329
|
|
|
161
330
|
# PLAIN FIXED_LEN_BYTE_ARRAY
|
|
162
331
|
class FixedBytesDecoder
|
|
332
|
+
# @param data [String] binary page data
|
|
333
|
+
# @param pos [Integer] offset of the first value
|
|
334
|
+
# @param width [Integer] the column's type_length
|
|
163
335
|
def initialize(data, pos, width)
|
|
164
336
|
@data = data
|
|
165
337
|
@pos = pos
|
|
166
338
|
@width = width
|
|
167
339
|
end
|
|
168
340
|
|
|
341
|
+
# @param n [Integer] number of values to decode
|
|
342
|
+
# @return [Array<String>] the next +n+ values as binary Strings
|
|
343
|
+
# @raise [FormatError] when the page holds fewer values
|
|
169
344
|
def read(n)
|
|
170
345
|
raise FormatError, "Truncated FIXED_LEN_BYTE_ARRAY data" if @pos + n * @width > @data.bytesize
|
|
171
346
|
out = Array.new(n) { |i| @data.byteslice(@pos + i * @width, @width) }
|
|
@@ -173,6 +348,9 @@ module Herringbone
|
|
|
173
348
|
out
|
|
174
349
|
end
|
|
175
350
|
|
|
351
|
+
# @param n [Integer] number of values to move past
|
|
352
|
+
# @return [void]
|
|
353
|
+
# @raise [FormatError] when the page holds fewer values
|
|
176
354
|
def skip(n)
|
|
177
355
|
raise FormatError, "Truncated FIXED_LEN_BYTE_ARRAY data" if @pos + n * @width > @data.bytesize
|
|
178
356
|
@pos += n * @width
|
|
@@ -181,11 +359,16 @@ module Herringbone
|
|
|
181
359
|
|
|
182
360
|
# PLAIN BYTE_ARRAY: 4-byte length, then the bytes
|
|
183
361
|
class ByteArrayDecoder
|
|
362
|
+
# @param data [String] binary page data
|
|
363
|
+
# @param pos [Integer] offset of the first length prefix
|
|
184
364
|
def initialize(data, pos)
|
|
185
365
|
@data = data
|
|
186
366
|
@pos = pos
|
|
187
367
|
end
|
|
188
368
|
|
|
369
|
+
# @param n [Integer] number of values to decode
|
|
370
|
+
# @return [Array<String>] the next +n+ values as binary Strings
|
|
371
|
+
# @raise [FormatError] when a length prefix or value runs past the page
|
|
189
372
|
def read(n)
|
|
190
373
|
data = @data
|
|
191
374
|
size = data.bytesize
|
|
@@ -206,6 +389,10 @@ module Herringbone
|
|
|
206
389
|
end
|
|
207
390
|
|
|
208
391
|
# Walks the length prefixes without creating Strings
|
|
392
|
+
#
|
|
393
|
+
# @param n [Integer] number of values to move past
|
|
394
|
+
# @return [void]
|
|
395
|
+
# @raise [FormatError] when a length prefix or value runs past the page
|
|
209
396
|
def skip(n)
|
|
210
397
|
data = @data
|
|
211
398
|
size = data.bytesize
|
|
@@ -221,21 +408,45 @@ module Herringbone
|
|
|
221
408
|
|
|
222
409
|
# PLAIN BOOLEAN: one bit per value, LSB first
|
|
223
410
|
class BooleanDecoder
|
|
411
|
+
# @param data [String] binary page data
|
|
412
|
+
# @param pos [Integer] offset of the first value; the rest of +data+ is unpacked at once
|
|
224
413
|
def initialize(data, pos)
|
|
225
414
|
@bits = data.byteslice(pos, data.bytesize - pos).unpack1("b*")
|
|
226
415
|
@i = 0
|
|
227
416
|
end
|
|
228
417
|
|
|
418
|
+
# @param n [Integer] number of values to decode
|
|
419
|
+
# @return [Array<Boolean>] the next +n+ values
|
|
420
|
+
# @raise [FormatError] when the page holds fewer values
|
|
229
421
|
def read(n)
|
|
230
422
|
raise FormatError, "Truncated BOOLEAN data" if @i + n > @bits.bytesize
|
|
231
423
|
out = Array.new(n) { |k| @bits.getbyte(@i + k) == 49 }
|
|
232
424
|
@i += n
|
|
233
425
|
out
|
|
234
426
|
end
|
|
427
|
+
|
|
428
|
+
# The next +n+ values as a Numo::Bit (read(as: :numo))
|
|
429
|
+
#
|
|
430
|
+
# @param n [Integer] number of values to decode
|
|
431
|
+
# @return [Numo::Bit] 1 for true
|
|
432
|
+
# @raise [FormatError] when the page holds fewer values
|
|
433
|
+
def read_numo(n)
|
|
434
|
+
raise FormatError, "Truncated BOOLEAN data" if @i + n > @bits.bytesize
|
|
435
|
+
out = Numo::UInt8.from_binary(@bits.byteslice(@i, n)).eq(49) # "0"/"1" characters
|
|
436
|
+
@i += n
|
|
437
|
+
out
|
|
438
|
+
end
|
|
235
439
|
end
|
|
236
440
|
|
|
237
441
|
# RLE_DICTIONARY / PLAIN_DICTIONARY indices mapped to the (already converted) dictionary
|
|
238
442
|
class DictionaryDecoder
|
|
443
|
+
# @return [Array] the chunk's dictionary values (converted, shared by all its pages)
|
|
444
|
+
attr_reader :dictionary
|
|
445
|
+
|
|
446
|
+
# @param data [String] binary page data
|
|
447
|
+
# @param pos [Integer] offset of the bit width byte that precedes the RLE/bit-packed indices
|
|
448
|
+
# @param dictionary [Array] values the indices point into, already converted
|
|
449
|
+
# @param path [String] dotted column path, for error messages
|
|
239
450
|
def initialize(data, pos, dictionary, path)
|
|
240
451
|
@indices = HybridDecoder.new(data, pos + 1, data.bytesize, data.getbyte(pos).to_i)
|
|
241
452
|
@dictionary = dictionary
|
|
@@ -243,38 +454,76 @@ module Herringbone
|
|
|
243
454
|
end
|
|
244
455
|
|
|
245
456
|
# Indices are decoded but not looked up
|
|
457
|
+
#
|
|
458
|
+
# @param n [Integer] number of values to move past
|
|
459
|
+
# @return [void]
|
|
460
|
+
# @raise [FormatError] when the indices run out
|
|
246
461
|
def skip(n)
|
|
247
462
|
@indices.read(n)
|
|
248
463
|
end
|
|
249
464
|
|
|
465
|
+
# @param n [Integer] number of values to decode; must be positive
|
|
466
|
+
# @return [Array] the dictionary values at the next +n+ indices
|
|
467
|
+
# @raise [FormatError] when an index is out of range or the indices run out
|
|
250
468
|
def read(n)
|
|
251
469
|
indices = @indices.read(n)
|
|
252
470
|
dict = @dictionary
|
|
253
471
|
raise FormatError, "Dictionary index out of range in #{@path}" if indices.max >= dict.size
|
|
254
472
|
indices.map! { |i| dict[i] }
|
|
255
473
|
end
|
|
474
|
+
|
|
475
|
+
# The next +n+ indices as a Numo::Int32, bounds-checked (read(as: :numo))
|
|
476
|
+
#
|
|
477
|
+
# @param n [Integer] number of indices to decode
|
|
478
|
+
# @return [Numo::Int32] indices into #dictionary
|
|
479
|
+
# @raise [FormatError] when an index is out of range or the indices run out
|
|
480
|
+
def read_indices_numo(n)
|
|
481
|
+
indices = @indices.read_numo(n, Numo::Int32)
|
|
482
|
+
raise FormatError, "Dictionary index out of range in #{@path}" if n.positive? && indices.max >= @dictionary.size
|
|
483
|
+
indices
|
|
484
|
+
end
|
|
256
485
|
end
|
|
257
486
|
|
|
258
487
|
# RLE-encoded BOOLEAN values
|
|
259
488
|
class RleBooleanDecoder
|
|
489
|
+
# @param data [String] binary page data
|
|
490
|
+
# @param pos [Integer] offset of the 4-byte length prefix of the RLE data
|
|
260
491
|
def initialize(data, pos)
|
|
261
492
|
len = data.byteslice(pos, 4).unpack1("V")
|
|
262
493
|
@bits = HybridDecoder.new(data, pos + 4, pos + 4 + len, 1)
|
|
263
494
|
end
|
|
264
495
|
|
|
496
|
+
# @param n [Integer] number of values to decode
|
|
497
|
+
# @return [Array<Boolean>] the next +n+ values
|
|
498
|
+
# @raise [FormatError] when the runs end before +n+ values
|
|
265
499
|
def read(n)
|
|
266
500
|
@bits.read(n).map! { |v| v == 1 }
|
|
267
501
|
end
|
|
502
|
+
|
|
503
|
+
# The next +n+ values as a Numo::Bit (read(as: :numo))
|
|
504
|
+
#
|
|
505
|
+
# @param n [Integer] number of values to decode
|
|
506
|
+
# @return [Numo::Bit] 1 for true
|
|
507
|
+
# @raise [FormatError] when the runs end before +n+ values
|
|
508
|
+
def read_numo(n)
|
|
509
|
+
@bits.read_flags(n)
|
|
510
|
+
end
|
|
268
511
|
end
|
|
269
512
|
|
|
270
513
|
# DELTA_LENGTH_BYTE_ARRAY: all lengths first (Integers), then the bytes sliced as needed
|
|
271
514
|
class DeltaLengthDecoder
|
|
515
|
+
# @param data [String] binary page data
|
|
516
|
+
# @param pos [Integer] offset of the DELTA_BINARY_PACKED lengths
|
|
517
|
+
# @raise [FormatError] when the lengths are malformed
|
|
272
518
|
def initialize(data, pos)
|
|
273
519
|
@lengths, @pos = Encodings::Delta.decode_binary_packed(data, pos, 32)
|
|
274
520
|
@data = data
|
|
275
521
|
@i = 0
|
|
276
522
|
end
|
|
277
523
|
|
|
524
|
+
# @param n [Integer] number of values to decode
|
|
525
|
+
# @return [Array<String>] the next +n+ values as binary Strings
|
|
526
|
+
# @raise [FormatError] when there are too few lengths or a value runs past the page
|
|
278
527
|
def read(n)
|
|
279
528
|
raise FormatError, "DELTA_LENGTH_BYTE_ARRAY has too few values" if @i + n > @lengths.size
|
|
280
529
|
data = @data
|
|
@@ -292,6 +541,9 @@ module Herringbone
|
|
|
292
541
|
|
|
293
542
|
# DELTA_BYTE_ARRAY: prefix lengths and suffixes, each value built from the previous one
|
|
294
543
|
class DeltaByteArrayDecoder
|
|
544
|
+
# @param data [String] binary page data
|
|
545
|
+
# @param pos [Integer] offset of the DELTA_BINARY_PACKED prefix lengths
|
|
546
|
+
# @raise [FormatError] when the prefix or suffix lengths are malformed
|
|
295
547
|
def initialize(data, pos)
|
|
296
548
|
@prefixes, pos = Encodings::Delta.decode_binary_packed(data, pos, 32)
|
|
297
549
|
@suffixes = DeltaLengthDecoder.new(data, pos)
|
|
@@ -299,13 +551,17 @@ module Herringbone
|
|
|
299
551
|
@i = 0
|
|
300
552
|
end
|
|
301
553
|
|
|
554
|
+
# @param n [Integer] number of values to decode
|
|
555
|
+
# @return [Array<String>] the next +n+ values as binary Strings
|
|
556
|
+
# @raise [FormatError] when there are too few values or a prefix is negative or longer than
|
|
557
|
+
# the previous value
|
|
302
558
|
def read(n)
|
|
303
559
|
raise FormatError, "DELTA_BYTE_ARRAY has too few values" if @i + n > @prefixes.size
|
|
304
560
|
suffixes = @suffixes.read(n)
|
|
305
561
|
prev = @prev
|
|
306
562
|
out = Array.new(n) do |k|
|
|
307
563
|
prefix = @prefixes[@i + k]
|
|
308
|
-
raise FormatError, "DELTA_BYTE_ARRAY prefix
|
|
564
|
+
raise FormatError, "DELTA_BYTE_ARRAY prefix length #{prefix} out of range" if prefix > prev.bytesize || prefix.negative?
|
|
309
565
|
prev = prefix.zero? ? suffixes[k] : prev.byteslice(0, prefix) + suffixes[k]
|
|
310
566
|
end
|
|
311
567
|
@prev = prev
|
|
@@ -316,6 +572,11 @@ module Herringbone
|
|
|
316
572
|
|
|
317
573
|
# BYTE_STREAM_SPLIT: byte k of every value lives in stream k; values are gathered per call
|
|
318
574
|
class ByteStreamSplitDecoder
|
|
575
|
+
# @param data [String] binary page data; the streams run to its end
|
|
576
|
+
# @param pos [Integer] offset of the first stream
|
|
577
|
+
# @param width [Integer] bytes per value (and number of streams)
|
|
578
|
+
# @param type [Integer] Format::Type of the column
|
|
579
|
+
# @param type_length [Integer, nil] value size in bytes for FIXED_LEN_BYTE_ARRAY, else unused
|
|
319
580
|
def initialize(data, pos, width, type, type_length)
|
|
320
581
|
@data = data
|
|
321
582
|
@pos = pos
|
|
@@ -326,6 +587,9 @@ module Herringbone
|
|
|
326
587
|
@i = 0
|
|
327
588
|
end
|
|
328
589
|
|
|
590
|
+
# @param n [Integer] number of values to decode
|
|
591
|
+
# @return [Array<Integer>, Array<Float>, Array<String>] the next +n+ values, decoded as PLAIN
|
|
592
|
+
# @raise [FormatError] when the streams hold fewer values
|
|
329
593
|
def read(n)
|
|
330
594
|
raise FormatError, "Truncated BYTE_STREAM_SPLIT data" if @i + n > @count
|
|
331
595
|
streams = Array.new(@width) { |k| @data.byteslice(@pos + k * @count + @i, n) }.join
|
|
@@ -333,6 +597,18 @@ module Herringbone
|
|
|
333
597
|
plain, = Encodings::ByteStreamSplit.decode(streams, 0, n, @width)
|
|
334
598
|
Encodings::Plain.decode(plain, 0, n, @type, @type_length).first
|
|
335
599
|
end
|
|
600
|
+
|
|
601
|
+
# The next +n+ values re-interleaved into PLAIN bytes by Numo (read(as: :numo))
|
|
602
|
+
#
|
|
603
|
+
# @param n [Integer] number of values
|
|
604
|
+
# @return [String] +n+ * width bytes in PLAIN layout
|
|
605
|
+
# @raise [FormatError] when the streams hold fewer values
|
|
606
|
+
def read_bytes(n)
|
|
607
|
+
raise FormatError, "Truncated BYTE_STREAM_SPLIT data" if @i + n > @count
|
|
608
|
+
streams = Array.new(@width) { |k| @data.byteslice(@pos + k * @count + @i, n) }.join
|
|
609
|
+
@i += n
|
|
610
|
+
Numo::UInt8.from_binary(streams, [@width, n]).transpose.to_binary.b
|
|
611
|
+
end
|
|
336
612
|
end
|
|
337
613
|
end
|
|
338
614
|
end
|