herringbone 0.1.0 → 0.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/README.md +198 -74
- data/bin/herringbone +38 -26
- data/lib/herringbone/active_record.rb +48 -19
- data/lib/herringbone/bloom_filter.rb +370 -0
- data/lib/herringbone/byte_values.rb +29 -0
- data/lib/herringbone/codecs/lz4.rb +66 -3
- data/lib/herringbone/codecs/snappy.rb +132 -3
- data/lib/herringbone/compression.rb +121 -16
- data/lib/herringbone/encodings/delta.rb +76 -19
- data/lib/herringbone/encodings/plain.rb +29 -7
- data/lib/herringbone/encodings/rle.rb +55 -3
- data/lib/herringbone/format.rb +203 -0
- data/lib/herringbone/inspector.rb +1784 -0
- data/lib/herringbone/io_buffer_support.rb +3 -1
- data/lib/herringbone/reader/column_chunk_reader.rb +388 -0
- data/lib/herringbone/reader/column_cursor.rb +225 -0
- data/lib/herringbone/reader/numo.rb +365 -0
- data/lib/herringbone/reader/page_stream.rb +615 -0
- data/lib/herringbone/reader/scan.rb +350 -0
- data/lib/herringbone/reader.rb +585 -272
- data/lib/herringbone/schema.rb +326 -90
- data/lib/herringbone/thrift.rb +155 -4
- data/lib/herringbone/types.rb +177 -42
- data/lib/herringbone/version.rb +2 -1
- data/lib/herringbone/visualizer.rb +1136 -0
- data/lib/herringbone/writer.rb +550 -105
- data/lib/herringbone/xxhash.rb +425 -0
- data/lib/herringbone.rb +64 -9
- metadata +14 -33
|
@@ -0,0 +1,615 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Herringbone
|
|
4
|
+
class Reader
|
|
5
|
+
# Incremental decoders for the contents of one data page. The page bytes are decoded as they
|
|
6
|
+
# are asked for, so a caller that takes a few hundred entries at a time never holds a whole
|
|
7
|
+
# page's worth of levels or values as Ruby objects.
|
|
8
|
+
module PageStream
|
|
9
|
+
# Values unpacked from a bit-packed run at a time (a multiple of 8)
|
|
10
|
+
CHUNK = 1024
|
|
11
|
+
|
|
12
|
+
# The levels and values of one data page
|
|
13
|
+
class Page
|
|
14
|
+
# @return [Integer] entries (levels) of the page not read yet
|
|
15
|
+
attr_reader :remaining
|
|
16
|
+
|
|
17
|
+
# @return [Proc, nil] physical value => Ruby value, still to be applied to #read_values
|
|
18
|
+
attr_reader :converter
|
|
19
|
+
|
|
20
|
+
# @param entries [Integer] number of entries (num_values of the page header, nulls included)
|
|
21
|
+
# @param defs [HybridDecoder, ArrayDecoder, nil] definition levels; nil when max level is 0
|
|
22
|
+
# @param reps [HybridDecoder, ArrayDecoder, nil] repetition levels; nil when max level is 0
|
|
23
|
+
# @param values [Object] value decoder responding to #read(n) (one of the decoders here)
|
|
24
|
+
# @param converter [Proc, nil] converter for the decoded values, nil when none is needed
|
|
25
|
+
def initialize(entries, defs, reps, values, converter)
|
|
26
|
+
@remaining = entries
|
|
27
|
+
@defs = defs
|
|
28
|
+
@reps = reps
|
|
29
|
+
@values = values
|
|
30
|
+
@converter = converter
|
|
31
|
+
end
|
|
32
|
+
|
|
33
|
+
# [definition_levels, repetition_levels] of the next +n+ entries (nil when a column has no
|
|
34
|
+
# levels of that kind)
|
|
35
|
+
#
|
|
36
|
+
# @param n [Integer] entries wanted; capped at #remaining
|
|
37
|
+
# @return [Array(Array<Integer>, Array<Integer>)] definition and repetition levels
|
|
38
|
+
def read_levels(n)
|
|
39
|
+
n = @remaining if n > @remaining
|
|
40
|
+
@remaining -= n
|
|
41
|
+
[@defs&.read(n), @reps&.read(n)]
|
|
42
|
+
end
|
|
43
|
+
|
|
44
|
+
# The next +n+ values, as physical values (apply #converter for Ruby values)
|
|
45
|
+
#
|
|
46
|
+
# @param n [Integer] number of non-null values
|
|
47
|
+
# @return [Array] decoded values
|
|
48
|
+
# @raise [FormatError] when the page holds fewer values
|
|
49
|
+
def read_values(n)
|
|
50
|
+
n.zero? ? [] : @values.read(n)
|
|
51
|
+
end
|
|
52
|
+
|
|
53
|
+
# Moves past the next +n+ values without returning them
|
|
54
|
+
#
|
|
55
|
+
# @param n [Integer] number of non-null values
|
|
56
|
+
# @return [void]
|
|
57
|
+
# @raise [FormatError] when the page holds fewer values
|
|
58
|
+
def skip_values(n)
|
|
59
|
+
return if n.zero?
|
|
60
|
+
@values.respond_to?(:skip) ? @values.skip(n) : @values.read(n)
|
|
61
|
+
end
|
|
62
|
+
|
|
63
|
+
# The value decoder (for read(as: :numo), which asks it for bytes or Numo arrays)
|
|
64
|
+
#
|
|
65
|
+
# @return [Object] one of the decoders in PageStream
|
|
66
|
+
def value_decoder = @values
|
|
67
|
+
|
|
68
|
+
# Which of the next +n+ entries are defined (definition level == +max_def+), as a
|
|
69
|
+
# Numo::Bit, or nil when the column has no definition levels. For columns without
|
|
70
|
+
# repetition levels; used by read(as: :numo).
|
|
71
|
+
#
|
|
72
|
+
# @param n [Integer] entries wanted; capped at #remaining
|
|
73
|
+
# @param max_def [Integer] the column's max definition level
|
|
74
|
+
# @return [Numo::Bit, nil] 1 where the entry holds a value
|
|
75
|
+
def read_validity_numo(n, max_def)
|
|
76
|
+
n = @remaining if n > @remaining
|
|
77
|
+
@remaining -= n
|
|
78
|
+
defs = @defs or return nil
|
|
79
|
+
return defs.read_flags(n) if max_def == 1 && defs.respond_to?(:read_flags)
|
|
80
|
+
levels = defs.respond_to?(:read_numo) ? defs.read_numo(n, Numo::UInt8) : Numo::UInt8.cast(defs.read(n))
|
|
81
|
+
levels.eq(max_def)
|
|
82
|
+
end
|
|
83
|
+
end
|
|
84
|
+
|
|
85
|
+
# A decoder over an Array that is already decoded (legacy encodings, booleans, deltas)
|
|
86
|
+
class ArrayDecoder
|
|
87
|
+
# @param values [Array] all of the page's decoded levels or values
|
|
88
|
+
def initialize(values)
|
|
89
|
+
@values = values
|
|
90
|
+
@i = 0
|
|
91
|
+
end
|
|
92
|
+
|
|
93
|
+
# @param n [Integer] number of entries to hand out
|
|
94
|
+
# @return [Array] the next +n+ entries
|
|
95
|
+
# @raise [FormatError] when fewer than +n+ are left
|
|
96
|
+
def read(n)
|
|
97
|
+
raise FormatError, "Page has fewer values than its levels require" if @i + n > @values.size
|
|
98
|
+
out = @values[@i, n]
|
|
99
|
+
@i += n
|
|
100
|
+
out
|
|
101
|
+
end
|
|
102
|
+
end
|
|
103
|
+
|
|
104
|
+
# The RLE / bit-packed hybrid, decoded run by run
|
|
105
|
+
class HybridDecoder
|
|
106
|
+
# @param data [String] binary page data
|
|
107
|
+
# @param pos [Integer] offset of the first run header
|
|
108
|
+
# @param limit [Integer] offset just past the encoded runs
|
|
109
|
+
# @param width [Integer] bit width of each value
|
|
110
|
+
def initialize(data, pos, limit, width)
|
|
111
|
+
@data = data
|
|
112
|
+
@pos = pos
|
|
113
|
+
@limit = limit
|
|
114
|
+
@width = width
|
|
115
|
+
@value_bytes = (width + 7) / 8
|
|
116
|
+
@left = 0 # values left in the current run
|
|
117
|
+
@rle = true
|
|
118
|
+
@value = 0
|
|
119
|
+
@buf = nil
|
|
120
|
+
@bi = 0
|
|
121
|
+
@groups = 0 # bit-packed groups of 8 not unpacked yet
|
|
122
|
+
end
|
|
123
|
+
|
|
124
|
+
# Bit-packed runs are unpacked up to CHUNK values at a time
|
|
125
|
+
#
|
|
126
|
+
# @param n [Integer] number of values to decode
|
|
127
|
+
# @return [Array<Integer>] the next +n+ values
|
|
128
|
+
# @raise [FormatError] when the runs end before +n+ values
|
|
129
|
+
def read(n)
|
|
130
|
+
out = []
|
|
131
|
+
while n > 0
|
|
132
|
+
next_run if @left.zero?
|
|
133
|
+
t = (n < @left) ? n : @left
|
|
134
|
+
if @rle
|
|
135
|
+
out.fill(@value, out.size, t)
|
|
136
|
+
else
|
|
137
|
+
if @buf.nil? || @bi >= @buf.size
|
|
138
|
+
g = (@groups < CHUNK / 8) ? @groups : CHUNK / 8
|
|
139
|
+
@buf = Encodings::RLE.unpack_bits(@data, @pos, g * 8, @width)
|
|
140
|
+
@pos += g * @width
|
|
141
|
+
@groups -= g
|
|
142
|
+
@bi = 0
|
|
143
|
+
end
|
|
144
|
+
avail = @buf.size - @bi
|
|
145
|
+
t = avail if t > avail
|
|
146
|
+
out.concat(@buf[@bi, t])
|
|
147
|
+
@bi += t
|
|
148
|
+
end
|
|
149
|
+
@left -= t
|
|
150
|
+
n -= t
|
|
151
|
+
end
|
|
152
|
+
out
|
|
153
|
+
end
|
|
154
|
+
|
|
155
|
+
# For a bit width of 1: the next +n+ values as a Numo::Bit (read(as: :numo)). Runs are
|
|
156
|
+
# collected as "0"/"1" characters, which costs less per run than Numo calls do (levels of
|
|
157
|
+
# columns with scattered nulls come in many short runs).
|
|
158
|
+
#
|
|
159
|
+
# @param n [Integer] number of values to decode
|
|
160
|
+
# @return [Numo::Bit] the next +n+ values
|
|
161
|
+
# @raise [ArgumentError] when the bit width is not 1
|
|
162
|
+
# @raise [FormatError] when the runs end before +n+ values
|
|
163
|
+
def read_flags(n)
|
|
164
|
+
raise ArgumentError, "read_flags needs a bit width of 1" unless @width == 1
|
|
165
|
+
out = String.new(capacity: n, encoding: Encoding::BINARY)
|
|
166
|
+
while n > 0
|
|
167
|
+
next_run if @left.zero?
|
|
168
|
+
t = (n < @left) ? n : @left
|
|
169
|
+
if @rle
|
|
170
|
+
out << ((@value == 1) ? "1" : "0") * t
|
|
171
|
+
elsif @buf && @bi < @buf.size
|
|
172
|
+
avail = @buf.size - @bi
|
|
173
|
+
t = avail if t > avail
|
|
174
|
+
out << @buf[@bi, t].join
|
|
175
|
+
@bi += t
|
|
176
|
+
else
|
|
177
|
+
groups = (t + 7) / 8
|
|
178
|
+
bits = @data.byteslice(@pos, groups)&.unpack1("b*") || +""
|
|
179
|
+
bits << "0" * (groups * 8 - bits.bytesize) if bits.bytesize < groups * 8 # truncated last run
|
|
180
|
+
@pos += groups
|
|
181
|
+
@groups -= groups
|
|
182
|
+
if groups * 8 > t
|
|
183
|
+
out << bits.byteslice(0, t)
|
|
184
|
+
@buf = bits.byteslice(t, groups * 8 - t).bytes.map! { |c| c - 48 }
|
|
185
|
+
else
|
|
186
|
+
out << bits
|
|
187
|
+
@buf = nil
|
|
188
|
+
end
|
|
189
|
+
@bi = 0
|
|
190
|
+
end
|
|
191
|
+
@left -= t
|
|
192
|
+
n -= t
|
|
193
|
+
end
|
|
194
|
+
Numo::UInt8.from_binary(out).eq(49)
|
|
195
|
+
end
|
|
196
|
+
|
|
197
|
+
# The next +n+ values as a Numo array of +klass+ (read(as: :numo)): RLE runs are
|
|
198
|
+
# filled and bit-packed runs unpacked by Numo, with no Ruby object per value. Leaves the
|
|
199
|
+
# decoder in a state #read can continue from.
|
|
200
|
+
#
|
|
201
|
+
# @param n [Integer] number of values to decode
|
|
202
|
+
# @param klass [Class] Numo integer class to fill, e.g. Numo::UInt8 or Numo::Int32
|
|
203
|
+
# @return [Numo::NArray] the next +n+ values as a +klass+ array
|
|
204
|
+
# @raise [FormatError] when the runs end before +n+ values
|
|
205
|
+
def read_numo(n, klass)
|
|
206
|
+
out = klass.zeros(n)
|
|
207
|
+
i = 0
|
|
208
|
+
while n > 0
|
|
209
|
+
next_run if @left.zero?
|
|
210
|
+
t = (n < @left) ? n : @left
|
|
211
|
+
if @rle
|
|
212
|
+
out[i...i + t] = @value
|
|
213
|
+
elsif @buf && @bi < @buf.size
|
|
214
|
+
# Values of this run that #read (or a previous call) already unpacked
|
|
215
|
+
avail = @buf.size - @bi
|
|
216
|
+
t = avail if t > avail
|
|
217
|
+
out[i...i + t] = @buf[@bi, t]
|
|
218
|
+
@bi += t
|
|
219
|
+
else
|
|
220
|
+
groups = (t + 7) / 8 # @left == @groups * 8 here, so these groups exist
|
|
221
|
+
vals = NumoColumns.unpack_bits(@data, @pos, groups * 8, @width)
|
|
222
|
+
@pos += groups * @width
|
|
223
|
+
@groups -= groups
|
|
224
|
+
# A partly used group goes to the buffer #read and this method take values from
|
|
225
|
+
@buf = (groups * 8 > t) ? vals[t..].to_a : nil
|
|
226
|
+
@bi = 0
|
|
227
|
+
out[i...i + t] = (groups * 8 > t) ? vals[0...t] : vals
|
|
228
|
+
end
|
|
229
|
+
@left -= t
|
|
230
|
+
n -= t
|
|
231
|
+
i += t
|
|
232
|
+
end
|
|
233
|
+
out
|
|
234
|
+
end
|
|
235
|
+
|
|
236
|
+
private
|
|
237
|
+
|
|
238
|
+
# Reads run headers until a non-empty run starts: an RLE run (count and one value) or
|
|
239
|
+
# a bit-packed run (groups of 8 values)
|
|
240
|
+
#
|
|
241
|
+
# @return [void]
|
|
242
|
+
# @raise [FormatError] when there are no more runs before +limit+
|
|
243
|
+
def next_run
|
|
244
|
+
while @left.zero?
|
|
245
|
+
raise FormatError, "RLE data exhausted" if @pos >= @limit
|
|
246
|
+
header, @pos = Encodings::RLE.read_uleb(@data, @pos)
|
|
247
|
+
if header & 1 == 1
|
|
248
|
+
@rle = false
|
|
249
|
+
@groups = header >> 1
|
|
250
|
+
@left = @groups * 8
|
|
251
|
+
@buf = nil
|
|
252
|
+
else
|
|
253
|
+
@rle = true
|
|
254
|
+
@left = header >> 1
|
|
255
|
+
v = 0
|
|
256
|
+
@value_bytes.times { |k| v |= @data.getbyte(@pos + k).to_i << (8 * k) }
|
|
257
|
+
@value = v
|
|
258
|
+
@pos += @value_bytes
|
|
259
|
+
end
|
|
260
|
+
end
|
|
261
|
+
end
|
|
262
|
+
end
|
|
263
|
+
|
|
264
|
+
# PLAIN INT32 / INT64 / FLOAT / DOUBLE / INT96
|
|
265
|
+
class FixedDecoder
|
|
266
|
+
# @param data [String] binary page data
|
|
267
|
+
# @param pos [Integer] offset of the first value
|
|
268
|
+
# @param format [String] String#unpack directive of one value, e.g. "l<"
|
|
269
|
+
# @param width [Integer] bytes per value
|
|
270
|
+
def initialize(data, pos, format, width)
|
|
271
|
+
@data = data
|
|
272
|
+
@pos = pos
|
|
273
|
+
@format = format
|
|
274
|
+
@width = width
|
|
275
|
+
end
|
|
276
|
+
|
|
277
|
+
# @param n [Integer] number of values to decode
|
|
278
|
+
# @return [Array<Integer>, Array<Float>] the next +n+ values
|
|
279
|
+
# @raise [FormatError] when the page holds fewer values
|
|
280
|
+
def read(n)
|
|
281
|
+
bytes = n * @width
|
|
282
|
+
raise FormatError, "Truncated PLAIN data" if @pos + bytes > @data.bytesize
|
|
283
|
+
out = @data.byteslice(@pos, bytes).unpack("#{@format}#{n}")
|
|
284
|
+
@pos += bytes
|
|
285
|
+
out
|
|
286
|
+
end
|
|
287
|
+
|
|
288
|
+
# @param n [Integer] number of values to move past
|
|
289
|
+
# @return [void]
|
|
290
|
+
# @raise [FormatError] when the page holds fewer values
|
|
291
|
+
def skip(n)
|
|
292
|
+
raise FormatError, "Truncated PLAIN data" if @pos + n * @width > @data.bytesize
|
|
293
|
+
@pos += n * @width
|
|
294
|
+
end
|
|
295
|
+
|
|
296
|
+
# The next +n+ values as their PLAIN (little-endian) bytes, for read(as: :numo)
|
|
297
|
+
#
|
|
298
|
+
# @param n [Integer] number of values
|
|
299
|
+
# @return [String] +n+ * width bytes
|
|
300
|
+
# @raise [FormatError] when the page holds fewer values
|
|
301
|
+
def read_bytes(n)
|
|
302
|
+
bytes = n * @width
|
|
303
|
+
raise FormatError, "Truncated PLAIN data" if @pos + bytes > @data.bytesize
|
|
304
|
+
out = @data.byteslice(@pos, bytes)
|
|
305
|
+
@pos += bytes
|
|
306
|
+
out
|
|
307
|
+
end
|
|
308
|
+
end
|
|
309
|
+
|
|
310
|
+
# PLAIN INT96 (legacy Impala/Spark timestamps): 8 bytes of nanoseconds, 4 of Julian day
|
|
311
|
+
class Int96Decoder < FixedDecoder
|
|
312
|
+
# @param data [String] binary page data
|
|
313
|
+
# @param pos [Integer] offset of the first value
|
|
314
|
+
def initialize(data, pos)
|
|
315
|
+
super(data, pos, "Q<L<", 12)
|
|
316
|
+
end
|
|
317
|
+
|
|
318
|
+
# @param n [Integer] number of values to decode
|
|
319
|
+
# @return [Array<Array(Integer, Integer)>] [nanoseconds of the day, Julian day] per value
|
|
320
|
+
# @raise [FormatError] when the page holds fewer values
|
|
321
|
+
def read(n)
|
|
322
|
+
bytes = n * 12
|
|
323
|
+
raise FormatError, "Truncated INT96 data" if @pos + bytes > @data.bytesize
|
|
324
|
+
out = @data.byteslice(@pos, bytes).unpack("Q<L<" * n).each_slice(2).to_a
|
|
325
|
+
@pos += bytes
|
|
326
|
+
out
|
|
327
|
+
end
|
|
328
|
+
end
|
|
329
|
+
|
|
330
|
+
# PLAIN FIXED_LEN_BYTE_ARRAY
|
|
331
|
+
class FixedBytesDecoder
|
|
332
|
+
# @param data [String] binary page data
|
|
333
|
+
# @param pos [Integer] offset of the first value
|
|
334
|
+
# @param width [Integer] the column's type_length
|
|
335
|
+
def initialize(data, pos, width)
|
|
336
|
+
@data = data
|
|
337
|
+
@pos = pos
|
|
338
|
+
@width = width
|
|
339
|
+
end
|
|
340
|
+
|
|
341
|
+
# @param n [Integer] number of values to decode
|
|
342
|
+
# @return [Array<String>] the next +n+ values as binary Strings
|
|
343
|
+
# @raise [FormatError] when the page holds fewer values
|
|
344
|
+
def read(n)
|
|
345
|
+
raise FormatError, "Truncated FIXED_LEN_BYTE_ARRAY data" if @pos + n * @width > @data.bytesize
|
|
346
|
+
out = Array.new(n) { |i| @data.byteslice(@pos + i * @width, @width) }
|
|
347
|
+
@pos += n * @width
|
|
348
|
+
out
|
|
349
|
+
end
|
|
350
|
+
|
|
351
|
+
# @param n [Integer] number of values to move past
|
|
352
|
+
# @return [void]
|
|
353
|
+
# @raise [FormatError] when the page holds fewer values
|
|
354
|
+
def skip(n)
|
|
355
|
+
raise FormatError, "Truncated FIXED_LEN_BYTE_ARRAY data" if @pos + n * @width > @data.bytesize
|
|
356
|
+
@pos += n * @width
|
|
357
|
+
end
|
|
358
|
+
end
|
|
359
|
+
|
|
360
|
+
# PLAIN BYTE_ARRAY: 4-byte length, then the bytes
|
|
361
|
+
class ByteArrayDecoder
|
|
362
|
+
# @param data [String] binary page data
|
|
363
|
+
# @param pos [Integer] offset of the first length prefix
|
|
364
|
+
def initialize(data, pos)
|
|
365
|
+
@data = data
|
|
366
|
+
@pos = pos
|
|
367
|
+
end
|
|
368
|
+
|
|
369
|
+
# @param n [Integer] number of values to decode
|
|
370
|
+
# @return [Array<String>] the next +n+ values as binary Strings
|
|
371
|
+
# @raise [FormatError] when a length prefix or value runs past the page
|
|
372
|
+
def read(n)
|
|
373
|
+
data = @data
|
|
374
|
+
size = data.bytesize
|
|
375
|
+
pos = @pos
|
|
376
|
+
out = Array.new(n)
|
|
377
|
+
i = 0
|
|
378
|
+
while i < n
|
|
379
|
+
raise FormatError, "Truncated BYTE_ARRAY data" if pos + 4 > size
|
|
380
|
+
len = data.getbyte(pos) | (data.getbyte(pos + 1) << 8) | (data.getbyte(pos + 2) << 16) | (data.getbyte(pos + 3) << 24)
|
|
381
|
+
pos += 4
|
|
382
|
+
raise FormatError, "BYTE_ARRAY value overruns page" if pos + len > size
|
|
383
|
+
out[i] = data.byteslice(pos, len)
|
|
384
|
+
pos += len
|
|
385
|
+
i += 1
|
|
386
|
+
end
|
|
387
|
+
@pos = pos
|
|
388
|
+
out
|
|
389
|
+
end
|
|
390
|
+
|
|
391
|
+
# Walks the length prefixes without creating Strings
|
|
392
|
+
#
|
|
393
|
+
# @param n [Integer] number of values to move past
|
|
394
|
+
# @return [void]
|
|
395
|
+
# @raise [FormatError] when a length prefix or value runs past the page
|
|
396
|
+
def skip(n)
|
|
397
|
+
data = @data
|
|
398
|
+
size = data.bytesize
|
|
399
|
+
pos = @pos
|
|
400
|
+
n.times do
|
|
401
|
+
raise FormatError, "Truncated BYTE_ARRAY data" if pos + 4 > size
|
|
402
|
+
pos += 4 + (data.getbyte(pos) | (data.getbyte(pos + 1) << 8) | (data.getbyte(pos + 2) << 16) | (data.getbyte(pos + 3) << 24))
|
|
403
|
+
end
|
|
404
|
+
raise FormatError, "BYTE_ARRAY value overruns page" if pos > size
|
|
405
|
+
@pos = pos
|
|
406
|
+
end
|
|
407
|
+
end
|
|
408
|
+
|
|
409
|
+
# PLAIN BOOLEAN: one bit per value, LSB first
|
|
410
|
+
class BooleanDecoder
|
|
411
|
+
# @param data [String] binary page data
|
|
412
|
+
# @param pos [Integer] offset of the first value; the rest of +data+ is unpacked at once
|
|
413
|
+
def initialize(data, pos)
|
|
414
|
+
@bits = data.byteslice(pos, data.bytesize - pos).unpack1("b*")
|
|
415
|
+
@i = 0
|
|
416
|
+
end
|
|
417
|
+
|
|
418
|
+
# @param n [Integer] number of values to decode
|
|
419
|
+
# @return [Array<Boolean>] the next +n+ values
|
|
420
|
+
# @raise [FormatError] when the page holds fewer values
|
|
421
|
+
def read(n)
|
|
422
|
+
raise FormatError, "Truncated BOOLEAN data" if @i + n > @bits.bytesize
|
|
423
|
+
out = Array.new(n) { |k| @bits.getbyte(@i + k) == 49 }
|
|
424
|
+
@i += n
|
|
425
|
+
out
|
|
426
|
+
end
|
|
427
|
+
|
|
428
|
+
# The next +n+ values as a Numo::Bit (read(as: :numo))
|
|
429
|
+
#
|
|
430
|
+
# @param n [Integer] number of values to decode
|
|
431
|
+
# @return [Numo::Bit] 1 for true
|
|
432
|
+
# @raise [FormatError] when the page holds fewer values
|
|
433
|
+
def read_numo(n)
|
|
434
|
+
raise FormatError, "Truncated BOOLEAN data" if @i + n > @bits.bytesize
|
|
435
|
+
out = Numo::UInt8.from_binary(@bits.byteslice(@i, n)).eq(49) # "0"/"1" characters
|
|
436
|
+
@i += n
|
|
437
|
+
out
|
|
438
|
+
end
|
|
439
|
+
end
|
|
440
|
+
|
|
441
|
+
# RLE_DICTIONARY / PLAIN_DICTIONARY indices mapped to the (already converted) dictionary
|
|
442
|
+
class DictionaryDecoder
|
|
443
|
+
# @return [Array] the chunk's dictionary values (converted, shared by all its pages)
|
|
444
|
+
attr_reader :dictionary
|
|
445
|
+
|
|
446
|
+
# @param data [String] binary page data
|
|
447
|
+
# @param pos [Integer] offset of the bit width byte that precedes the RLE/bit-packed indices
|
|
448
|
+
# @param dictionary [Array] values the indices point into, already converted
|
|
449
|
+
# @param path [String] dotted column path, for error messages
|
|
450
|
+
def initialize(data, pos, dictionary, path)
|
|
451
|
+
@indices = HybridDecoder.new(data, pos + 1, data.bytesize, data.getbyte(pos).to_i)
|
|
452
|
+
@dictionary = dictionary
|
|
453
|
+
@path = path
|
|
454
|
+
end
|
|
455
|
+
|
|
456
|
+
# Indices are decoded but not looked up
|
|
457
|
+
#
|
|
458
|
+
# @param n [Integer] number of values to move past
|
|
459
|
+
# @return [void]
|
|
460
|
+
# @raise [FormatError] when the indices run out
|
|
461
|
+
def skip(n)
|
|
462
|
+
@indices.read(n)
|
|
463
|
+
end
|
|
464
|
+
|
|
465
|
+
# @param n [Integer] number of values to decode; must be positive
|
|
466
|
+
# @return [Array] the dictionary values at the next +n+ indices
|
|
467
|
+
# @raise [FormatError] when an index is out of range or the indices run out
|
|
468
|
+
def read(n)
|
|
469
|
+
indices = @indices.read(n)
|
|
470
|
+
dict = @dictionary
|
|
471
|
+
raise FormatError, "Dictionary index out of range in #{@path}" if indices.max >= dict.size
|
|
472
|
+
indices.map! { |i| dict[i] }
|
|
473
|
+
end
|
|
474
|
+
|
|
475
|
+
# The next +n+ indices as a Numo::Int32, bounds-checked (read(as: :numo))
|
|
476
|
+
#
|
|
477
|
+
# @param n [Integer] number of indices to decode
|
|
478
|
+
# @return [Numo::Int32] indices into #dictionary
|
|
479
|
+
# @raise [FormatError] when an index is out of range or the indices run out
|
|
480
|
+
def read_indices_numo(n)
|
|
481
|
+
indices = @indices.read_numo(n, Numo::Int32)
|
|
482
|
+
raise FormatError, "Dictionary index out of range in #{@path}" if n.positive? && indices.max >= @dictionary.size
|
|
483
|
+
indices
|
|
484
|
+
end
|
|
485
|
+
end
|
|
486
|
+
|
|
487
|
+
# RLE-encoded BOOLEAN values
|
|
488
|
+
class RleBooleanDecoder
|
|
489
|
+
# @param data [String] binary page data
|
|
490
|
+
# @param pos [Integer] offset of the 4-byte length prefix of the RLE data
|
|
491
|
+
def initialize(data, pos)
|
|
492
|
+
len = data.byteslice(pos, 4).unpack1("V")
|
|
493
|
+
@bits = HybridDecoder.new(data, pos + 4, pos + 4 + len, 1)
|
|
494
|
+
end
|
|
495
|
+
|
|
496
|
+
# @param n [Integer] number of values to decode
|
|
497
|
+
# @return [Array<Boolean>] the next +n+ values
|
|
498
|
+
# @raise [FormatError] when the runs end before +n+ values
|
|
499
|
+
def read(n)
|
|
500
|
+
@bits.read(n).map! { |v| v == 1 }
|
|
501
|
+
end
|
|
502
|
+
|
|
503
|
+
# The next +n+ values as a Numo::Bit (read(as: :numo))
|
|
504
|
+
#
|
|
505
|
+
# @param n [Integer] number of values to decode
|
|
506
|
+
# @return [Numo::Bit] 1 for true
|
|
507
|
+
# @raise [FormatError] when the runs end before +n+ values
|
|
508
|
+
def read_numo(n)
|
|
509
|
+
@bits.read_flags(n)
|
|
510
|
+
end
|
|
511
|
+
end
|
|
512
|
+
|
|
513
|
+
# DELTA_LENGTH_BYTE_ARRAY: all lengths first (Integers), then the bytes sliced as needed
|
|
514
|
+
class DeltaLengthDecoder
|
|
515
|
+
# @param data [String] binary page data
|
|
516
|
+
# @param pos [Integer] offset of the DELTA_BINARY_PACKED lengths
|
|
517
|
+
# @raise [FormatError] when the lengths are malformed
|
|
518
|
+
def initialize(data, pos)
|
|
519
|
+
@lengths, @pos = Encodings::Delta.decode_binary_packed(data, pos, 32)
|
|
520
|
+
@data = data
|
|
521
|
+
@i = 0
|
|
522
|
+
end
|
|
523
|
+
|
|
524
|
+
# @param n [Integer] number of values to decode
|
|
525
|
+
# @return [Array<String>] the next +n+ values as binary Strings
|
|
526
|
+
# @raise [FormatError] when there are too few lengths or a value runs past the page
|
|
527
|
+
def read(n)
|
|
528
|
+
raise FormatError, "DELTA_LENGTH_BYTE_ARRAY has too few values" if @i + n > @lengths.size
|
|
529
|
+
data = @data
|
|
530
|
+
out = Array.new(n) do |k|
|
|
531
|
+
len = @lengths[@i + k]
|
|
532
|
+
raise FormatError, "DELTA_LENGTH_BYTE_ARRAY value overruns page" if len.negative? || @pos + len > data.bytesize
|
|
533
|
+
s = data.byteslice(@pos, len)
|
|
534
|
+
@pos += len
|
|
535
|
+
s
|
|
536
|
+
end
|
|
537
|
+
@i += n
|
|
538
|
+
out
|
|
539
|
+
end
|
|
540
|
+
end
|
|
541
|
+
|
|
542
|
+
# DELTA_BYTE_ARRAY: prefix lengths and suffixes, each value built from the previous one
|
|
543
|
+
class DeltaByteArrayDecoder
|
|
544
|
+
# @param data [String] binary page data
|
|
545
|
+
# @param pos [Integer] offset of the DELTA_BINARY_PACKED prefix lengths
|
|
546
|
+
# @raise [FormatError] when the prefix or suffix lengths are malformed
|
|
547
|
+
def initialize(data, pos)
|
|
548
|
+
@prefixes, pos = Encodings::Delta.decode_binary_packed(data, pos, 32)
|
|
549
|
+
@suffixes = DeltaLengthDecoder.new(data, pos)
|
|
550
|
+
@prev = "".b
|
|
551
|
+
@i = 0
|
|
552
|
+
end
|
|
553
|
+
|
|
554
|
+
# @param n [Integer] number of values to decode
|
|
555
|
+
# @return [Array<String>] the next +n+ values as binary Strings
|
|
556
|
+
# @raise [FormatError] when there are too few values or a prefix is negative or longer than
|
|
557
|
+
# the previous value
|
|
558
|
+
def read(n)
|
|
559
|
+
raise FormatError, "DELTA_BYTE_ARRAY has too few values" if @i + n > @prefixes.size
|
|
560
|
+
suffixes = @suffixes.read(n)
|
|
561
|
+
prev = @prev
|
|
562
|
+
out = Array.new(n) do |k|
|
|
563
|
+
prefix = @prefixes[@i + k]
|
|
564
|
+
raise FormatError, "DELTA_BYTE_ARRAY prefix length #{prefix} out of range" if prefix > prev.bytesize || prefix.negative?
|
|
565
|
+
prev = prefix.zero? ? suffixes[k] : prev.byteslice(0, prefix) + suffixes[k]
|
|
566
|
+
end
|
|
567
|
+
@prev = prev
|
|
568
|
+
@i += n
|
|
569
|
+
out
|
|
570
|
+
end
|
|
571
|
+
end
|
|
572
|
+
|
|
573
|
+
# BYTE_STREAM_SPLIT: byte k of every value lives in stream k; values are gathered per call
|
|
574
|
+
class ByteStreamSplitDecoder
|
|
575
|
+
# @param data [String] binary page data; the streams run to its end
|
|
576
|
+
# @param pos [Integer] offset of the first stream
|
|
577
|
+
# @param width [Integer] bytes per value (and number of streams)
|
|
578
|
+
# @param type [Integer] Format::Type of the column
|
|
579
|
+
# @param type_length [Integer, nil] value size in bytes for FIXED_LEN_BYTE_ARRAY, else unused
|
|
580
|
+
def initialize(data, pos, width, type, type_length)
|
|
581
|
+
@data = data
|
|
582
|
+
@pos = pos
|
|
583
|
+
@width = width
|
|
584
|
+
@count = (data.bytesize - pos) / width
|
|
585
|
+
@type = type
|
|
586
|
+
@type_length = type_length
|
|
587
|
+
@i = 0
|
|
588
|
+
end
|
|
589
|
+
|
|
590
|
+
# @param n [Integer] number of values to decode
|
|
591
|
+
# @return [Array<Integer>, Array<Float>, Array<String>] the next +n+ values, decoded as PLAIN
|
|
592
|
+
# @raise [FormatError] when the streams hold fewer values
|
|
593
|
+
def read(n)
|
|
594
|
+
raise FormatError, "Truncated BYTE_STREAM_SPLIT data" if @i + n > @count
|
|
595
|
+
streams = Array.new(@width) { |k| @data.byteslice(@pos + k * @count + @i, n) }.join
|
|
596
|
+
@i += n
|
|
597
|
+
plain, = Encodings::ByteStreamSplit.decode(streams, 0, n, @width)
|
|
598
|
+
Encodings::Plain.decode(plain, 0, n, @type, @type_length).first
|
|
599
|
+
end
|
|
600
|
+
|
|
601
|
+
# The next +n+ values re-interleaved into PLAIN bytes by Numo (read(as: :numo))
|
|
602
|
+
#
|
|
603
|
+
# @param n [Integer] number of values
|
|
604
|
+
# @return [String] +n+ * width bytes in PLAIN layout
|
|
605
|
+
# @raise [FormatError] when the streams hold fewer values
|
|
606
|
+
def read_bytes(n)
|
|
607
|
+
raise FormatError, "Truncated BYTE_STREAM_SPLIT data" if @i + n > @count
|
|
608
|
+
streams = Array.new(@width) { |k| @data.byteslice(@pos + k * @count + @i, n) }.join
|
|
609
|
+
@i += n
|
|
610
|
+
Numo::UInt8.from_binary(streams, [@width, n]).transpose.to_binary.b
|
|
611
|
+
end
|
|
612
|
+
end
|
|
613
|
+
end
|
|
614
|
+
end
|
|
615
|
+
end
|