herringbone 0.1.0 → 0.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/README.md +198 -74
- data/bin/herringbone +38 -26
- data/lib/herringbone/active_record.rb +48 -19
- data/lib/herringbone/bloom_filter.rb +370 -0
- data/lib/herringbone/byte_values.rb +29 -0
- data/lib/herringbone/codecs/lz4.rb +66 -3
- data/lib/herringbone/codecs/snappy.rb +132 -3
- data/lib/herringbone/compression.rb +121 -16
- data/lib/herringbone/encodings/delta.rb +76 -19
- data/lib/herringbone/encodings/plain.rb +29 -7
- data/lib/herringbone/encodings/rle.rb +55 -3
- data/lib/herringbone/format.rb +203 -0
- data/lib/herringbone/inspector.rb +1784 -0
- data/lib/herringbone/io_buffer_support.rb +3 -1
- data/lib/herringbone/reader/column_chunk_reader.rb +388 -0
- data/lib/herringbone/reader/column_cursor.rb +225 -0
- data/lib/herringbone/reader/numo.rb +365 -0
- data/lib/herringbone/reader/page_stream.rb +615 -0
- data/lib/herringbone/reader/scan.rb +350 -0
- data/lib/herringbone/reader.rb +585 -272
- data/lib/herringbone/schema.rb +326 -90
- data/lib/herringbone/thrift.rb +155 -4
- data/lib/herringbone/types.rb +177 -42
- data/lib/herringbone/version.rb +2 -1
- data/lib/herringbone/visualizer.rb +1136 -0
- data/lib/herringbone/writer.rb +550 -105
- data/lib/herringbone/xxhash.rb +425 -0
- data/lib/herringbone.rb +64 -9
- metadata +14 -33
|
@@ -7,6 +7,7 @@ module Herringbone
|
|
|
7
7
|
module Plain
|
|
8
8
|
module_function
|
|
9
9
|
|
|
10
|
+
# Fixed-width numeric types: pack/unpack directive and byte width (all little-endian).
|
|
10
11
|
FORMATS = {
|
|
11
12
|
Format::Type::INT32 => ["l<", 4],
|
|
12
13
|
Format::Type::INT64 => ["q<", 8],
|
|
@@ -16,43 +17,57 @@ module Herringbone
|
|
|
16
17
|
|
|
17
18
|
# Decodes +count+ values of +type+ from +data+ starting at +pos+.
|
|
18
19
|
# Returns [values, new_pos].
|
|
20
|
+
# INT96 values come back as [nanoseconds_of_day, julian_day] pairs.
|
|
21
|
+
# @param data [String] binary page data
|
|
22
|
+
# @param pos [Integer] byte offset of the first value
|
|
23
|
+
# @param count [Integer] number of values to decode
|
|
24
|
+
# @param type [Integer] physical type, a Format::Type constant
|
|
25
|
+
# @param type_length [Integer, nil] byte width, required for FIXED_LEN_BYTE_ARRAY
|
|
26
|
+
# @return [Array(Array, Integer)] the decoded values and the offset just past them
|
|
27
|
+
# @raise [FormatError] if the data is truncated or the type is unknown
|
|
19
28
|
def decode(data, pos, count, type, type_length = nil)
|
|
20
29
|
case type
|
|
21
30
|
when Format::Type::BOOLEAN
|
|
22
31
|
nbytes = (count + 7) / 8
|
|
23
32
|
bits = data.byteslice(pos, nbytes).unpack1("b*")
|
|
24
|
-
raise
|
|
33
|
+
raise FormatError, "Truncated BOOLEAN data" if bits.bytesize < count
|
|
25
34
|
[Array.new(count) { |i| bits.getbyte(i) == 49 }, pos + nbytes]
|
|
26
35
|
when Format::Type::INT32, Format::Type::INT64, Format::Type::FLOAT, Format::Type::DOUBLE
|
|
27
36
|
fmt, width = FORMATS[type]
|
|
28
37
|
nbytes = count * width
|
|
29
|
-
raise
|
|
38
|
+
raise FormatError, "Truncated PLAIN data" if pos + nbytes > data.bytesize
|
|
30
39
|
[data.byteslice(pos, nbytes).unpack("#{fmt}#{count}"), pos + nbytes]
|
|
31
40
|
when Format::Type::INT96
|
|
32
41
|
nbytes = count * 12
|
|
33
|
-
raise
|
|
42
|
+
raise FormatError, "Truncated INT96 data" if pos + nbytes > data.bytesize
|
|
34
43
|
vals = data.byteslice(pos, nbytes).unpack("Q<L<" * count).each_slice(2).map { |nanos, day| [nanos, day] }
|
|
35
44
|
[vals, pos + nbytes]
|
|
36
45
|
when Format::Type::BYTE_ARRAY
|
|
37
46
|
decode_byte_arrays(data, pos, count)
|
|
38
47
|
when Format::Type::FIXED_LEN_BYTE_ARRAY
|
|
39
48
|
nbytes = count * type_length
|
|
40
|
-
raise
|
|
49
|
+
raise FormatError, "Truncated FIXED_LEN_BYTE_ARRAY data" if pos + nbytes > data.bytesize
|
|
41
50
|
[Array.new(count) { |i| data.byteslice(pos + i * type_length, type_length) }, pos + nbytes]
|
|
42
51
|
else
|
|
43
|
-
raise
|
|
52
|
+
raise FormatError, "Unknown physical type #{type}"
|
|
44
53
|
end
|
|
45
54
|
end
|
|
46
55
|
|
|
56
|
+
# Decodes PLAIN BYTE_ARRAY values: each is a 4-byte little-endian length followed by the bytes.
|
|
57
|
+
# @param data [String] binary page data
|
|
58
|
+
# @param pos [Integer] byte offset of the first length prefix
|
|
59
|
+
# @param count [Integer] number of values to decode
|
|
60
|
+
# @return [Array(Array<String>, Integer)] binary slices of +data+ and the offset just past them
|
|
61
|
+
# @raise [FormatError] if a length prefix or value runs past the end of +data+
|
|
47
62
|
def decode_byte_arrays(data, pos, count)
|
|
48
63
|
out = Array.new(count)
|
|
49
64
|
size = data.bytesize
|
|
50
65
|
i = 0
|
|
51
66
|
while i < count
|
|
52
|
-
raise
|
|
67
|
+
raise FormatError, "Truncated BYTE_ARRAY data" if pos + 4 > size
|
|
53
68
|
len = data.getbyte(pos) | (data.getbyte(pos + 1) << 8) | (data.getbyte(pos + 2) << 16) | (data.getbyte(pos + 3) << 24)
|
|
54
69
|
pos += 4
|
|
55
|
-
raise
|
|
70
|
+
raise FormatError, "BYTE_ARRAY value overruns page" if pos + len > size
|
|
56
71
|
out[i] = data.byteslice(pos, len)
|
|
57
72
|
pos += len
|
|
58
73
|
i += 1
|
|
@@ -60,6 +75,13 @@ module Herringbone
|
|
|
60
75
|
[out, pos]
|
|
61
76
|
end
|
|
62
77
|
|
|
78
|
+
# Encodes values of a physical type with PLAIN. Inverse of decode; INT96 values are
|
|
79
|
+
# [nanoseconds_of_day, julian_day] pairs and booleans are packed LSB-first, 8 per byte.
|
|
80
|
+
# @param values [Array] values in their physical Ruby form, without nulls
|
|
81
|
+
# @param type [Integer] physical type, a Format::Type constant
|
|
82
|
+
# @param type_length [Integer, nil] byte width, required for FIXED_LEN_BYTE_ARRAY
|
|
83
|
+
# @return [String] encoded bytes in ASCII-8BIT
|
|
84
|
+
# @raise [EncodeError] if a FIXED_LEN_BYTE_ARRAY value has the wrong size or the type is unknown
|
|
63
85
|
def encode(values, type, type_length = nil)
|
|
64
86
|
case type
|
|
65
87
|
when Format::Type::BOOLEAN
|
|
@@ -1,6 +1,9 @@
|
|
|
1
1
|
# frozen_string_literal: true
|
|
2
2
|
|
|
3
3
|
module Herringbone
|
|
4
|
+
# Decoders and encoders for the Parquet value and level encodings: PLAIN, the RLE / bit-packed
|
|
5
|
+
# hybrid, the DELTA_* family and BYTE_STREAM_SPLIT. They work on binary Strings and byte offsets,
|
|
6
|
+
# and know nothing about pages or columns.
|
|
4
7
|
module Encodings
|
|
5
8
|
# Bit packing (LSB-first, as used by Parquet) and the RLE / bit-packed hybrid encoding
|
|
6
9
|
# used for repetition/definition levels, dictionary indices and RLE booleans.
|
|
@@ -9,6 +12,11 @@ module Herringbone
|
|
|
9
12
|
|
|
10
13
|
# Unpacks +count+ values of +width+ bits each, starting at byte +offset+ of +data+.
|
|
11
14
|
# Missing trailing bytes are treated as zeroes.
|
|
15
|
+
# @param data [String] binary input
|
|
16
|
+
# @param offset [Integer] byte offset of the first packed value
|
|
17
|
+
# @param count [Integer] number of values to unpack
|
|
18
|
+
# @param width [Integer] bits per value, 0..64
|
|
19
|
+
# @return [Array<Integer>] +count+ unsigned values
|
|
12
20
|
def unpack_bits(data, offset, count, width)
|
|
13
21
|
return Array.new(count, 0) if width.zero?
|
|
14
22
|
nbytes = (count * width + 7) / 8
|
|
@@ -22,7 +30,7 @@ module Herringbone
|
|
|
22
30
|
|
|
23
31
|
# Pad to a whole number of 32-bit words plus one spare word, so reads never go out of range
|
|
24
32
|
pad = (-chunk.bytesize % 4) + 4
|
|
25
|
-
words = (
|
|
33
|
+
words = (chunk + ("\0" * pad)).unpack("V*")
|
|
26
34
|
mask = (1 << width) - 1
|
|
27
35
|
out = Array.new(count)
|
|
28
36
|
bitpos = 0
|
|
@@ -41,6 +49,10 @@ module Herringbone
|
|
|
41
49
|
end
|
|
42
50
|
|
|
43
51
|
# Slow path for widths above 32 bits (used by DELTA_BINARY_PACKED with 64-bit values)
|
|
52
|
+
# @param chunk [String] packed bytes, starting at the first value
|
|
53
|
+
# @param count [Integer] number of values to unpack
|
|
54
|
+
# @param width [Integer] bits per value, 33..64
|
|
55
|
+
# @return [Array<Integer>] +count+ unsigned values
|
|
44
56
|
def unpack_wide(chunk, count, width)
|
|
45
57
|
mask = (1 << width) - 1
|
|
46
58
|
out = Array.new(count)
|
|
@@ -61,6 +73,9 @@ module Herringbone
|
|
|
61
73
|
end
|
|
62
74
|
|
|
63
75
|
# Packs +values+ with +width+ bits each; the value count is padded up to a multiple of 8.
|
|
76
|
+
# @param values [Array<Integer>] non-negative integers that fit in +width+ bits
|
|
77
|
+
# @param width [Integer] bits per value
|
|
78
|
+
# @return [String] packed bytes in ASCII-8BIT, +width+ bytes per group of 8 values
|
|
64
79
|
def pack_bits(values, width)
|
|
65
80
|
return "".b if width.zero? || values.empty?
|
|
66
81
|
n = values.size
|
|
@@ -93,12 +108,17 @@ module Herringbone
|
|
|
93
108
|
out
|
|
94
109
|
end
|
|
95
110
|
|
|
111
|
+
# Reads an unsigned LEB128 varint (hybrid run headers, DELTA_BINARY_PACKED headers).
|
|
112
|
+
# @param data [String] binary input
|
|
113
|
+
# @param pos [Integer] byte offset of the varint
|
|
114
|
+
# @return [Array(Integer, Integer)] the decoded value and the offset just past it
|
|
115
|
+
# @raise [FormatError] if the input ends inside the varint
|
|
96
116
|
def read_uleb(data, pos)
|
|
97
117
|
result = 0
|
|
98
118
|
shift = 0
|
|
99
119
|
while true
|
|
100
120
|
b = data.getbyte(pos)
|
|
101
|
-
raise
|
|
121
|
+
raise FormatError, "Truncated varint" unless b
|
|
102
122
|
pos += 1
|
|
103
123
|
result |= (b & 0x7F) << shift
|
|
104
124
|
return [result, pos] if b < 0x80
|
|
@@ -106,6 +126,10 @@ module Herringbone
|
|
|
106
126
|
end
|
|
107
127
|
end
|
|
108
128
|
|
|
129
|
+
# Appends +n+ as an unsigned LEB128 varint.
|
|
130
|
+
# @param out [String] binary output buffer, appended to
|
|
131
|
+
# @param n [Integer] non-negative integer to encode
|
|
132
|
+
# @return [String] +out+
|
|
109
133
|
def write_uleb(out, n)
|
|
110
134
|
while n >= 0x80
|
|
111
135
|
out << ((n & 0x7F) | 0x80)
|
|
@@ -116,11 +140,18 @@ module Herringbone
|
|
|
116
140
|
|
|
117
141
|
# Decodes the RLE/bit-packed hybrid from +data+ between +pos+ and +limit+,
|
|
118
142
|
# returning exactly +count+ values (missing values are an error).
|
|
143
|
+
# @param data [String] binary input
|
|
144
|
+
# @param pos [Integer] byte offset of the first run header
|
|
145
|
+
# @param limit [Integer] byte offset just past the encoded data
|
|
146
|
+
# @param width [Integer] bits per value
|
|
147
|
+
# @param count [Integer] number of values to decode
|
|
148
|
+
# @return [Array<Integer>] exactly +count+ values
|
|
149
|
+
# @raise [FormatError] if the runs end before +count+ values were produced
|
|
119
150
|
def decode_hybrid(data, pos, limit, width, count)
|
|
120
151
|
out = []
|
|
121
152
|
value_bytes = (width + 7) / 8
|
|
122
153
|
while out.size < count
|
|
123
|
-
raise
|
|
154
|
+
raise FormatError, "RLE data exhausted (#{out.size}/#{count} values)" if pos >= limit
|
|
124
155
|
header, pos = read_uleb(data, pos)
|
|
125
156
|
if header & 1 == 1
|
|
126
157
|
groups = header >> 1
|
|
@@ -145,6 +176,10 @@ module Herringbone
|
|
|
145
176
|
|
|
146
177
|
# Encodes +values+ with the RLE/bit-packed hybrid. Repeated runs of 8+ equal values
|
|
147
178
|
# become RLE runs, everything else goes into bit-packed groups of 8.
|
|
179
|
+
# Output has no length prefix; callers that need one (levels in data page v1) add it.
|
|
180
|
+
# @param values [Array<Integer>] non-negative integers that fit in +width+ bits
|
|
181
|
+
# @param width [Integer] bits per value
|
|
182
|
+
# @return [String] encoded runs in ASCII-8BIT
|
|
148
183
|
def encode_hybrid(values, width)
|
|
149
184
|
out = String.new(encoding: Encoding::BINARY)
|
|
150
185
|
n = values.size
|
|
@@ -181,6 +216,13 @@ module Herringbone
|
|
|
181
216
|
out
|
|
182
217
|
end
|
|
183
218
|
|
|
219
|
+
# Appends values[from...to] as one bit-packed run, zero-padded to whole groups of 8.
|
|
220
|
+
# @param out [String] binary output buffer, appended to
|
|
221
|
+
# @param values [Array<Integer>] all values being encoded
|
|
222
|
+
# @param from [Integer] index of the first literal value
|
|
223
|
+
# @param to [Integer] index just past the last literal value
|
|
224
|
+
# @param width [Integer] bits per value
|
|
225
|
+
# @return [String] +out+
|
|
184
226
|
def flush_literals(out, values, from, to, width)
|
|
185
227
|
count = to - from
|
|
186
228
|
groups = (count + 7) / 8
|
|
@@ -188,13 +230,23 @@ module Herringbone
|
|
|
188
230
|
out << pack_bits(values[from, count], width)
|
|
189
231
|
end
|
|
190
232
|
|
|
233
|
+
# Number of bits needed to store values up to +max_value+ (0 for 0).
|
|
234
|
+
# @param max_value [Integer, nil] largest value to encode; nil counts as 0
|
|
235
|
+
# @return [Integer] bit width
|
|
191
236
|
def bit_width(max_value)
|
|
192
237
|
max_value.to_i.bit_length
|
|
193
238
|
end
|
|
194
239
|
|
|
195
240
|
# Legacy BIT_PACKED level encoding (deprecated): MSB-first bit order, no header.
|
|
241
|
+
# @param data [String] binary input
|
|
242
|
+
# @param pos [Integer] byte offset of the packed levels
|
|
243
|
+
# @param width [Integer] bits per value
|
|
244
|
+
# @param count [Integer] number of values to decode
|
|
245
|
+
# @return [Array<Integer>] +count+ levels
|
|
246
|
+
# @raise [FormatError] if +data+ ends before +count+ levels
|
|
196
247
|
def decode_legacy_bit_packed(data, pos, width, count)
|
|
197
248
|
nbytes = (count * width + 7) / 8
|
|
249
|
+
raise FormatError, "Truncated BIT_PACKED levels" if pos + nbytes > data.bytesize
|
|
198
250
|
bits = data.byteslice(pos, nbytes).unpack1("B*")
|
|
199
251
|
Array.new(count) { |i| bits[i * width, width].to_i(2) }
|
|
200
252
|
end
|
data/lib/herringbone/format.rb
CHANGED
|
@@ -4,92 +4,163 @@ module Herringbone
|
|
|
4
4
|
# Parquet file metadata structures, mirroring parquet.thrift.
|
|
5
5
|
# Enums are plain i32 on the wire; constants below give them names.
|
|
6
6
|
module Format
|
|
7
|
+
# Physical storage types (+Type+ enum in parquet.thrift).
|
|
7
8
|
module Type
|
|
9
|
+
# Single bit, bit-packed in PLAIN encoding
|
|
8
10
|
BOOLEAN = 0
|
|
11
|
+
# 32-bit signed little-endian integer
|
|
9
12
|
INT32 = 1
|
|
13
|
+
# 64-bit signed little-endian integer
|
|
10
14
|
INT64 = 2
|
|
15
|
+
# 96-bit legacy timestamp (nanoseconds of day + Julian day), deprecated by the spec
|
|
11
16
|
INT96 = 3
|
|
17
|
+
# IEEE 754 single precision
|
|
12
18
|
FLOAT = 4
|
|
19
|
+
# IEEE 754 double precision
|
|
13
20
|
DOUBLE = 5
|
|
21
|
+
# Variable-length bytes, length-prefixed in PLAIN encoding
|
|
14
22
|
BYTE_ARRAY = 6
|
|
23
|
+
# Bytes of the fixed length given by +SchemaElement#type_length+
|
|
15
24
|
FIXED_LEN_BYTE_ARRAY = 7
|
|
25
|
+
# Constant name for each enum value
|
|
16
26
|
NAMES = constants.to_h { |c| [const_get(c), c] }.freeze
|
|
17
27
|
end
|
|
18
28
|
|
|
29
|
+
# Legacy type annotations (+ConvertedType+ enum), superseded by LogicalType but still
|
|
30
|
+
# written alongside it for older readers.
|
|
19
31
|
module ConvertedType
|
|
32
|
+
# UTF-8 encoded BYTE_ARRAY
|
|
20
33
|
UTF8 = 0
|
|
34
|
+
# Group annotated as a map
|
|
21
35
|
MAP = 1
|
|
36
|
+
# Repeated key/value group inside a MAP
|
|
22
37
|
MAP_KEY_VALUE = 2
|
|
38
|
+
# Group annotated as a list
|
|
23
39
|
LIST = 3
|
|
40
|
+
# BYTE_ARRAY holding an enum label
|
|
24
41
|
ENUM = 4
|
|
42
|
+
# Decimal with scale and precision from the SchemaElement
|
|
25
43
|
DECIMAL = 5
|
|
44
|
+
# INT32 days since the Unix epoch
|
|
26
45
|
DATE = 6
|
|
46
|
+
# INT32 milliseconds since midnight
|
|
27
47
|
TIME_MILLIS = 7
|
|
48
|
+
# INT64 microseconds since midnight
|
|
28
49
|
TIME_MICROS = 8
|
|
50
|
+
# INT64 milliseconds since the Unix epoch, UTC
|
|
29
51
|
TIMESTAMP_MILLIS = 9
|
|
52
|
+
# INT64 microseconds since the Unix epoch, UTC
|
|
30
53
|
TIMESTAMP_MICROS = 10
|
|
54
|
+
# Unsigned 8-bit integer stored in INT32
|
|
31
55
|
UINT_8 = 11
|
|
56
|
+
# Unsigned 16-bit integer stored in INT32
|
|
32
57
|
UINT_16 = 12
|
|
58
|
+
# Unsigned 32-bit integer stored in INT32
|
|
33
59
|
UINT_32 = 13
|
|
60
|
+
# Unsigned 64-bit integer stored in INT64
|
|
34
61
|
UINT_64 = 14
|
|
62
|
+
# Signed 8-bit integer stored in INT32
|
|
35
63
|
INT_8 = 15
|
|
64
|
+
# Signed 16-bit integer stored in INT32
|
|
36
65
|
INT_16 = 16
|
|
66
|
+
# Signed 32-bit integer stored in INT32
|
|
37
67
|
INT_32 = 17
|
|
68
|
+
# Signed 64-bit integer stored in INT64
|
|
38
69
|
INT_64 = 18
|
|
70
|
+
# UTF-8 JSON document in a BYTE_ARRAY
|
|
39
71
|
JSON = 19
|
|
72
|
+
# BSON document in a BYTE_ARRAY
|
|
40
73
|
BSON = 20
|
|
74
|
+
# 12-byte FIXED_LEN_BYTE_ARRAY of months, days and milliseconds
|
|
41
75
|
INTERVAL = 21
|
|
76
|
+
# Constant name for each enum value
|
|
42
77
|
NAMES = constants.to_h { |c| [const_get(c), c] }.freeze
|
|
43
78
|
end
|
|
44
79
|
|
|
80
|
+
# Field repetition (+FieldRepetitionType+ enum).
|
|
45
81
|
module Repetition
|
|
82
|
+
# Exactly one value
|
|
46
83
|
REQUIRED = 0
|
|
84
|
+
# Zero or one value
|
|
47
85
|
OPTIONAL = 1
|
|
86
|
+
# Zero or more values
|
|
48
87
|
REPEATED = 2
|
|
88
|
+
# Constant name for each enum value
|
|
49
89
|
NAMES = constants.to_h { |c| [const_get(c), c] }.freeze
|
|
50
90
|
end
|
|
51
91
|
|
|
92
|
+
# Value and level encodings (+Encoding+ enum). Value 1 (GROUP_VAR_INT) was never used.
|
|
52
93
|
module Encoding
|
|
94
|
+
# Values back to back in their natural binary form
|
|
53
95
|
PLAIN = 0
|
|
96
|
+
# Deprecated name for a dictionary-encoded page (and a PLAIN dictionary page) in format v1
|
|
54
97
|
PLAIN_DICTIONARY = 2
|
|
98
|
+
# RLE / bit-packing hybrid, used for levels and booleans
|
|
55
99
|
RLE = 3
|
|
100
|
+
# Deprecated bit-packed levels
|
|
56
101
|
BIT_PACKED = 4
|
|
102
|
+
# Delta encoding for INT32/INT64
|
|
57
103
|
DELTA_BINARY_PACKED = 5
|
|
104
|
+
# Delta-encoded lengths followed by the concatenated bytes
|
|
58
105
|
DELTA_LENGTH_BYTE_ARRAY = 6
|
|
106
|
+
# Incremental (prefix-shared) encoding for byte arrays
|
|
59
107
|
DELTA_BYTE_ARRAY = 7
|
|
108
|
+
# Dictionary indices in RLE / bit-packing hybrid form
|
|
60
109
|
RLE_DICTIONARY = 8
|
|
110
|
+
# Bytes of each value split into separate streams
|
|
61
111
|
BYTE_STREAM_SPLIT = 9
|
|
112
|
+
# Constant name for each enum value
|
|
62
113
|
NAMES = constants.to_h { |c| [const_get(c), c] }.freeze
|
|
63
114
|
end
|
|
64
115
|
|
|
116
|
+
# Page compression codecs (+CompressionCodec+ enum).
|
|
65
117
|
module Codec
|
|
118
|
+
# No compression
|
|
66
119
|
UNCOMPRESSED = 0
|
|
120
|
+
# Raw Snappy
|
|
67
121
|
SNAPPY = 1
|
|
122
|
+
# Gzip (deflate with gzip header)
|
|
68
123
|
GZIP = 2
|
|
124
|
+
# LZO
|
|
69
125
|
LZO = 3
|
|
126
|
+
# Brotli
|
|
70
127
|
BROTLI = 4
|
|
128
|
+
# Deprecated Hadoop-framed LZ4
|
|
71
129
|
LZ4 = 5
|
|
130
|
+
# Zstandard
|
|
72
131
|
ZSTD = 6
|
|
132
|
+
# LZ4 block format without framing
|
|
73
133
|
LZ4_RAW = 7
|
|
134
|
+
# Constant name for each enum value
|
|
74
135
|
NAMES = constants.to_h { |c| [const_get(c), c] }.freeze
|
|
75
136
|
end
|
|
76
137
|
|
|
138
|
+
# Page types (+PageType+ enum).
|
|
77
139
|
module PageType
|
|
140
|
+
# Data page, format v1
|
|
78
141
|
DATA_PAGE = 0
|
|
142
|
+
# Index page (never written in practice)
|
|
79
143
|
INDEX_PAGE = 1
|
|
144
|
+
# Dictionary page, precedes the data pages of a column chunk
|
|
80
145
|
DICTIONARY_PAGE = 2
|
|
146
|
+
# Data page, format v2 (levels stored uncompressed ahead of the values)
|
|
81
147
|
DATA_PAGE_V2 = 3
|
|
148
|
+
# Constant name for each enum value
|
|
82
149
|
NAMES = constants.to_h { |c| [const_get(c), c] }.freeze
|
|
83
150
|
end
|
|
84
151
|
|
|
152
|
+
# Short alias for the struct base class used by every definition below
|
|
85
153
|
S = Thrift::Struct
|
|
86
154
|
|
|
155
|
+
# Byte and level-histogram statistics for a page or column chunk.
|
|
87
156
|
class SizeStatistics < S
|
|
88
157
|
field 1, :unencoded_byte_array_data_bytes, :i64
|
|
89
158
|
field 2, :repetition_level_histogram, [:list, :i64]
|
|
90
159
|
field 3, :definition_level_histogram, [:list, :i64]
|
|
91
160
|
end
|
|
92
161
|
|
|
162
|
+
# Min/max and count statistics for a page or column chunk. +max+/+min+ are the deprecated
|
|
163
|
+
# signed-order fields; +max_value+/+min_value+ use the column's sort order.
|
|
93
164
|
class Statistics < S
|
|
94
165
|
field 1, :max, :binary
|
|
95
166
|
field 2, :min, :binary
|
|
@@ -102,40 +173,77 @@ module Herringbone
|
|
|
102
173
|
end
|
|
103
174
|
|
|
104
175
|
# Empty marker structs used inside unions
|
|
176
|
+
|
|
177
|
+
# STRING logical type.
|
|
105
178
|
class StringType < S; end
|
|
179
|
+
|
|
180
|
+
# UUID logical type (16-byte FIXED_LEN_BYTE_ARRAY).
|
|
106
181
|
class UUIDType < S; end
|
|
182
|
+
|
|
183
|
+
# MAP logical type.
|
|
107
184
|
class MapType < S; end
|
|
185
|
+
|
|
186
|
+
# LIST logical type.
|
|
108
187
|
class ListType < S; end
|
|
188
|
+
|
|
189
|
+
# ENUM logical type.
|
|
109
190
|
class EnumType < S; end
|
|
191
|
+
|
|
192
|
+
# DATE logical type.
|
|
110
193
|
class DateType < S; end
|
|
194
|
+
|
|
195
|
+
# FLOAT16 logical type (2-byte FIXED_LEN_BYTE_ARRAY, little-endian half precision).
|
|
111
196
|
class Float16Type < S; end
|
|
197
|
+
|
|
198
|
+
# UNKNOWN logical type: the column is always null.
|
|
112
199
|
class NullType < S; end
|
|
200
|
+
|
|
201
|
+
# JSON logical type.
|
|
113
202
|
class JsonType < S; end
|
|
203
|
+
|
|
204
|
+
# BSON logical type.
|
|
114
205
|
class BsonType < S; end
|
|
206
|
+
|
|
207
|
+
# Millisecond member of the TimeUnit union.
|
|
115
208
|
class MilliSeconds < S; end
|
|
209
|
+
|
|
210
|
+
# Microsecond member of the TimeUnit union.
|
|
116
211
|
class MicroSeconds < S; end
|
|
212
|
+
|
|
213
|
+
# Nanosecond member of the TimeUnit union.
|
|
117
214
|
class NanoSeconds < S; end
|
|
215
|
+
|
|
216
|
+
# ColumnOrder member: values are ordered according to their (logical) type.
|
|
118
217
|
class TypeDefinedOrder < S; end
|
|
218
|
+
|
|
219
|
+
# Header of an index page; has no fields in the spec.
|
|
119
220
|
class IndexPageHeader < S; end
|
|
120
221
|
|
|
222
|
+
# VARIANT logical type.
|
|
121
223
|
class VariantType < S
|
|
122
224
|
field 1, :specification_version, :byte
|
|
123
225
|
end
|
|
124
226
|
|
|
227
|
+
# DECIMAL logical type: unscaled integer value divided by 10 to the power of +scale+.
|
|
125
228
|
class DecimalType < S
|
|
126
229
|
field 1, :scale, :i32
|
|
127
230
|
field 2, :precision, :i32
|
|
128
231
|
end
|
|
129
232
|
|
|
233
|
+
# Union of time units for TIME and TIMESTAMP; exactly one member is set.
|
|
130
234
|
class TimeUnit < S
|
|
131
235
|
field 1, :millis, MilliSeconds
|
|
132
236
|
field 2, :micros, MicroSeconds
|
|
133
237
|
field 3, :nanos, NanoSeconds
|
|
134
238
|
|
|
239
|
+
# @return [TimeUnit] a unit with the +millis+ member set
|
|
135
240
|
def self.millis = new(millis: MilliSeconds.new)
|
|
241
|
+
# @return [TimeUnit] a unit with the +micros+ member set
|
|
136
242
|
def self.micros = new(micros: MicroSeconds.new)
|
|
243
|
+
# @return [TimeUnit] a unit with the +nanos+ member set
|
|
137
244
|
def self.nanos = new(nanos: NanoSeconds.new)
|
|
138
245
|
|
|
246
|
+
# @return [Symbol, nil] +:millis+, +:micros+ or +:nanos+, or nil when no member is set
|
|
139
247
|
def to_sym
|
|
140
248
|
if millis then :millis
|
|
141
249
|
elsif micros then :micros
|
|
@@ -144,21 +252,26 @@ module Herringbone
|
|
|
144
252
|
end
|
|
145
253
|
end
|
|
146
254
|
|
|
255
|
+
# TIMESTAMP logical type.
|
|
147
256
|
class TimestampType < S
|
|
148
257
|
field 1, :is_adjusted_to_utc, :bool
|
|
149
258
|
field 2, :unit, TimeUnit
|
|
150
259
|
end
|
|
151
260
|
|
|
261
|
+
# TIME logical type.
|
|
152
262
|
class TimeType < S
|
|
153
263
|
field 1, :is_adjusted_to_utc, :bool
|
|
154
264
|
field 2, :unit, TimeUnit
|
|
155
265
|
end
|
|
156
266
|
|
|
267
|
+
# INTEGER logical type (bit width 8, 16, 32 or 64; signed or unsigned).
|
|
157
268
|
class IntType < S
|
|
158
269
|
field 1, :bit_width, :byte
|
|
159
270
|
field 2, :is_signed, :bool
|
|
160
271
|
end
|
|
161
272
|
|
|
273
|
+
# Union of logical type annotations; exactly one member is set. Field id 9 is reserved
|
|
274
|
+
# (for INTERVAL) in parquet.thrift.
|
|
162
275
|
class LogicalType < S
|
|
163
276
|
field 1, :string, StringType
|
|
164
277
|
field 2, :map, MapType
|
|
@@ -177,6 +290,8 @@ module Herringbone
|
|
|
177
290
|
field 16, :variant, VariantType
|
|
178
291
|
|
|
179
292
|
# Returns [kind_symbol, payload] for whichever union member is set
|
|
293
|
+
# @return [Array(Symbol, Thrift::Struct), nil] member name and its struct, or nil when
|
|
294
|
+
# no member is set
|
|
180
295
|
def kind
|
|
181
296
|
self.class.fields.each do |f|
|
|
182
297
|
v = instance_variable_get(f.ivar)
|
|
@@ -186,6 +301,8 @@ module Herringbone
|
|
|
186
301
|
end
|
|
187
302
|
end
|
|
188
303
|
|
|
304
|
+
# One node of the schema, flattened depth-first into +FileMetaData#schema+. Groups carry
|
|
305
|
+
# +num_children+; leaves carry the physical +type+.
|
|
189
306
|
class SchemaElement < S
|
|
190
307
|
field 1, :type, :i32
|
|
191
308
|
field 2, :type_length, :i32
|
|
@@ -199,6 +316,7 @@ module Herringbone
|
|
|
199
316
|
field 10, :logical_type, LogicalType
|
|
200
317
|
end
|
|
201
318
|
|
|
319
|
+
# Header of a v1 data page: levels and values share one compressed block.
|
|
202
320
|
class DataPageHeader < S
|
|
203
321
|
field 1, :num_values, :i32
|
|
204
322
|
field 2, :encoding, :i32
|
|
@@ -207,12 +325,15 @@ module Herringbone
|
|
|
207
325
|
field 5, :statistics, Statistics
|
|
208
326
|
end
|
|
209
327
|
|
|
328
|
+
# Header of a dictionary page.
|
|
210
329
|
class DictionaryPageHeader < S
|
|
211
330
|
field 1, :num_values, :i32
|
|
212
331
|
field 2, :encoding, :i32
|
|
213
332
|
field 3, :is_sorted, :bool
|
|
214
333
|
end
|
|
215
334
|
|
|
335
|
+
# Header of a v2 data page: levels are stored uncompressed before the (optionally
|
|
336
|
+
# compressed) values, with their byte lengths given here.
|
|
216
337
|
class DataPageHeaderV2 < S
|
|
217
338
|
field 1, :num_values, :i32
|
|
218
339
|
field 2, :num_nulls, :i32
|
|
@@ -224,6 +345,7 @@ module Herringbone
|
|
|
224
345
|
field 8, :statistics, Statistics
|
|
225
346
|
end
|
|
226
347
|
|
|
348
|
+
# Header preceding every page; +type+ (a PageType) says which sub-header is set.
|
|
227
349
|
class PageHeader < S
|
|
228
350
|
field 1, :type, :i32
|
|
229
351
|
field 2, :uncompressed_page_size, :i32
|
|
@@ -235,23 +357,27 @@ module Herringbone
|
|
|
235
357
|
field 8, :data_page_header_v2, DataPageHeaderV2
|
|
236
358
|
end
|
|
237
359
|
|
|
360
|
+
# Application-defined key/value metadata entry.
|
|
238
361
|
class KeyValue < S
|
|
239
362
|
field 1, :key, :string
|
|
240
363
|
field 2, :value, :string
|
|
241
364
|
end
|
|
242
365
|
|
|
366
|
+
# Sort order of one column within a row group.
|
|
243
367
|
class SortingColumn < S
|
|
244
368
|
field 1, :column_idx, :i32
|
|
245
369
|
field 2, :descending, :bool
|
|
246
370
|
field 3, :nulls_first, :bool
|
|
247
371
|
end
|
|
248
372
|
|
|
373
|
+
# Number of pages of a given type and encoding in a column chunk.
|
|
249
374
|
class PageEncodingStats < S
|
|
250
375
|
field 1, :page_type, :i32
|
|
251
376
|
field 2, :encoding, :i32
|
|
252
377
|
field 3, :count, :i32
|
|
253
378
|
end
|
|
254
379
|
|
|
380
|
+
# Metadata of a column chunk: where its pages are, how they are encoded and compressed.
|
|
255
381
|
class ColumnMetaData < S
|
|
256
382
|
field 1, :type, :i32
|
|
257
383
|
field 2, :encodings, [:list, :i32]
|
|
@@ -271,6 +397,8 @@ module Herringbone
|
|
|
271
397
|
field 16, :size_statistics, SizeStatistics
|
|
272
398
|
end
|
|
273
399
|
|
|
400
|
+
# One column of a row group, plus the locations of its page index. Field id 8
|
|
401
|
+
# (crypto_metadata) is not supported.
|
|
274
402
|
class ColumnChunk < S
|
|
275
403
|
field 1, :file_path, :string
|
|
276
404
|
field 2, :file_offset, :i64
|
|
@@ -282,6 +410,7 @@ module Herringbone
|
|
|
282
410
|
field 9, :encrypted_column_metadata, :binary
|
|
283
411
|
end
|
|
284
412
|
|
|
413
|
+
# A horizontal slice of the file, holding one ColumnChunk per leaf column.
|
|
285
414
|
class RowGroup < S
|
|
286
415
|
field 1, :columns, [:list, ColumnChunk]
|
|
287
416
|
field 2, :total_byte_size, :i64
|
|
@@ -292,10 +421,50 @@ module Herringbone
|
|
|
292
421
|
field 7, :ordinal, :i16
|
|
293
422
|
end
|
|
294
423
|
|
|
424
|
+
# Ordering of the per-page min/max values in a ColumnIndex (+BoundaryOrder+ enum).
|
|
425
|
+
module BoundaryOrder
|
|
426
|
+
# No particular order
|
|
427
|
+
UNORDERED = 0
|
|
428
|
+
# Min and max values both non-decreasing from page to page
|
|
429
|
+
ASCENDING = 1
|
|
430
|
+
# Min and max values both non-increasing from page to page
|
|
431
|
+
DESCENDING = 2
|
|
432
|
+
# Constant name for each enum value
|
|
433
|
+
NAMES = constants.to_h { |c| [const_get(c), c] }.freeze
|
|
434
|
+
end
|
|
435
|
+
|
|
436
|
+
# Page index structures (stored between the row groups and the footer)
|
|
437
|
+
|
|
438
|
+
# Location of one data page within the file.
|
|
439
|
+
class PageLocation < S
|
|
440
|
+
field 1, :offset, :i64
|
|
441
|
+
field 2, :compressed_page_size, :i32
|
|
442
|
+
field 3, :first_row_index, :i64
|
|
443
|
+
end
|
|
444
|
+
|
|
445
|
+
# Offset index of a column chunk: the location of each of its data pages.
|
|
446
|
+
class OffsetIndex < S
|
|
447
|
+
field 1, :page_locations, [:list, PageLocation]
|
|
448
|
+
field 2, :unencoded_byte_array_data_bytes, [:list, :i64]
|
|
449
|
+
end
|
|
450
|
+
|
|
451
|
+
# Column index of a column chunk: per-page min/max values and null information.
|
|
452
|
+
class ColumnIndex < S
|
|
453
|
+
field 1, :null_pages, [:list, :bool]
|
|
454
|
+
field 2, :min_values, [:list, :binary]
|
|
455
|
+
field 3, :max_values, [:list, :binary]
|
|
456
|
+
field 4, :boundary_order, :i32
|
|
457
|
+
field 5, :null_counts, [:list, :i64]
|
|
458
|
+
field 6, :repetition_level_histograms, [:list, :i64]
|
|
459
|
+
field 7, :definition_level_histograms, [:list, :i64]
|
|
460
|
+
end
|
|
461
|
+
|
|
462
|
+
# Union describing how min/max statistics of a column are ordered.
|
|
295
463
|
class ColumnOrder < S
|
|
296
464
|
field 1, :type_order, TypeDefinedOrder
|
|
297
465
|
end
|
|
298
466
|
|
|
467
|
+
# The file footer: schema, row groups and file-level metadata.
|
|
299
468
|
class FileMetaData < S
|
|
300
469
|
field 1, :version, :i32
|
|
301
470
|
field 2, :schema, [:list, SchemaElement]
|
|
@@ -305,5 +474,39 @@ module Herringbone
|
|
|
305
474
|
field 6, :created_by, :string
|
|
306
475
|
field 7, :column_orders, [:list, ColumnOrder]
|
|
307
476
|
end
|
|
477
|
+
|
|
478
|
+
# Bloom filters (BloomFilter.md). Each union has a single member, an empty struct.
|
|
479
|
+
|
|
480
|
+
# Split block bloom filter algorithm.
|
|
481
|
+
class SplitBlockAlgorithm < S; end
|
|
482
|
+
|
|
483
|
+
# XXH64 hash with seed 0.
|
|
484
|
+
class XxHash < S; end
|
|
485
|
+
|
|
486
|
+
# Bloom filter bitset stored uncompressed.
|
|
487
|
+
class BloomFilterUncompressed < S; end
|
|
488
|
+
|
|
489
|
+
# Union of bloom filter algorithms.
|
|
490
|
+
class BloomFilterAlgorithm < S
|
|
491
|
+
field 1, :block, SplitBlockAlgorithm
|
|
492
|
+
end
|
|
493
|
+
|
|
494
|
+
# Union of hash functions used to feed the bloom filter.
|
|
495
|
+
class BloomFilterHash < S
|
|
496
|
+
field 1, :xxhash, XxHash
|
|
497
|
+
end
|
|
498
|
+
|
|
499
|
+
# Union of bloom filter compressions.
|
|
500
|
+
class BloomFilterCompression < S
|
|
501
|
+
field 1, :uncompressed, BloomFilterUncompressed
|
|
502
|
+
end
|
|
503
|
+
|
|
504
|
+
# Header preceding the bitset of a column chunk's bloom filter.
|
|
505
|
+
class BloomFilterHeader < S
|
|
506
|
+
field 1, :num_bytes, :i32
|
|
507
|
+
field 2, :algorithm, BloomFilterAlgorithm
|
|
508
|
+
field 3, :hash_function, BloomFilterHash # "hash" in parquet.thrift; renamed to keep Object#hash
|
|
509
|
+
field 4, :compression, BloomFilterCompression
|
|
510
|
+
end
|
|
308
511
|
end
|
|
309
512
|
end
|