herringbone 0.1.0 → 0.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -7,6 +7,7 @@ module Herringbone
7
7
  module Plain
8
8
  module_function
9
9
 
10
+ # Fixed-width numeric types: pack/unpack directive and byte width (all little-endian).
10
11
  FORMATS = {
11
12
  Format::Type::INT32 => ["l<", 4],
12
13
  Format::Type::INT64 => ["q<", 8],
@@ -16,43 +17,57 @@ module Herringbone
16
17
 
17
18
  # Decodes +count+ values of +type+ from +data+ starting at +pos+.
18
19
  # Returns [values, new_pos].
20
+ # INT96 values come back as [nanoseconds_of_day, julian_day] pairs.
21
+ # @param data [String] binary page data
22
+ # @param pos [Integer] byte offset of the first value
23
+ # @param count [Integer] number of values to decode
24
+ # @param type [Integer] physical type, a Format::Type constant
25
+ # @param type_length [Integer, nil] byte width, required for FIXED_LEN_BYTE_ARRAY
26
+ # @return [Array(Array, Integer)] the decoded values and the offset just past them
27
+ # @raise [FormatError] if the data is truncated or the type is unknown
19
28
  def decode(data, pos, count, type, type_length = nil)
20
29
  case type
21
30
  when Format::Type::BOOLEAN
22
31
  nbytes = (count + 7) / 8
23
32
  bits = data.byteslice(pos, nbytes).unpack1("b*")
24
- raise DecodeError, "Truncated BOOLEAN data" if bits.bytesize < count
33
+ raise FormatError, "Truncated BOOLEAN data" if bits.bytesize < count
25
34
  [Array.new(count) { |i| bits.getbyte(i) == 49 }, pos + nbytes]
26
35
  when Format::Type::INT32, Format::Type::INT64, Format::Type::FLOAT, Format::Type::DOUBLE
27
36
  fmt, width = FORMATS[type]
28
37
  nbytes = count * width
29
- raise DecodeError, "Truncated PLAIN data" if pos + nbytes > data.bytesize
38
+ raise FormatError, "Truncated PLAIN data" if pos + nbytes > data.bytesize
30
39
  [data.byteslice(pos, nbytes).unpack("#{fmt}#{count}"), pos + nbytes]
31
40
  when Format::Type::INT96
32
41
  nbytes = count * 12
33
- raise DecodeError, "Truncated INT96 data" if pos + nbytes > data.bytesize
42
+ raise FormatError, "Truncated INT96 data" if pos + nbytes > data.bytesize
34
43
  vals = data.byteslice(pos, nbytes).unpack("Q<L<" * count).each_slice(2).map { |nanos, day| [nanos, day] }
35
44
  [vals, pos + nbytes]
36
45
  when Format::Type::BYTE_ARRAY
37
46
  decode_byte_arrays(data, pos, count)
38
47
  when Format::Type::FIXED_LEN_BYTE_ARRAY
39
48
  nbytes = count * type_length
40
- raise DecodeError, "Truncated FIXED_LEN_BYTE_ARRAY data" if pos + nbytes > data.bytesize
49
+ raise FormatError, "Truncated FIXED_LEN_BYTE_ARRAY data" if pos + nbytes > data.bytesize
41
50
  [Array.new(count) { |i| data.byteslice(pos + i * type_length, type_length) }, pos + nbytes]
42
51
  else
43
- raise DecodeError, "Unknown physical type #{type}"
52
+ raise FormatError, "Unknown physical type #{type}"
44
53
  end
45
54
  end
46
55
 
56
+ # Decodes PLAIN BYTE_ARRAY values: each is a 4-byte little-endian length followed by the bytes.
57
+ # @param data [String] binary page data
58
+ # @param pos [Integer] byte offset of the first length prefix
59
+ # @param count [Integer] number of values to decode
60
+ # @return [Array(Array<String>, Integer)] binary slices of +data+ and the offset just past them
61
+ # @raise [FormatError] if a length prefix or value runs past the end of +data+
47
62
  def decode_byte_arrays(data, pos, count)
48
63
  out = Array.new(count)
49
64
  size = data.bytesize
50
65
  i = 0
51
66
  while i < count
52
- raise DecodeError, "Truncated BYTE_ARRAY data" if pos + 4 > size
67
+ raise FormatError, "Truncated BYTE_ARRAY data" if pos + 4 > size
53
68
  len = data.getbyte(pos) | (data.getbyte(pos + 1) << 8) | (data.getbyte(pos + 2) << 16) | (data.getbyte(pos + 3) << 24)
54
69
  pos += 4
55
- raise DecodeError, "BYTE_ARRAY value overruns page" if pos + len > size
70
+ raise FormatError, "BYTE_ARRAY value overruns page" if pos + len > size
56
71
  out[i] = data.byteslice(pos, len)
57
72
  pos += len
58
73
  i += 1
@@ -60,6 +75,13 @@ module Herringbone
60
75
  [out, pos]
61
76
  end
62
77
 
78
+ # Encodes values of a physical type with PLAIN. Inverse of decode; INT96 values are
79
+ # [nanoseconds_of_day, julian_day] pairs and booleans are packed LSB-first, 8 per byte.
80
+ # @param values [Array] values in their physical Ruby form, without nulls
81
+ # @param type [Integer] physical type, a Format::Type constant
82
+ # @param type_length [Integer, nil] byte width, required for FIXED_LEN_BYTE_ARRAY
83
+ # @return [String] encoded bytes in ASCII-8BIT
84
+ # @raise [EncodeError] if a FIXED_LEN_BYTE_ARRAY value has the wrong size or the type is unknown
63
85
  def encode(values, type, type_length = nil)
64
86
  case type
65
87
  when Format::Type::BOOLEAN
@@ -1,6 +1,9 @@
1
1
  # frozen_string_literal: true
2
2
 
3
3
  module Herringbone
4
+ # Decoders and encoders for the Parquet value and level encodings: PLAIN, the RLE / bit-packed
5
+ # hybrid, the DELTA_* family and BYTE_STREAM_SPLIT. They work on binary Strings and byte offsets,
6
+ # and know nothing about pages or columns.
4
7
  module Encodings
5
8
  # Bit packing (LSB-first, as used by Parquet) and the RLE / bit-packed hybrid encoding
6
9
  # used for repetition/definition levels, dictionary indices and RLE booleans.
@@ -9,6 +12,11 @@ module Herringbone
9
12
 
10
13
  # Unpacks +count+ values of +width+ bits each, starting at byte +offset+ of +data+.
11
14
  # Missing trailing bytes are treated as zeroes.
15
+ # @param data [String] binary input
16
+ # @param offset [Integer] byte offset of the first packed value
17
+ # @param count [Integer] number of values to unpack
18
+ # @param width [Integer] bits per value, 0..64
19
+ # @return [Array<Integer>] +count+ unsigned values
12
20
  def unpack_bits(data, offset, count, width)
13
21
  return Array.new(count, 0) if width.zero?
14
22
  nbytes = (count * width + 7) / 8
@@ -22,7 +30,7 @@ module Herringbone
22
30
 
23
31
  # Pad to a whole number of 32-bit words plus one spare word, so reads never go out of range
24
32
  pad = (-chunk.bytesize % 4) + 4
25
- words = (pad.zero? ? chunk : chunk + ("\0" * pad)).unpack("V*")
33
+ words = (chunk + ("\0" * pad)).unpack("V*")
26
34
  mask = (1 << width) - 1
27
35
  out = Array.new(count)
28
36
  bitpos = 0
@@ -41,6 +49,10 @@ module Herringbone
41
49
  end
42
50
 
43
51
  # Slow path for widths above 32 bits (used by DELTA_BINARY_PACKED with 64-bit values)
52
+ # @param chunk [String] packed bytes, starting at the first value
53
+ # @param count [Integer] number of values to unpack
54
+ # @param width [Integer] bits per value, 33..64
55
+ # @return [Array<Integer>] +count+ unsigned values
44
56
  def unpack_wide(chunk, count, width)
45
57
  mask = (1 << width) - 1
46
58
  out = Array.new(count)
@@ -61,6 +73,9 @@ module Herringbone
61
73
  end
62
74
 
63
75
  # Packs +values+ with +width+ bits each; the value count is padded up to a multiple of 8.
76
+ # @param values [Array<Integer>] non-negative integers that fit in +width+ bits
77
+ # @param width [Integer] bits per value
78
+ # @return [String] packed bytes in ASCII-8BIT, +width+ bytes per group of 8 values
64
79
  def pack_bits(values, width)
65
80
  return "".b if width.zero? || values.empty?
66
81
  n = values.size
@@ -93,12 +108,17 @@ module Herringbone
93
108
  out
94
109
  end
95
110
 
111
+ # Reads an unsigned LEB128 varint (hybrid run headers, DELTA_BINARY_PACKED headers).
112
+ # @param data [String] binary input
113
+ # @param pos [Integer] byte offset of the varint
114
+ # @return [Array(Integer, Integer)] the decoded value and the offset just past it
115
+ # @raise [FormatError] if the input ends inside the varint
96
116
  def read_uleb(data, pos)
97
117
  result = 0
98
118
  shift = 0
99
119
  while true
100
120
  b = data.getbyte(pos)
101
- raise DecodeError, "Truncated varint" unless b
121
+ raise FormatError, "Truncated varint" unless b
102
122
  pos += 1
103
123
  result |= (b & 0x7F) << shift
104
124
  return [result, pos] if b < 0x80
@@ -106,6 +126,10 @@ module Herringbone
106
126
  end
107
127
  end
108
128
 
129
+ # Appends +n+ as an unsigned LEB128 varint.
130
+ # @param out [String] binary output buffer, appended to
131
+ # @param n [Integer] non-negative integer to encode
132
+ # @return [String] +out+
109
133
  def write_uleb(out, n)
110
134
  while n >= 0x80
111
135
  out << ((n & 0x7F) | 0x80)
@@ -116,11 +140,18 @@ module Herringbone
116
140
 
117
141
  # Decodes the RLE/bit-packed hybrid from +data+ between +pos+ and +limit+,
118
142
  # returning exactly +count+ values (missing values are an error).
143
+ # @param data [String] binary input
144
+ # @param pos [Integer] byte offset of the first run header
145
+ # @param limit [Integer] byte offset just past the encoded data
146
+ # @param width [Integer] bits per value
147
+ # @param count [Integer] number of values to decode
148
+ # @return [Array<Integer>] exactly +count+ values
149
+ # @raise [FormatError] if the runs end before +count+ values were produced
119
150
  def decode_hybrid(data, pos, limit, width, count)
120
151
  out = []
121
152
  value_bytes = (width + 7) / 8
122
153
  while out.size < count
123
- raise DecodeError, "RLE data exhausted (#{out.size}/#{count} values)" if pos >= limit
154
+ raise FormatError, "RLE data exhausted (#{out.size}/#{count} values)" if pos >= limit
124
155
  header, pos = read_uleb(data, pos)
125
156
  if header & 1 == 1
126
157
  groups = header >> 1
@@ -145,6 +176,10 @@ module Herringbone
145
176
 
146
177
  # Encodes +values+ with the RLE/bit-packed hybrid. Repeated runs of 8+ equal values
147
178
  # become RLE runs, everything else goes into bit-packed groups of 8.
179
+ # Output has no length prefix; callers that need one (levels in data page v1) add it.
180
+ # @param values [Array<Integer>] non-negative integers that fit in +width+ bits
181
+ # @param width [Integer] bits per value
182
+ # @return [String] encoded runs in ASCII-8BIT
148
183
  def encode_hybrid(values, width)
149
184
  out = String.new(encoding: Encoding::BINARY)
150
185
  n = values.size
@@ -181,6 +216,13 @@ module Herringbone
181
216
  out
182
217
  end
183
218
 
219
+ # Appends values[from...to] as one bit-packed run, zero-padded to whole groups of 8.
220
+ # @param out [String] binary output buffer, appended to
221
+ # @param values [Array<Integer>] all values being encoded
222
+ # @param from [Integer] index of the first literal value
223
+ # @param to [Integer] index just past the last literal value
224
+ # @param width [Integer] bits per value
225
+ # @return [String] +out+
184
226
  def flush_literals(out, values, from, to, width)
185
227
  count = to - from
186
228
  groups = (count + 7) / 8
@@ -188,13 +230,23 @@ module Herringbone
188
230
  out << pack_bits(values[from, count], width)
189
231
  end
190
232
 
233
+ # Number of bits needed to store values up to +max_value+ (0 for 0).
234
+ # @param max_value [Integer, nil] largest value to encode; nil counts as 0
235
+ # @return [Integer] bit width
191
236
  def bit_width(max_value)
192
237
  max_value.to_i.bit_length
193
238
  end
194
239
 
195
240
  # Legacy BIT_PACKED level encoding (deprecated): MSB-first bit order, no header.
241
+ # @param data [String] binary input
242
+ # @param pos [Integer] byte offset of the packed levels
243
+ # @param width [Integer] bits per value
244
+ # @param count [Integer] number of values to decode
245
+ # @return [Array<Integer>] +count+ levels
246
+ # @raise [FormatError] if +data+ ends before +count+ levels
196
247
  def decode_legacy_bit_packed(data, pos, width, count)
197
248
  nbytes = (count * width + 7) / 8
249
+ raise FormatError, "Truncated BIT_PACKED levels" if pos + nbytes > data.bytesize
198
250
  bits = data.byteslice(pos, nbytes).unpack1("B*")
199
251
  Array.new(count) { |i| bits[i * width, width].to_i(2) }
200
252
  end
@@ -4,92 +4,163 @@ module Herringbone
4
4
  # Parquet file metadata structures, mirroring parquet.thrift.
5
5
  # Enums are plain i32 on the wire; constants below give them names.
6
6
  module Format
7
+ # Physical storage types (+Type+ enum in parquet.thrift).
7
8
  module Type
9
+ # Single bit, bit-packed in PLAIN encoding
8
10
  BOOLEAN = 0
11
+ # 32-bit signed little-endian integer
9
12
  INT32 = 1
13
+ # 64-bit signed little-endian integer
10
14
  INT64 = 2
15
+ # 96-bit legacy timestamp (nanoseconds of day + Julian day), deprecated by the spec
11
16
  INT96 = 3
17
+ # IEEE 754 single precision
12
18
  FLOAT = 4
19
+ # IEEE 754 double precision
13
20
  DOUBLE = 5
21
+ # Variable-length bytes, length-prefixed in PLAIN encoding
14
22
  BYTE_ARRAY = 6
23
+ # Bytes of the fixed length given by +SchemaElement#type_length+
15
24
  FIXED_LEN_BYTE_ARRAY = 7
25
+ # Constant name for each enum value
16
26
  NAMES = constants.to_h { |c| [const_get(c), c] }.freeze
17
27
  end
18
28
 
29
+ # Legacy type annotations (+ConvertedType+ enum), superseded by LogicalType but still
30
+ # written alongside it for older readers.
19
31
  module ConvertedType
32
+ # UTF-8 encoded BYTE_ARRAY
20
33
  UTF8 = 0
34
+ # Group annotated as a map
21
35
  MAP = 1
36
+ # Repeated key/value group inside a MAP
22
37
  MAP_KEY_VALUE = 2
38
+ # Group annotated as a list
23
39
  LIST = 3
40
+ # BYTE_ARRAY holding an enum label
24
41
  ENUM = 4
42
+ # Decimal with scale and precision from the SchemaElement
25
43
  DECIMAL = 5
44
+ # INT32 days since the Unix epoch
26
45
  DATE = 6
46
+ # INT32 milliseconds since midnight
27
47
  TIME_MILLIS = 7
48
+ # INT64 microseconds since midnight
28
49
  TIME_MICROS = 8
50
+ # INT64 milliseconds since the Unix epoch, UTC
29
51
  TIMESTAMP_MILLIS = 9
52
+ # INT64 microseconds since the Unix epoch, UTC
30
53
  TIMESTAMP_MICROS = 10
54
+ # Unsigned 8-bit integer stored in INT32
31
55
  UINT_8 = 11
56
+ # Unsigned 16-bit integer stored in INT32
32
57
  UINT_16 = 12
58
+ # Unsigned 32-bit integer stored in INT32
33
59
  UINT_32 = 13
60
+ # Unsigned 64-bit integer stored in INT64
34
61
  UINT_64 = 14
62
+ # Signed 8-bit integer stored in INT32
35
63
  INT_8 = 15
64
+ # Signed 16-bit integer stored in INT32
36
65
  INT_16 = 16
66
+ # Signed 32-bit integer stored in INT32
37
67
  INT_32 = 17
68
+ # Signed 64-bit integer stored in INT64
38
69
  INT_64 = 18
70
+ # UTF-8 JSON document in a BYTE_ARRAY
39
71
  JSON = 19
72
+ # BSON document in a BYTE_ARRAY
40
73
  BSON = 20
74
+ # 12-byte FIXED_LEN_BYTE_ARRAY of months, days and milliseconds
41
75
  INTERVAL = 21
76
+ # Constant name for each enum value
42
77
  NAMES = constants.to_h { |c| [const_get(c), c] }.freeze
43
78
  end
44
79
 
80
+ # Field repetition (+FieldRepetitionType+ enum).
45
81
  module Repetition
82
+ # Exactly one value
46
83
  REQUIRED = 0
84
+ # Zero or one value
47
85
  OPTIONAL = 1
86
+ # Zero or more values
48
87
  REPEATED = 2
88
+ # Constant name for each enum value
49
89
  NAMES = constants.to_h { |c| [const_get(c), c] }.freeze
50
90
  end
51
91
 
92
+ # Value and level encodings (+Encoding+ enum). Value 1 (GROUP_VAR_INT) was never used.
52
93
  module Encoding
94
+ # Values back to back in their natural binary form
53
95
  PLAIN = 0
96
+ # Deprecated name for a dictionary-encoded page (and a PLAIN dictionary page) in format v1
54
97
  PLAIN_DICTIONARY = 2
98
+ # RLE / bit-packing hybrid, used for levels and booleans
55
99
  RLE = 3
100
+ # Deprecated bit-packed levels
56
101
  BIT_PACKED = 4
102
+ # Delta encoding for INT32/INT64
57
103
  DELTA_BINARY_PACKED = 5
104
+ # Delta-encoded lengths followed by the concatenated bytes
58
105
  DELTA_LENGTH_BYTE_ARRAY = 6
106
+ # Incremental (prefix-shared) encoding for byte arrays
59
107
  DELTA_BYTE_ARRAY = 7
108
+ # Dictionary indices in RLE / bit-packing hybrid form
60
109
  RLE_DICTIONARY = 8
110
+ # Bytes of each value split into separate streams
61
111
  BYTE_STREAM_SPLIT = 9
112
+ # Constant name for each enum value
62
113
  NAMES = constants.to_h { |c| [const_get(c), c] }.freeze
63
114
  end
64
115
 
116
+ # Page compression codecs (+CompressionCodec+ enum).
65
117
  module Codec
118
+ # No compression
66
119
  UNCOMPRESSED = 0
120
+ # Raw Snappy
67
121
  SNAPPY = 1
122
+ # Gzip (deflate with gzip header)
68
123
  GZIP = 2
124
+ # LZO
69
125
  LZO = 3
126
+ # Brotli
70
127
  BROTLI = 4
128
+ # Deprecated Hadoop-framed LZ4
71
129
  LZ4 = 5
130
+ # Zstandard
72
131
  ZSTD = 6
132
+ # LZ4 block format without framing
73
133
  LZ4_RAW = 7
134
+ # Constant name for each enum value
74
135
  NAMES = constants.to_h { |c| [const_get(c), c] }.freeze
75
136
  end
76
137
 
138
+ # Page types (+PageType+ enum).
77
139
  module PageType
140
+ # Data page, format v1
78
141
  DATA_PAGE = 0
142
+ # Index page (never written in practice)
79
143
  INDEX_PAGE = 1
144
+ # Dictionary page, precedes the data pages of a column chunk
80
145
  DICTIONARY_PAGE = 2
146
+ # Data page, format v2 (levels stored uncompressed ahead of the values)
81
147
  DATA_PAGE_V2 = 3
148
+ # Constant name for each enum value
82
149
  NAMES = constants.to_h { |c| [const_get(c), c] }.freeze
83
150
  end
84
151
 
152
+ # Short alias for the struct base class used by every definition below
85
153
  S = Thrift::Struct
86
154
 
155
+ # Byte and level-histogram statistics for a page or column chunk.
87
156
  class SizeStatistics < S
88
157
  field 1, :unencoded_byte_array_data_bytes, :i64
89
158
  field 2, :repetition_level_histogram, [:list, :i64]
90
159
  field 3, :definition_level_histogram, [:list, :i64]
91
160
  end
92
161
 
162
+ # Min/max and count statistics for a page or column chunk. +max+/+min+ are the deprecated
163
+ # signed-order fields; +max_value+/+min_value+ use the column's sort order.
93
164
  class Statistics < S
94
165
  field 1, :max, :binary
95
166
  field 2, :min, :binary
@@ -102,40 +173,77 @@ module Herringbone
102
173
  end
103
174
 
104
175
  # Empty marker structs used inside unions
176
+
177
+ # STRING logical type.
105
178
  class StringType < S; end
179
+
180
+ # UUID logical type (16-byte FIXED_LEN_BYTE_ARRAY).
106
181
  class UUIDType < S; end
182
+
183
+ # MAP logical type.
107
184
  class MapType < S; end
185
+
186
+ # LIST logical type.
108
187
  class ListType < S; end
188
+
189
+ # ENUM logical type.
109
190
  class EnumType < S; end
191
+
192
+ # DATE logical type.
110
193
  class DateType < S; end
194
+
195
+ # FLOAT16 logical type (2-byte FIXED_LEN_BYTE_ARRAY, little-endian half precision).
111
196
  class Float16Type < S; end
197
+
198
+ # UNKNOWN logical type: the column is always null.
112
199
  class NullType < S; end
200
+
201
+ # JSON logical type.
113
202
  class JsonType < S; end
203
+
204
+ # BSON logical type.
114
205
  class BsonType < S; end
206
+
207
+ # Millisecond member of the TimeUnit union.
115
208
  class MilliSeconds < S; end
209
+
210
+ # Microsecond member of the TimeUnit union.
116
211
  class MicroSeconds < S; end
212
+
213
+ # Nanosecond member of the TimeUnit union.
117
214
  class NanoSeconds < S; end
215
+
216
+ # ColumnOrder member: values are ordered according to their (logical) type.
118
217
  class TypeDefinedOrder < S; end
218
+
219
+ # Header of an index page; has no fields in the spec.
119
220
  class IndexPageHeader < S; end
120
221
 
222
+ # VARIANT logical type.
121
223
  class VariantType < S
122
224
  field 1, :specification_version, :byte
123
225
  end
124
226
 
227
+ # DECIMAL logical type: unscaled integer value divided by 10 to the power of +scale+.
125
228
  class DecimalType < S
126
229
  field 1, :scale, :i32
127
230
  field 2, :precision, :i32
128
231
  end
129
232
 
233
+ # Union of time units for TIME and TIMESTAMP; exactly one member is set.
130
234
  class TimeUnit < S
131
235
  field 1, :millis, MilliSeconds
132
236
  field 2, :micros, MicroSeconds
133
237
  field 3, :nanos, NanoSeconds
134
238
 
239
+ # @return [TimeUnit] a unit with the +millis+ member set
135
240
  def self.millis = new(millis: MilliSeconds.new)
241
+ # @return [TimeUnit] a unit with the +micros+ member set
136
242
  def self.micros = new(micros: MicroSeconds.new)
243
+ # @return [TimeUnit] a unit with the +nanos+ member set
137
244
  def self.nanos = new(nanos: NanoSeconds.new)
138
245
 
246
+ # @return [Symbol, nil] +:millis+, +:micros+ or +:nanos+, or nil when no member is set
139
247
  def to_sym
140
248
  if millis then :millis
141
249
  elsif micros then :micros
@@ -144,21 +252,26 @@ module Herringbone
144
252
  end
145
253
  end
146
254
 
255
+ # TIMESTAMP logical type.
147
256
  class TimestampType < S
148
257
  field 1, :is_adjusted_to_utc, :bool
149
258
  field 2, :unit, TimeUnit
150
259
  end
151
260
 
261
+ # TIME logical type.
152
262
  class TimeType < S
153
263
  field 1, :is_adjusted_to_utc, :bool
154
264
  field 2, :unit, TimeUnit
155
265
  end
156
266
 
267
+ # INTEGER logical type (bit width 8, 16, 32 or 64; signed or unsigned).
157
268
  class IntType < S
158
269
  field 1, :bit_width, :byte
159
270
  field 2, :is_signed, :bool
160
271
  end
161
272
 
273
+ # Union of logical type annotations; exactly one member is set. Field id 9 is reserved
274
+ # (for INTERVAL) in parquet.thrift.
162
275
  class LogicalType < S
163
276
  field 1, :string, StringType
164
277
  field 2, :map, MapType
@@ -177,6 +290,8 @@ module Herringbone
177
290
  field 16, :variant, VariantType
178
291
 
179
292
  # Returns [kind_symbol, payload] for whichever union member is set
293
+ # @return [Array(Symbol, Thrift::Struct), nil] member name and its struct, or nil when
294
+ # no member is set
180
295
  def kind
181
296
  self.class.fields.each do |f|
182
297
  v = instance_variable_get(f.ivar)
@@ -186,6 +301,8 @@ module Herringbone
186
301
  end
187
302
  end
188
303
 
304
+ # One node of the schema, flattened depth-first into +FileMetaData#schema+. Groups carry
305
+ # +num_children+; leaves carry the physical +type+.
189
306
  class SchemaElement < S
190
307
  field 1, :type, :i32
191
308
  field 2, :type_length, :i32
@@ -199,6 +316,7 @@ module Herringbone
199
316
  field 10, :logical_type, LogicalType
200
317
  end
201
318
 
319
+ # Header of a v1 data page: levels and values share one compressed block.
202
320
  class DataPageHeader < S
203
321
  field 1, :num_values, :i32
204
322
  field 2, :encoding, :i32
@@ -207,12 +325,15 @@ module Herringbone
207
325
  field 5, :statistics, Statistics
208
326
  end
209
327
 
328
+ # Header of a dictionary page.
210
329
  class DictionaryPageHeader < S
211
330
  field 1, :num_values, :i32
212
331
  field 2, :encoding, :i32
213
332
  field 3, :is_sorted, :bool
214
333
  end
215
334
 
335
+ # Header of a v2 data page: levels are stored uncompressed before the (optionally
336
+ # compressed) values, with their byte lengths given here.
216
337
  class DataPageHeaderV2 < S
217
338
  field 1, :num_values, :i32
218
339
  field 2, :num_nulls, :i32
@@ -224,6 +345,7 @@ module Herringbone
224
345
  field 8, :statistics, Statistics
225
346
  end
226
347
 
348
+ # Header preceding every page; +type+ (a PageType) says which sub-header is set.
227
349
  class PageHeader < S
228
350
  field 1, :type, :i32
229
351
  field 2, :uncompressed_page_size, :i32
@@ -235,23 +357,27 @@ module Herringbone
235
357
  field 8, :data_page_header_v2, DataPageHeaderV2
236
358
  end
237
359
 
360
+ # Application-defined key/value metadata entry.
238
361
  class KeyValue < S
239
362
  field 1, :key, :string
240
363
  field 2, :value, :string
241
364
  end
242
365
 
366
+ # Sort order of one column within a row group.
243
367
  class SortingColumn < S
244
368
  field 1, :column_idx, :i32
245
369
  field 2, :descending, :bool
246
370
  field 3, :nulls_first, :bool
247
371
  end
248
372
 
373
+ # Number of pages of a given type and encoding in a column chunk.
249
374
  class PageEncodingStats < S
250
375
  field 1, :page_type, :i32
251
376
  field 2, :encoding, :i32
252
377
  field 3, :count, :i32
253
378
  end
254
379
 
380
+ # Metadata of a column chunk: where its pages are, how they are encoded and compressed.
255
381
  class ColumnMetaData < S
256
382
  field 1, :type, :i32
257
383
  field 2, :encodings, [:list, :i32]
@@ -271,6 +397,8 @@ module Herringbone
271
397
  field 16, :size_statistics, SizeStatistics
272
398
  end
273
399
 
400
+ # One column of a row group, plus the locations of its page index. Field id 8
401
+ # (crypto_metadata) is not supported.
274
402
  class ColumnChunk < S
275
403
  field 1, :file_path, :string
276
404
  field 2, :file_offset, :i64
@@ -282,6 +410,7 @@ module Herringbone
282
410
  field 9, :encrypted_column_metadata, :binary
283
411
  end
284
412
 
413
+ # A horizontal slice of the file, holding one ColumnChunk per leaf column.
285
414
  class RowGroup < S
286
415
  field 1, :columns, [:list, ColumnChunk]
287
416
  field 2, :total_byte_size, :i64
@@ -292,10 +421,50 @@ module Herringbone
292
421
  field 7, :ordinal, :i16
293
422
  end
294
423
 
424
+ # Ordering of the per-page min/max values in a ColumnIndex (+BoundaryOrder+ enum).
425
+ module BoundaryOrder
426
+ # No particular order
427
+ UNORDERED = 0
428
+ # Min and max values both non-decreasing from page to page
429
+ ASCENDING = 1
430
+ # Min and max values both non-increasing from page to page
431
+ DESCENDING = 2
432
+ # Constant name for each enum value
433
+ NAMES = constants.to_h { |c| [const_get(c), c] }.freeze
434
+ end
435
+
436
+ # Page index structures (stored between the row groups and the footer)
437
+
438
+ # Location of one data page within the file.
439
+ class PageLocation < S
440
+ field 1, :offset, :i64
441
+ field 2, :compressed_page_size, :i32
442
+ field 3, :first_row_index, :i64
443
+ end
444
+
445
+ # Offset index of a column chunk: the location of each of its data pages.
446
+ class OffsetIndex < S
447
+ field 1, :page_locations, [:list, PageLocation]
448
+ field 2, :unencoded_byte_array_data_bytes, [:list, :i64]
449
+ end
450
+
451
+ # Column index of a column chunk: per-page min/max values and null information.
452
+ class ColumnIndex < S
453
+ field 1, :null_pages, [:list, :bool]
454
+ field 2, :min_values, [:list, :binary]
455
+ field 3, :max_values, [:list, :binary]
456
+ field 4, :boundary_order, :i32
457
+ field 5, :null_counts, [:list, :i64]
458
+ field 6, :repetition_level_histograms, [:list, :i64]
459
+ field 7, :definition_level_histograms, [:list, :i64]
460
+ end
461
+
462
+ # Union describing how min/max statistics of a column are ordered.
295
463
  class ColumnOrder < S
296
464
  field 1, :type_order, TypeDefinedOrder
297
465
  end
298
466
 
467
+ # The file footer: schema, row groups and file-level metadata.
299
468
  class FileMetaData < S
300
469
  field 1, :version, :i32
301
470
  field 2, :schema, [:list, SchemaElement]
@@ -305,5 +474,39 @@ module Herringbone
305
474
  field 6, :created_by, :string
306
475
  field 7, :column_orders, [:list, ColumnOrder]
307
476
  end
477
+
478
+ # Bloom filters (BloomFilter.md). Each union has a single member, an empty struct.
479
+
480
+ # Split block bloom filter algorithm.
481
+ class SplitBlockAlgorithm < S; end
482
+
483
+ # XXH64 hash with seed 0.
484
+ class XxHash < S; end
485
+
486
+ # Bloom filter bitset stored uncompressed.
487
+ class BloomFilterUncompressed < S; end
488
+
489
+ # Union of bloom filter algorithms.
490
+ class BloomFilterAlgorithm < S
491
+ field 1, :block, SplitBlockAlgorithm
492
+ end
493
+
494
+ # Union of hash functions used to feed the bloom filter.
495
+ class BloomFilterHash < S
496
+ field 1, :xxhash, XxHash
497
+ end
498
+
499
+ # Union of bloom filter compressions.
500
+ class BloomFilterCompression < S
501
+ field 1, :uncompressed, BloomFilterUncompressed
502
+ end
503
+
504
+ # Header preceding the bitset of a column chunk's bloom filter.
505
+ class BloomFilterHeader < S
506
+ field 1, :num_bytes, :i32
507
+ field 2, :algorithm, BloomFilterAlgorithm
508
+ field 3, :hash_function, BloomFilterHash # "hash" in parquet.thrift; renamed to keep Object#hash
509
+ field 4, :compression, BloomFilterCompression
510
+ end
308
511
  end
309
512
  end