herringbone 0.1.0 → 0.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,270 @@
1
+ # frozen_string_literal: true
2
+
3
+ module Herringbone
4
+ # A Parquet Split Block Bloom Filter (parquet-format BloomFilter.md).
5
+ #
6
+ # The bitset is made of 32-byte blocks of eight 32-bit words. A value is hashed with XXH64 over
7
+ # the PLAIN encoding of its physical value (without the length prefix for BYTE_ARRAY); the upper
8
+ # 32 bits of the hash pick a block, and the lower 32 bits set one bit in each of its words.
9
+ #
10
+ # A filter answers "definitely not in the column chunk" or "maybe": +might_contain?+ never
11
+ # returns false for a value that was inserted. Values are given as Ruby values and converted
12
+ # like the writer converts them (the column's encoder), so a Date, Time, BigDecimal or UUID
13
+ # String hashes the same bytes as the stored value. Nulls are never in a bloom filter.
14
+ # The writer builds them (bloom_filters: option) and reads with where: consult them.
15
+ class BloomFilter
16
+ SALT = [0x47b6137b, 0x44974d91, 0x8824ad5b, 0xa2b7289d, 0x705495c7, 0x2df1424b, 0x9efc4947, 0x5c6bfb31].freeze
17
+ S0, S1, S2, S3, S4, S5, S6, S7 = SALT
18
+ # Low 16 bits of the salts: (lo * salt) mod 2**32 is computed as
19
+ # (lo & 0xFFFF) * salt + ((lo >> 16) * (salt & 0xFFFF) << 16), which stays a Fixnum
20
+ L0, L1, L2, L3, L4, L5, L6, L7 = SALT.map { |s| s & 0xFFFF }
21
+ BLOCK_BYTES = 32
22
+ MIN_BYTES = 32
23
+ MAX_BYTES = 128 * 1024 * 1024
24
+ # Default cap for filters sized by the writer (as in parquet-mr)
25
+ DEFAULT_MAX_BYTES = 1024 * 1024
26
+ DEFAULT_FPP = 0.01
27
+ M32 = 0xFFFF_FFFF
28
+ M64 = 0xFFFF_FFFF_FFFF_FFFF
29
+ T = Format::Type
30
+ # Physical types a bloom filter can be built for (the spec does not define BOOLEAN hashing)
31
+ TYPES = [T::INT32, T::INT64, T::INT96, T::FLOAT, T::DOUBLE, T::BYTE_ARRAY, T::FIXED_LEN_BYTE_ARRAY].freeze
32
+
33
+ # Bitset size in bytes for +ndv+ distinct values at false positive probability +fpp+, per the
34
+ # spec's formula (m = -8 * ndv / ln(1 - fpp ** (1/8)) bits), rounded up to a power of two
35
+ # and clamped to MIN_BYTES..max_bytes
36
+ def self.optimal_num_bytes(ndv, fpp = DEFAULT_FPP, max_bytes: DEFAULT_MAX_BYTES)
37
+ fpp = Float(fpp)
38
+ raise ArgumentError, "fpp must be between 0 and 1, got #{fpp}" unless fpp > 0 && fpp < 1
39
+ max_bytes = [[Integer(max_bytes), MAX_BYTES].min, MIN_BYTES].max
40
+ max_bytes = 1 << (max_bytes.bit_length - 1) # a power of two
41
+ ndv = [Integer(ndv), 1].max
42
+ bits = -8.0 * ndv / Math.log(1 - fpp**(1.0 / 8))
43
+ bytes = (bits / 8).ceil
44
+ return max_bytes if bytes >= max_bytes
45
+ return MIN_BYTES if bytes <= MIN_BYTES
46
+ 1 << (bytes - 1).bit_length
47
+ end
48
+
49
+ # XXH64 of the PLAIN encoding of a physical value of +type+ (as the column encoders produce it)
50
+ def self.hash_physical(value, type)
51
+ case type
52
+ when T::INT32 then XXHash.xxh64_u32(value)
53
+ when T::INT64 then XXHash.xxh64_u64(value)
54
+ when T::FLOAT then XXHash.xxh64_u32([value].pack("e").unpack1("L<"))
55
+ when T::DOUBLE then XXHash.xxh64_u64([value].pack("E").unpack1("Q<"))
56
+ when T::INT96 then XXHash.xxh64(value.pack("Q<L<"))
57
+ when T::BYTE_ARRAY, T::FIXED_LEN_BYTE_ARRAY then XXHash.xxh64(value)
58
+ else raise UnsupportedError, "Bloom filters are not defined for #{T::NAMES.fetch(type, type)} values"
59
+ end
60
+ end
61
+
62
+ # Hashes of many physical values of +type+, converting them in bulk. With +distinct+, each
63
+ # distinct physical value is hashed once (floats are compared by their bytes, so -0.0 and 0.0,
64
+ # or NaNs with different payloads, stay apart as they hash differently).
65
+ def self.hash_physical_all(values, type, distinct: false)
66
+ case type
67
+ when T::INT32 then XXHash.xxh64_u32_all(distinct ? values.uniq : values)
68
+ when T::INT64 then XXHash.xxh64_u64_all(distinct ? values.uniq : values)
69
+ when T::FLOAT
70
+ words = values.pack("e*").unpack("L<*")
71
+ XXHash.xxh64_u32_all(distinct ? words.uniq! || words : words)
72
+ when T::DOUBLE
73
+ lanes = values.pack("E*").unpack("Q<*")
74
+ XXHash.xxh64_u64_all(distinct ? lanes.uniq! || lanes : lanes)
75
+ when T::INT96 then XXHash.xxh64_all((distinct ? values.uniq : values).map { |v| v.pack("Q<L<") })
76
+ when T::BYTE_ARRAY, T::FIXED_LEN_BYTE_ARRAY then XXHash.xxh64_all(distinct ? values.uniq : values)
77
+ else raise UnsupportedError, "Bloom filters are not defined for #{T::NAMES.fetch(type, type)} values"
78
+ end
79
+ end
80
+
81
+ # Reads a filter (header and bitset) from +buf+ at +pos+. Returns nil for algorithms, hashes
82
+ # or compressions this implementation does not know.
83
+ def self.decode(buf, pos = 0, column: nil)
84
+ reader = Thrift::Reader.new(buf, pos)
85
+ header = reader.read_struct(Format::BloomFilterHeader)
86
+ return nil unless supported_header?(header)
87
+ bitset = buf.byteslice(reader.pos, header.num_bytes)
88
+ raise FormatError, "Truncated bloom filter" if bitset.nil? || bitset.bytesize != header.num_bytes
89
+ new(bitset: bitset, column: column)
90
+ end
91
+
92
+ def self.supported_header?(header)
93
+ n = header.num_bytes
94
+ n.is_a?(Integer) && n >= BLOCK_BYTES && (n % BLOCK_BYTES).zero? && n <= MAX_BYTES &&
95
+ header.algorithm&.block && header.hash_function&.xxhash &&
96
+ (header.compression.nil? || header.compression.uncompressed)
97
+ end
98
+
99
+ attr_reader :column
100
+
101
+ # A filter of +num_bytes+ (a multiple of 32, normally a power of two), or one using an existing
102
+ # +bitset+ String. With a +column+ (a Schema::Column), values are Ruby values converted with
103
+ # the column's encoder; without one, values must be Strings and their bytes are hashed.
104
+ def initialize(num_bytes = nil, bitset: nil, column: nil)
105
+ if bitset
106
+ @words = bitset.unpack("V*")
107
+ else
108
+ num_bytes = Integer(num_bytes || MIN_BYTES)
109
+ unless num_bytes >= BLOCK_BYTES && (num_bytes % BLOCK_BYTES).zero? && num_bytes <= MAX_BYTES
110
+ raise ArgumentError, "Bloom filter size must be a multiple of 32 bytes up to 128MB, got #{num_bytes}"
111
+ end
112
+ @words = Array.new(num_bytes / 4, 0)
113
+ end
114
+ @num_blocks = @words.size / 8
115
+ raise ArgumentError, "Bloom filter bitset must be a multiple of 32 bytes" if @num_blocks.zero? || @words.size % 8 != 0
116
+ @column = column
117
+ if column && !TYPES.include?(column.type)
118
+ raise UnsupportedError, "Bloom filters are not supported for #{T::NAMES[column.type]} column #{column.dotted_path}"
119
+ end
120
+ end
121
+
122
+ def num_bytes = @words.size * 4
123
+
124
+ def insert(value)
125
+ insert_hash(hash_of(value))
126
+ self
127
+ end
128
+
129
+ def might_contain?(value)
130
+ might_contain_hash?(hash_of(value))
131
+ end
132
+
133
+ # XXH64 hash the filter uses for a Ruby +value+
134
+ def hash_of(value)
135
+ raise ArgumentError, "Nulls are not recorded in bloom filters" if value.nil?
136
+ if @column
137
+ begin
138
+ physical = @column.encoder.call(value)
139
+ rescue ArgumentError, TypeError, NoMethodError, RangeError, EncodeError => e
140
+ raise ArgumentError, "#{value.inspect} is not a valid value for #{@column.dotted_path}: #{e.message}"
141
+ end
142
+ self.class.hash_physical(physical, @column.type)
143
+ else
144
+ raise ArgumentError, "Without a column, bloom filter values must be Strings, got #{value.class}" unless value.is_a?(String)
145
+ XXHash.xxh64(value)
146
+ end
147
+ end
148
+
149
+ def insert_hash(h)
150
+ i = (((h >> 32) * @num_blocks) >> 32) << 3
151
+ x0 = h & 0xFFFF
152
+ x1 = (h >> 16) & 0xFFFF
153
+ w = @words
154
+ w[i] |= 1 << (((x0 * S0 + (x1 * L0 << 16)) & M32) >> 27)
155
+ w[i + 1] |= 1 << (((x0 * S1 + (x1 * L1 << 16)) & M32) >> 27)
156
+ w[i + 2] |= 1 << (((x0 * S2 + (x1 * L2 << 16)) & M32) >> 27)
157
+ w[i + 3] |= 1 << (((x0 * S3 + (x1 * L3 << 16)) & M32) >> 27)
158
+ w[i + 4] |= 1 << (((x0 * S4 + (x1 * L4 << 16)) & M32) >> 27)
159
+ w[i + 5] |= 1 << (((x0 * S5 + (x1 * L5 << 16)) & M32) >> 27)
160
+ w[i + 6] |= 1 << (((x0 * S6 + (x1 * L6 << 16)) & M32) >> 27)
161
+ w[i + 7] |= 1 << (((x0 * S7 + (x1 * L7 << 16)) & M32) >> 27)
162
+ self
163
+ end
164
+
165
+ BITS = Array.new(32) { |i| 1 << i }.freeze
166
+
167
+ # Inserts many hashes at once: faster than insert_hash, as the hashes (mostly Bignums) are
168
+ # split into 32-bit halves in bulk and the rest is Fixnum arithmetic
169
+ def insert_hashes(hashes)
170
+ halves = hashes.pack("Q<*").unpack("V*")
171
+ w = @words
172
+ num_blocks = @num_blocks
173
+ bit = BITS
174
+ j = 0
175
+ n = halves.size
176
+ while j < n
177
+ lo = halves[j]
178
+ i = halves[j + 1] * num_blocks / 4_294_967_296 * 8
179
+ x0 = lo & 0xFFFF
180
+ x1 = lo / 65_536
181
+ w[i] |= bit[(x0 * S0 + x1 * L0 * 65536) / 134_217_728 & 31]
182
+ w[i + 1] |= bit[(x0 * S1 + x1 * L1 * 65536) / 134_217_728 & 31]
183
+ w[i + 2] |= bit[(x0 * S2 + x1 * L2 * 65536) / 134_217_728 & 31]
184
+ w[i + 3] |= bit[(x0 * S3 + x1 * L3 * 65536) / 134_217_728 & 31]
185
+ w[i + 4] |= bit[(x0 * S4 + x1 * L4 * 65536) / 134_217_728 & 31]
186
+ w[i + 5] |= bit[(x0 * S5 + x1 * L5 * 65536) / 134_217_728 & 31]
187
+ w[i + 6] |= bit[(x0 * S6 + x1 * L6 * 65536) / 134_217_728 & 31]
188
+ w[i + 7] |= bit[(x0 * S7 + x1 * L7 * 65536) / 134_217_728 & 31]
189
+ j += 2
190
+ end
191
+ self
192
+ end
193
+
194
+ def might_contain_hash?(h)
195
+ i = (((h >> 32) * @num_blocks) >> 32) << 3
196
+ x0 = h & 0xFFFF
197
+ x1 = (h >> 16) & 0xFFFF
198
+ w = @words
199
+ w[i][(((x0 * S0 + (x1 * L0 << 16)) & M32) >> 27)] == 1 &&
200
+ w[i + 1][(((x0 * S1 + (x1 * L1 << 16)) & M32) >> 27)] == 1 &&
201
+ w[i + 2][(((x0 * S2 + (x1 * L2 << 16)) & M32) >> 27)] == 1 &&
202
+ w[i + 3][(((x0 * S3 + (x1 * L3 << 16)) & M32) >> 27)] == 1 &&
203
+ w[i + 4][(((x0 * S4 + (x1 * L4 << 16)) & M32) >> 27)] == 1 &&
204
+ w[i + 5][(((x0 * S5 + (x1 * L5 << 16)) & M32) >> 27)] == 1 &&
205
+ w[i + 6][(((x0 * S6 + (x1 * L6 << 16)) & M32) >> 27)] == 1 &&
206
+ w[i + 7][(((x0 * S7 + (x1 * L7 << 16)) & M32) >> 27)] == 1
207
+ end
208
+
209
+ # The raw bitset (little-endian 32-bit words)
210
+ def bitset = @words.pack("V*")
211
+
212
+ # Header and bitset, as stored in a Parquet file
213
+ def encode
214
+ header.encode << bitset
215
+ end
216
+
217
+ def inspect
218
+ "#<#{self.class.name} #{num_bytes} bytes#{" for #{@column.dotted_path}" if @column}>"
219
+ end
220
+
221
+ private
222
+
223
+ def header
224
+ Format::BloomFilterHeader.new(
225
+ num_bytes: num_bytes,
226
+ algorithm: Format::BloomFilterAlgorithm.new(block: Format::SplitBlockAlgorithm.new),
227
+ hash_function: Format::BloomFilterHash.new(xxhash: Format::XxHash.new),
228
+ compression: Format::BloomFilterCompression.new(uncompressed: Format::BloomFilterUncompressed.new)
229
+ )
230
+ end
231
+ end
232
+
233
+ class Reader
234
+ # Internal (used by reads with where:): the bloom filter of a column chunk. +column+ is a
235
+ # dotted path ("a.b"), an Array path or a Schema::Column. Returns a BloomFilter, or nil when
236
+ # the chunk has none (or one of an unknown kind).
237
+ def bloom_filter(row_group_index, column)
238
+ col = bloom_filter_column(column)
239
+ rg = row_groups.fetch(row_group_index) { raise IndexError, "No row group #{row_group_index}" }
240
+ meta = rg.columns.fetch(col.index).meta_data
241
+ offset = meta&.bloom_filter_offset
242
+ return nil unless offset
243
+ length = meta.bloom_filter_length
244
+ @io.seek(offset)
245
+ if length
246
+ buf = @io.read(length)
247
+ raise FormatError, "Truncated bloom filter" if buf.nil? || buf.bytesize < length
248
+ return BloomFilter.decode(buf, column: col)
249
+ end
250
+ # Without a length (older writers), read the header first, then the bitset it announces
251
+ head = @io.read(256) || "".b
252
+ reader = Thrift::Reader.new(head)
253
+ header = reader.read_struct(Format::BloomFilterHeader)
254
+ return nil unless BloomFilter.supported_header?(header)
255
+ @io.seek(offset + reader.pos)
256
+ bitset = @io.read(header.num_bytes)
257
+ raise FormatError, "Truncated bloom filter" if bitset.nil? || bitset.bytesize != header.num_bytes
258
+ BloomFilter.new(bitset: bitset, column: col)
259
+ end
260
+
261
+ private
262
+
263
+ def bloom_filter_column(column)
264
+ return column if column.is_a?(Schema::Column)
265
+ col = schema.column(column.is_a?(Array) ? column.map(&:to_s) : column.to_s)
266
+ raise ArgumentError, "No column #{column.inspect}" unless col
267
+ col
268
+ end
269
+ end
270
+ end
@@ -2,25 +2,82 @@
2
2
 
3
3
  require "zlib"
4
4
  require "stringio"
5
- require "zstd-ruby"
6
- require "brotli"
7
5
 
8
6
  module Herringbone
9
- # Dispatches page (de)compression by Parquet codec id. Snappy and LZ4 are pure Ruby,
10
- # GZIP uses zlib, ZSTD and Brotli use the zstd-ruby and brotli gems.
7
+ # Raised when a file uses (or a writer asks for) a codec whose library is not installed
8
+ class MissingCodecError < UnsupportedError
9
+ attr_reader :codec, :gem_name
10
+
11
+ def initialize(codec, gem_name, load_error)
12
+ @codec = codec
13
+ @gem_name = gem_name
14
+ super("#{codec} compression needs the \"#{gem_name}\" gem, which could not be loaded " \
15
+ "(#{load_error.message}). Add `gem \"#{gem_name}\"` to your Gemfile to use #{codec}.")
16
+ end
17
+ end
18
+
19
+ # Dispatches page (de)compression by Parquet codec id. Snappy and LZ4 are pure Ruby and GZIP
20
+ # uses zlib, so those always work. ZSTD and Brotli come from optional gems (zstd-ruby, brotli),
21
+ # which are required on first use; if they are missing, MissingCodecError says what to add.
11
22
  module Compression
12
23
  module_function
13
24
 
14
- CODECS_BY_NAME = {
15
- none: Format::Codec::UNCOMPRESSED, uncompressed: Format::Codec::UNCOMPRESSED,
16
- snappy: Format::Codec::SNAPPY, gzip: Format::Codec::GZIP, brotli: Format::Codec::BROTLI,
17
- lz4: Format::Codec::LZ4_RAW, lz4_raw: Format::Codec::LZ4_RAW, lz4_hadoop: Format::Codec::LZ4,
18
- zstd: Format::Codec::ZSTD
25
+ # Codecs backed by optional native gems: codec id => [name, gem, require path, constant]
26
+ LIBRARIES = {
27
+ Format::Codec::ZSTD => ["ZSTD", "zstd-ruby", "zstd-ruby", :Zstd],
28
+ Format::Codec::BROTLI => ["Brotli", "brotli", "brotli", :Brotli]
29
+ }.freeze
30
+
31
+ @libraries = {}
32
+ @library_mutex = Mutex.new
33
+
34
+ # The library module for a codec backed by an optional gem, requiring it on first use
35
+ def library(codec)
36
+ @libraries.fetch(codec) do
37
+ @library_mutex.synchronize do
38
+ @libraries.fetch(codec) do
39
+ name, gem_name, path, const = LIBRARIES.fetch(codec)
40
+ begin
41
+ require_library(path)
42
+ rescue LoadError => e
43
+ raise MissingCodecError.new(name, gem_name, e)
44
+ end
45
+ @libraries[codec] = Object.const_get(const)
46
+ end
47
+ end
48
+ end
49
+ end
50
+
51
+ def require_library(path)
52
+ require path
53
+ end
54
+
55
+ # Raises MissingCodecError (or UnsupportedError) unless +codec+ can be used
56
+ def ensure_available!(codec)
57
+ codec = codec_id(codec)
58
+ return library(codec) if LIBRARIES.key?(codec)
59
+ return if SUPPORTED.include?(codec)
60
+ raise UnsupportedError, "#{Format::Codec::NAMES[codec] || codec} compression is not supported"
61
+ end
62
+
63
+ SUPPORTED = [
64
+ Format::Codec::UNCOMPRESSED, Format::Codec::SNAPPY, Format::Codec::GZIP,
65
+ Format::Codec::LZ4_RAW, Format::Codec::LZ4
66
+ ].freeze
67
+
68
+ # Codec id => name, as given to the writer's compression: option and listed by Herringbone.codecs
69
+ NAMES = {
70
+ Format::Codec::UNCOMPRESSED => :none, Format::Codec::SNAPPY => :snappy, Format::Codec::GZIP => :gzip,
71
+ Format::Codec::LZ4_RAW => :lz4, Format::Codec::LZ4 => :lz4_hadoop, Format::Codec::ZSTD => :zstd,
72
+ Format::Codec::BROTLI => :brotli, Format::Codec::LZO => :lzo
19
73
  }.freeze
74
+ CODECS_BY_NAME = NAMES.invert.freeze
20
75
 
21
76
  def codec_id(name)
22
77
  return name if name.is_a?(Integer)
23
- CODECS_BY_NAME.fetch(name.to_s.downcase.to_sym) { raise ArgumentError, "Unknown compression codec #{name.inspect}" }
78
+ CODECS_BY_NAME.fetch(name.to_s.downcase.to_sym) do
79
+ raise ArgumentError, "Unknown compression codec #{name.inspect}, expected one of #{NAMES.values.map(&:inspect).join(", ")}"
80
+ end
24
81
  end
25
82
 
26
83
  def decompress(codec, data, uncompressed_size)
@@ -31,14 +88,14 @@ module Herringbone
31
88
  when Format::Codec::GZIP then gunzip(data)
32
89
  when Format::Codec::LZ4_RAW then Codecs::LZ4.decompress_block(data, uncompressed_size)
33
90
  when Format::Codec::LZ4 then Codecs::LZ4.decompress_hadoop(data, uncompressed_size)
34
- when Format::Codec::ZSTD then ::Zstd.decompress(data)
35
- when Format::Codec::BROTLI then ::Brotli.inflate(data)
91
+ when Format::Codec::ZSTD then library(codec).decompress(data)
92
+ when Format::Codec::BROTLI then library(codec).inflate(data)
36
93
  else
37
94
  raise UnsupportedError, "Unsupported compression codec #{Format::Codec::NAMES[codec] || codec}"
38
95
  end
39
96
  out = out.b
40
97
  if out.bytesize != uncompressed_size
41
- raise DecodeError, "Decompressed #{out.bytesize} bytes, expected #{uncompressed_size}"
98
+ raise FormatError, "Decompressed #{out.bytesize} bytes, expected #{uncompressed_size}"
42
99
  end
43
100
  out
44
101
  end
@@ -50,8 +107,8 @@ module Herringbone
50
107
  when Format::Codec::GZIP then Zlib.gzip(data)
51
108
  when Format::Codec::LZ4_RAW then Codecs::LZ4.compress_block(data)
52
109
  when Format::Codec::LZ4 then Codecs::LZ4.compress_hadoop(data)
53
- when Format::Codec::ZSTD then ::Zstd.compress(data)
54
- when Format::Codec::BROTLI then ::Brotli.deflate(data)
110
+ when Format::Codec::ZSTD then library(codec).compress(data)
111
+ when Format::Codec::BROTLI then library(codec).deflate(data)
55
112
  else
56
113
  raise UnsupportedError, "Unsupported compression codec #{Format::Codec::NAMES[codec] || codec}"
57
114
  end.b
@@ -61,7 +118,7 @@ module Herringbone
61
118
  def gunzip(data)
62
119
  io = StringIO.new(data)
63
120
  out = String.new(encoding: Encoding::BINARY)
64
- loop do
121
+ while true
65
122
  gz = Zlib::GzipReader.new(io)
66
123
  out << gz.read
67
124
  unused = gz.unused
@@ -25,9 +25,9 @@ module Herringbone
25
25
  miniblocks, pos = RLE.read_uleb(data, pos)
26
26
  total, pos = RLE.read_uleb(data, pos)
27
27
  first, pos = RLE.read_uleb(data, pos)
28
- raise DecodeError, "Invalid DELTA_BINARY_PACKED header" if miniblocks.zero? || block_size % miniblocks != 0
28
+ raise FormatError, "Invalid DELTA_BINARY_PACKED header" if miniblocks.zero? || block_size % miniblocks != 0
29
29
  per_mini = block_size / miniblocks
30
- raise DecodeError, "Invalid miniblock size #{per_mini}" if per_mini % 8 != 0
30
+ raise FormatError, "Invalid miniblock size #{per_mini}" if per_mini % 8 != 0
31
31
  total = count if count && count < total
32
32
  values = []
33
33
  return [values, pos] if total.zero?
@@ -42,7 +42,7 @@ module Herringbone
42
42
  pos += miniblocks
43
43
  widths.each do |w|
44
44
  break if values.size >= total
45
- raise DecodeError, "Invalid delta bit width #{w}" if w > bits
45
+ raise FormatError, "Invalid delta bit width #{w}" if w > bits
46
46
  deltas = RLE.unpack_bits(data, pos, per_mini, w)
47
47
  pos += per_mini * w / 8
48
48
  take = total - values.size
@@ -84,11 +84,11 @@ module Herringbone
84
84
 
85
85
  def decode_length_byte_array(data, pos, count)
86
86
  lengths, pos = decode_binary_packed(data, pos, 32, count)
87
- raise DecodeError, "DELTA_LENGTH_BYTE_ARRAY has #{lengths.size} lengths, need #{count}" if lengths.size < count
87
+ raise FormatError, "DELTA_LENGTH_BYTE_ARRAY has #{lengths.size} lengths, need #{count}" if lengths.size < count
88
88
  out = Array.new(count)
89
89
  size = data.bytesize
90
90
  lengths.each_with_index do |len, i|
91
- raise DecodeError, "DELTA_LENGTH_BYTE_ARRAY value overruns page" if len.negative? || pos + len > size
91
+ raise FormatError, "DELTA_LENGTH_BYTE_ARRAY value overruns page" if len.negative? || pos + len > size
92
92
  out[i] = data.byteslice(pos, len)
93
93
  pos += len
94
94
  end
@@ -105,7 +105,7 @@ module Herringbone
105
105
  prev = "".b
106
106
  out = Array.new(count) do |i|
107
107
  prefix = prefixes[i]
108
- raise DecodeError, "DELTA_BYTE_ARRAY prefix longer than previous value" if prefix > prev.bytesize
108
+ raise FormatError, "DELTA_BYTE_ARRAY prefix longer than previous value" if prefix > prev.bytesize
109
109
  prev = prefix.zero? ? suffixes[i] : prev.byteslice(0, prefix) + suffixes[i]
110
110
  end
111
111
  [out, pos]
@@ -134,7 +134,7 @@ module Herringbone
134
134
  # Returns the value bytes re-interleaved into PLAIN layout
135
135
  def decode(data, pos, count, width)
136
136
  nbytes = count * width
137
- raise DecodeError, "Truncated BYTE_STREAM_SPLIT data" if pos + nbytes > data.bytesize
137
+ raise FormatError, "Truncated BYTE_STREAM_SPLIT data" if pos + nbytes > data.bytesize
138
138
  streams = Array.new(width) { |k| data.byteslice(pos + k * count, count).unpack("C*") }
139
139
  [streams[0].zip(*streams[1..]).flatten.pack("C*"), pos + nbytes]
140
140
  end
@@ -21,26 +21,26 @@ module Herringbone
21
21
  when Format::Type::BOOLEAN
22
22
  nbytes = (count + 7) / 8
23
23
  bits = data.byteslice(pos, nbytes).unpack1("b*")
24
- raise DecodeError, "Truncated BOOLEAN data" if bits.bytesize < count
24
+ raise FormatError, "Truncated BOOLEAN data" if bits.bytesize < count
25
25
  [Array.new(count) { |i| bits.getbyte(i) == 49 }, pos + nbytes]
26
26
  when Format::Type::INT32, Format::Type::INT64, Format::Type::FLOAT, Format::Type::DOUBLE
27
27
  fmt, width = FORMATS[type]
28
28
  nbytes = count * width
29
- raise DecodeError, "Truncated PLAIN data" if pos + nbytes > data.bytesize
29
+ raise FormatError, "Truncated PLAIN data" if pos + nbytes > data.bytesize
30
30
  [data.byteslice(pos, nbytes).unpack("#{fmt}#{count}"), pos + nbytes]
31
31
  when Format::Type::INT96
32
32
  nbytes = count * 12
33
- raise DecodeError, "Truncated INT96 data" if pos + nbytes > data.bytesize
33
+ raise FormatError, "Truncated INT96 data" if pos + nbytes > data.bytesize
34
34
  vals = data.byteslice(pos, nbytes).unpack("Q<L<" * count).each_slice(2).map { |nanos, day| [nanos, day] }
35
35
  [vals, pos + nbytes]
36
36
  when Format::Type::BYTE_ARRAY
37
37
  decode_byte_arrays(data, pos, count)
38
38
  when Format::Type::FIXED_LEN_BYTE_ARRAY
39
39
  nbytes = count * type_length
40
- raise DecodeError, "Truncated FIXED_LEN_BYTE_ARRAY data" if pos + nbytes > data.bytesize
40
+ raise FormatError, "Truncated FIXED_LEN_BYTE_ARRAY data" if pos + nbytes > data.bytesize
41
41
  [Array.new(count) { |i| data.byteslice(pos + i * type_length, type_length) }, pos + nbytes]
42
42
  else
43
- raise DecodeError, "Unknown physical type #{type}"
43
+ raise FormatError, "Unknown physical type #{type}"
44
44
  end
45
45
  end
46
46
 
@@ -49,10 +49,10 @@ module Herringbone
49
49
  size = data.bytesize
50
50
  i = 0
51
51
  while i < count
52
- raise DecodeError, "Truncated BYTE_ARRAY data" if pos + 4 > size
52
+ raise FormatError, "Truncated BYTE_ARRAY data" if pos + 4 > size
53
53
  len = data.getbyte(pos) | (data.getbyte(pos + 1) << 8) | (data.getbyte(pos + 2) << 16) | (data.getbyte(pos + 3) << 24)
54
54
  pos += 4
55
- raise DecodeError, "BYTE_ARRAY value overruns page" if pos + len > size
55
+ raise FormatError, "BYTE_ARRAY value overruns page" if pos + len > size
56
56
  out[i] = data.byteslice(pos, len)
57
57
  pos += len
58
58
  i += 1
@@ -98,7 +98,7 @@ module Herringbone
98
98
  shift = 0
99
99
  while true
100
100
  b = data.getbyte(pos)
101
- raise DecodeError, "Truncated varint" unless b
101
+ raise FormatError, "Truncated varint" unless b
102
102
  pos += 1
103
103
  result |= (b & 0x7F) << shift
104
104
  return [result, pos] if b < 0x80
@@ -120,7 +120,7 @@ module Herringbone
120
120
  out = []
121
121
  value_bytes = (width + 7) / 8
122
122
  while out.size < count
123
- raise DecodeError, "RLE data exhausted (#{out.size}/#{count} values)" if pos >= limit
123
+ raise FormatError, "RLE data exhausted (#{out.size}/#{count} values)" if pos >= limit
124
124
  header, pos = read_uleb(data, pos)
125
125
  if header & 1 == 1
126
126
  groups = header >> 1
@@ -292,6 +292,35 @@ module Herringbone
292
292
  field 7, :ordinal, :i16
293
293
  end
294
294
 
295
+ module BoundaryOrder
296
+ UNORDERED = 0
297
+ ASCENDING = 1
298
+ DESCENDING = 2
299
+ NAMES = constants.to_h { |c| [const_get(c), c] }.freeze
300
+ end
301
+
302
+ # Page index structures (stored between the row groups and the footer)
303
+ class PageLocation < S
304
+ field 1, :offset, :i64
305
+ field 2, :compressed_page_size, :i32
306
+ field 3, :first_row_index, :i64
307
+ end
308
+
309
+ class OffsetIndex < S
310
+ field 1, :page_locations, [:list, PageLocation]
311
+ field 2, :unencoded_byte_array_data_bytes, [:list, :i64]
312
+ end
313
+
314
+ class ColumnIndex < S
315
+ field 1, :null_pages, [:list, :bool]
316
+ field 2, :min_values, [:list, :binary]
317
+ field 3, :max_values, [:list, :binary]
318
+ field 4, :boundary_order, :i32
319
+ field 5, :null_counts, [:list, :i64]
320
+ field 6, :repetition_level_histograms, [:list, :i64]
321
+ field 7, :definition_level_histograms, [:list, :i64]
322
+ end
323
+
295
324
  class ColumnOrder < S
296
325
  field 1, :type_order, TypeDefinedOrder
297
326
  end
@@ -305,5 +334,29 @@ module Herringbone
305
334
  field 6, :created_by, :string
306
335
  field 7, :column_orders, [:list, ColumnOrder]
307
336
  end
337
+
338
+ # Bloom filters (BloomFilter.md). Each union has a single member, an empty struct.
339
+ class SplitBlockAlgorithm < S; end
340
+ class XxHash < S; end
341
+ class BloomFilterUncompressed < S; end
342
+
343
+ class BloomFilterAlgorithm < S
344
+ field 1, :block, SplitBlockAlgorithm
345
+ end
346
+
347
+ class BloomFilterHash < S
348
+ field 1, :xxhash, XxHash
349
+ end
350
+
351
+ class BloomFilterCompression < S
352
+ field 1, :uncompressed, BloomFilterUncompressed
353
+ end
354
+
355
+ class BloomFilterHeader < S
356
+ field 1, :num_bytes, :i32
357
+ field 2, :algorithm, BloomFilterAlgorithm
358
+ field 3, :hash_function, BloomFilterHash # "hash" in parquet.thrift; renamed to keep Object#hash
359
+ field 4, :compression, BloomFilterCompression
360
+ end
308
361
  end
309
362
  end