herringbone 0.1.0 → 0.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,370 @@
1
+ # frozen_string_literal: true
2
+
3
+ module Herringbone
4
+ # A Parquet Split Block Bloom Filter (parquet-format BloomFilter.md).
5
+ #
6
+ # The bitset is made of 32-byte blocks of eight 32-bit words. A value is hashed with XXH64 over
7
+ # the PLAIN encoding of its physical value (without the length prefix for BYTE_ARRAY); the upper
8
+ # 32 bits of the hash pick a block, and the lower 32 bits set one bit in each of its words.
9
+ #
10
+ # A filter answers "definitely not in the column chunk" or "maybe": +might_contain?+ never
11
+ # returns false for a value that was inserted. Values are given as Ruby values and converted
12
+ # like the writer converts them (the column's encoder), so a Date, Time, BigDecimal or UUID
13
+ # String hashes the same bytes as the stored value. Nulls are never in a bloom filter.
14
+ # The writer builds them (bloom_filters: option) and reads with where: consult them.
15
+ class BloomFilter
16
+ # The spec's eight salt constants, one per word of a block
17
+ SALT = [0x47b6137b, 0x44974d91, 0x8824ad5b, 0xa2b7289d, 0x705495c7, 0x2df1424b, 0x9efc4947, 0x5c6bfb31].freeze
18
+ S0, S1, S2, S3, S4, S5, S6, S7 = SALT
19
+ # Low 16 bits of the salts: (lo * salt) mod 2**32 is computed as
20
+ # (lo & 0xFFFF) * salt + ((lo >> 16) * (salt & 0xFFFF) << 16), which stays a Fixnum
21
+ L0, L1, L2, L3, L4, L5, L6, L7 = SALT.map { |s| s & 0xFFFF }
22
+ # Size of a block: eight 32-bit words
23
+ BLOCK_BYTES = 32
24
+ # Smallest bitset: a single block
25
+ MIN_BYTES = 32
26
+ # Largest bitset accepted when reading or building (128 MiB, the cap parquet-mr uses)
27
+ MAX_BYTES = 128 * 1024 * 1024
28
+ # Default cap for filters sized by the writer (as in parquet-mr)
29
+ DEFAULT_MAX_BYTES = 1024 * 1024
30
+ # Default false positive probability for filters sized by the writer
31
+ DEFAULT_FPP = 0.01
32
+ # 32-bit mask
33
+ M32 = 0xFFFF_FFFF
34
+ # 64-bit mask
35
+ M64 = 0xFFFF_FFFF_FFFF_FFFF
36
+ # Shorthand for the physical type constants
37
+ T = Format::Type
38
+ # Physical types a bloom filter can be built for (the spec does not define BOOLEAN hashing)
39
+ TYPES = [T::INT32, T::INT64, T::INT96, T::FLOAT, T::DOUBLE, T::BYTE_ARRAY, T::FIXED_LEN_BYTE_ARRAY].freeze
40
+
41
+ # Bitset size in bytes for +ndv+ distinct values at false positive probability +fpp+, per the
42
+ # spec's formula (m = -8 * ndv / ln(1 - fpp ** (1/8)) bits), rounded up to a power of two
43
+ # and clamped to MIN_BYTES..max_bytes
44
+ #
45
+ # @param ndv [Integer] expected number of distinct values (values below 1 count as 1)
46
+ # @param fpp [Float] false positive probability, strictly between 0 and 1
47
+ # @param max_bytes [Integer] upper bound, clamped to MIN_BYTES..MAX_BYTES and rounded down to a
48
+ # power of two
49
+ # @return [Integer] bitset size in bytes, a power of two
50
+ # @raise [ArgumentError] when +fpp+ is not between 0 and 1
51
+ def self.optimal_num_bytes(ndv, fpp = DEFAULT_FPP, max_bytes: DEFAULT_MAX_BYTES)
52
+ fpp = Float(fpp)
53
+ raise ArgumentError, "fpp must be between 0 and 1, got #{fpp}" unless fpp > 0 && fpp < 1
54
+ max_bytes = Integer(max_bytes).clamp(MIN_BYTES, MAX_BYTES)
55
+ max_bytes = 1 << (max_bytes.bit_length - 1) # a power of two
56
+ ndv = [Integer(ndv), 1].max
57
+ bits = -8.0 * ndv / Math.log(1 - fpp**(1.0 / 8))
58
+ bytes = (bits / 8).ceil
59
+ return max_bytes if bytes >= max_bytes
60
+ return MIN_BYTES if bytes <= MIN_BYTES
61
+ 1 << (bytes - 1).bit_length
62
+ end
63
+
64
+ # XXH64 of the PLAIN encoding of a physical value of +type+ (as the column encoders produce it)
65
+ #
66
+ # @param value [Integer, Float, String, Array<Integer>] the physical value; for INT96 the
67
+ # +[nanos_of_day, julian_day]+ pair the encoder produces
68
+ # @param type [Integer] Format::Type physical type
69
+ # @return [Integer] the unsigned 64-bit hash
70
+ # @raise [UnsupportedError] for BOOLEAN (or any type not in TYPES)
71
+ def self.hash_physical(value, type)
72
+ case type
73
+ when T::INT32 then XXHash.xxh64_u32(value)
74
+ when T::INT64 then XXHash.xxh64_u64(value)
75
+ when T::FLOAT then XXHash.xxh64_u32([value].pack("e").unpack1("L<"))
76
+ when T::DOUBLE then XXHash.xxh64_u64([value].pack("E").unpack1("Q<"))
77
+ when T::INT96 then XXHash.xxh64(value.pack("Q<L<"))
78
+ when T::BYTE_ARRAY, T::FIXED_LEN_BYTE_ARRAY then XXHash.xxh64(value)
79
+ else raise UnsupportedError, "Bloom filters are not defined for #{T::NAMES.fetch(type, type)} values"
80
+ end
81
+ end
82
+
83
+ # Hashes of many physical values of +type+, converting them in bulk. With +distinct+, each
84
+ # distinct physical value is hashed once (floats are compared by their bytes, so -0.0 and 0.0,
85
+ # or NaNs with different payloads, stay apart as they hash differently).
86
+ #
87
+ # @param values [Array] physical values, as for .hash_physical
88
+ # @param type [Integer] Format::Type physical type
89
+ # @param distinct [Boolean] hash each distinct value once (the result is then shorter)
90
+ # @return [Array<Integer>] the unsigned 64-bit hashes
91
+ # @raise [UnsupportedError] for BOOLEAN (or any type not in TYPES)
92
+ def self.hash_physical_all(values, type, distinct: false)
93
+ case type
94
+ when T::INT32 then XXHash.xxh64_u32_all(distinct ? values.uniq : values)
95
+ when T::INT64 then XXHash.xxh64_u64_all(distinct ? values.uniq : values)
96
+ when T::FLOAT
97
+ words = values.pack("e*").unpack("L<*")
98
+ XXHash.xxh64_u32_all(distinct ? words.uniq! || words : words)
99
+ when T::DOUBLE
100
+ lanes = values.pack("E*").unpack("Q<*")
101
+ XXHash.xxh64_u64_all(distinct ? lanes.uniq! || lanes : lanes)
102
+ when T::INT96 then XXHash.xxh64_all((distinct ? values.uniq : values).map { |v| v.pack("Q<L<") })
103
+ when T::BYTE_ARRAY, T::FIXED_LEN_BYTE_ARRAY then XXHash.xxh64_all(distinct ? values.uniq : values)
104
+ else raise UnsupportedError, "Bloom filters are not defined for #{T::NAMES.fetch(type, type)} values"
105
+ end
106
+ end
107
+
108
+ # Reads a filter (header and bitset) from +buf+ at +pos+. Returns nil for algorithms, hashes
109
+ # or compressions this implementation does not know.
110
+ #
111
+ # @param buf [String] bytes holding the BloomFilterHeader followed by the bitset
112
+ # @param pos [Integer] byte offset of the header in +buf+
113
+ # @param column [Schema::Column, nil] column the filter belongs to, for converting values
114
+ # @return [BloomFilter, nil] the filter, or nil when it is of an unsupported kind
115
+ # @raise [FormatError] when the bitset is shorter than the header announces
116
+ # @raise [Thrift::Error] when the header cannot be decoded
117
+ def self.decode(buf, pos = 0, column: nil)
118
+ reader = Thrift::Reader.new(buf, pos)
119
+ header = reader.read_struct(Format::BloomFilterHeader)
120
+ return nil unless supported_header?(header)
121
+ bitset = buf.byteslice(reader.pos, header.num_bytes)
122
+ raise FormatError, "Truncated bloom filter" if bitset.nil? || bitset.bytesize != header.num_bytes
123
+ new(bitset: bitset, column: column)
124
+ end
125
+
126
+ # Whether a header describes a filter this implementation can read: split block algorithm,
127
+ # XXH64, uncompressed, and a size that is a multiple of 32 bytes up to MAX_BYTES
128
+ #
129
+ # @param header [Format::BloomFilterHeader] decoded header
130
+ # @return [Boolean] true when supported
131
+ def self.supported_header?(header)
132
+ n = header.num_bytes
133
+ n.is_a?(Integer) && n >= BLOCK_BYTES && (n % BLOCK_BYTES).zero? && n <= MAX_BYTES &&
134
+ !(header.algorithm && header.algorithm.block).nil? &&
135
+ !(header.hash_function && header.hash_function.xxhash).nil? &&
136
+ (header.compression.nil? || !header.compression.uncompressed.nil?)
137
+ end
138
+
139
+ # @return [Schema::Column, nil] column whose encoder converts values, nil for raw Strings
140
+ attr_reader :column
141
+
142
+ # A filter of +num_bytes+ (a multiple of 32, normally a power of two), or one using an existing
143
+ # +bitset+ String. With a +column+ (a Schema::Column), values are Ruby values converted with
144
+ # the column's encoder; without one, values must be Strings and their bytes are hashed.
145
+ #
146
+ # @param num_bytes [Integer, nil] bitset size for an empty filter (MIN_BYTES when nil); ignored
147
+ # with +bitset+
148
+ # @param bitset [String, nil] existing bitset (little-endian 32-bit words)
149
+ # @param column [Schema::Column, nil] column the filter is for
150
+ # @raise [ArgumentError] when the size (or +bitset+'s size) is not a multiple of 32 bytes or
151
+ # exceeds MAX_BYTES
152
+ # @raise [UnsupportedError] when the column's physical type cannot have a bloom filter
153
+ def initialize(num_bytes = nil, bitset: nil, column: nil)
154
+ num_bytes = bitset ? bitset.bytesize : Integer(num_bytes || MIN_BYTES)
155
+ unless num_bytes >= BLOCK_BYTES && (num_bytes % BLOCK_BYTES).zero? && num_bytes <= MAX_BYTES
156
+ raise ArgumentError, "Bloom filter size must be a multiple of 32 bytes up to 128MB, got #{num_bytes}"
157
+ end
158
+ @words = bitset ? bitset.unpack("V*") : Array.new(num_bytes / 4, 0)
159
+ @num_blocks = @words.size / 8
160
+ @column = column
161
+ if column && !TYPES.include?(column.type)
162
+ raise UnsupportedError, "Bloom filters are not supported for #{T::NAMES[column.type]} column #{column.dotted_path}"
163
+ end
164
+ end
165
+
166
+ # @return [Integer] bitset size in bytes
167
+ def num_bytes = @words.size * 4
168
+
169
+ # Adds a value to the filter
170
+ #
171
+ # @param value [Object] Ruby value, converted with the column's encoder (a String without a
172
+ # column)
173
+ # @return [BloomFilter] self
174
+ # @raise [ArgumentError] for nil or a value the column cannot encode
175
+ def insert(value)
176
+ insert_hash(hash_of(value))
177
+ self
178
+ end
179
+
180
+ # Whether the value may be in the filter. Never false for an inserted value; true for a
181
+ # value that was not inserted with about the false positive probability the filter was sized for.
182
+ #
183
+ # @param value [Object] Ruby value, converted with the column's encoder (a String without a
184
+ # column)
185
+ # @return [Boolean] false when the value is definitely absent
186
+ # @raise [ArgumentError] for nil or a value the column cannot encode
187
+ def might_contain?(value)
188
+ might_contain_hash?(hash_of(value))
189
+ end
190
+
191
+ # XXH64 hash the filter uses for a Ruby +value+
192
+ #
193
+ # @param value [Object] Ruby value, converted with the column's encoder (a String without a
194
+ # column)
195
+ # @return [Integer] the unsigned 64-bit hash
196
+ # @raise [ArgumentError] for nil, a value the column cannot encode, or (without a column) a
197
+ # non-String
198
+ def hash_of(value)
199
+ raise ArgumentError, "Nulls are not recorded in bloom filters" if value.nil?
200
+ if @column
201
+ begin
202
+ physical = @column.encoder.call(value)
203
+ rescue ArgumentError, TypeError, NoMethodError, RangeError, EncodeError => e
204
+ raise ArgumentError, "#{value.inspect} is not a valid value for #{@column.dotted_path}: #{e.message}"
205
+ end
206
+ self.class.hash_physical(physical, @column.type)
207
+ else
208
+ raise ArgumentError, "Without a column, bloom filter values must be Strings, got #{value.class}" unless value.is_a?(String)
209
+ XXHash.xxh64(value)
210
+ end
211
+ end
212
+
213
+ # Sets the bits for a hash: the upper 32 bits pick the block, the lower 32 bits (multiplied
214
+ # by each salt) one bit in each of its eight words
215
+ #
216
+ # @param h [Integer] unsigned 64-bit XXH64 hash
217
+ # @return [BloomFilter] self
218
+ def insert_hash(h)
219
+ i = (((h >> 32) * @num_blocks) >> 32) << 3
220
+ x0 = h & 0xFFFF
221
+ x1 = (h >> 16) & 0xFFFF
222
+ w = @words
223
+ w[i] |= 1 << (((x0 * S0 + (x1 * L0 << 16)) & M32) >> 27)
224
+ w[i + 1] |= 1 << (((x0 * S1 + (x1 * L1 << 16)) & M32) >> 27)
225
+ w[i + 2] |= 1 << (((x0 * S2 + (x1 * L2 << 16)) & M32) >> 27)
226
+ w[i + 3] |= 1 << (((x0 * S3 + (x1 * L3 << 16)) & M32) >> 27)
227
+ w[i + 4] |= 1 << (((x0 * S4 + (x1 * L4 << 16)) & M32) >> 27)
228
+ w[i + 5] |= 1 << (((x0 * S5 + (x1 * L5 << 16)) & M32) >> 27)
229
+ w[i + 6] |= 1 << (((x0 * S6 + (x1 * L6 << 16)) & M32) >> 27)
230
+ w[i + 7] |= 1 << (((x0 * S7 + (x1 * L7 << 16)) & M32) >> 27)
231
+ self
232
+ end
233
+
234
+ # Single-bit masks, +BITS[i] == 1 << i+
235
+ BITS = Array.new(32) { |i| 1 << i }.freeze
236
+
237
+ # Inserts many hashes at once: faster than insert_hash, as the hashes (mostly Bignums) are
238
+ # split into 32-bit halves in bulk and the rest is Fixnum arithmetic
239
+ #
240
+ # @param hashes [Array<Integer>] unsigned 64-bit XXH64 hashes
241
+ # @return [BloomFilter] self
242
+ def insert_hashes(hashes)
243
+ halves = hashes.pack("Q<*").unpack("V*")
244
+ w = @words
245
+ num_blocks = @num_blocks
246
+ bit = BITS
247
+ j = 0
248
+ n = halves.size
249
+ while j < n
250
+ lo = halves[j]
251
+ i = halves[j + 1] * num_blocks / 4_294_967_296 * 8
252
+ x0 = lo & 0xFFFF
253
+ x1 = lo / 65_536
254
+ w[i] |= bit[(x0 * S0 + x1 * L0 * 65536) / 134_217_728 & 31]
255
+ w[i + 1] |= bit[(x0 * S1 + x1 * L1 * 65536) / 134_217_728 & 31]
256
+ w[i + 2] |= bit[(x0 * S2 + x1 * L2 * 65536) / 134_217_728 & 31]
257
+ w[i + 3] |= bit[(x0 * S3 + x1 * L3 * 65536) / 134_217_728 & 31]
258
+ w[i + 4] |= bit[(x0 * S4 + x1 * L4 * 65536) / 134_217_728 & 31]
259
+ w[i + 5] |= bit[(x0 * S5 + x1 * L5 * 65536) / 134_217_728 & 31]
260
+ w[i + 6] |= bit[(x0 * S6 + x1 * L6 * 65536) / 134_217_728 & 31]
261
+ w[i + 7] |= bit[(x0 * S7 + x1 * L7 * 65536) / 134_217_728 & 31]
262
+ j += 2
263
+ end
264
+ self
265
+ end
266
+
267
+ # Whether all the bits for a hash are set (see #insert_hash)
268
+ #
269
+ # @param h [Integer] unsigned 64-bit XXH64 hash
270
+ # @return [Boolean] false when the hashed value is definitely absent
271
+ def might_contain_hash?(h)
272
+ i = (((h >> 32) * @num_blocks) >> 32) << 3
273
+ x0 = h & 0xFFFF
274
+ x1 = (h >> 16) & 0xFFFF
275
+ w = @words
276
+ w[i][(((x0 * S0 + (x1 * L0 << 16)) & M32) >> 27)] == 1 &&
277
+ w[i + 1][(((x0 * S1 + (x1 * L1 << 16)) & M32) >> 27)] == 1 &&
278
+ w[i + 2][(((x0 * S2 + (x1 * L2 << 16)) & M32) >> 27)] == 1 &&
279
+ w[i + 3][(((x0 * S3 + (x1 * L3 << 16)) & M32) >> 27)] == 1 &&
280
+ w[i + 4][(((x0 * S4 + (x1 * L4 << 16)) & M32) >> 27)] == 1 &&
281
+ w[i + 5][(((x0 * S5 + (x1 * L5 << 16)) & M32) >> 27)] == 1 &&
282
+ w[i + 6][(((x0 * S6 + (x1 * L6 << 16)) & M32) >> 27)] == 1 &&
283
+ w[i + 7][(((x0 * S7 + (x1 * L7 << 16)) & M32) >> 27)] == 1
284
+ end
285
+
286
+ # The raw bitset (little-endian 32-bit words)
287
+ #
288
+ # @return [String] binary String of #num_bytes bytes
289
+ def bitset = @words.pack("V*")
290
+
291
+ # Header and bitset, as stored in a Parquet file
292
+ #
293
+ # @return [String] Thrift-encoded BloomFilterHeader followed by the bitset (binary)
294
+ def encode
295
+ header.encode << bitset
296
+ end
297
+
298
+ # @return [String] size and, when set, the column's dotted path
299
+ def inspect
300
+ "#<#{self.class.name} #{num_bytes} bytes#{" for #{@column.dotted_path}" if @column}>"
301
+ end
302
+
303
+ private
304
+
305
+ # Header for this filter: split block algorithm, XXH64 hash, uncompressed
306
+ #
307
+ # @return [Format::BloomFilterHeader] the header
308
+ def header
309
+ Format::BloomFilterHeader.new(
310
+ num_bytes: num_bytes,
311
+ algorithm: Format::BloomFilterAlgorithm.new(block: Format::SplitBlockAlgorithm.new),
312
+ hash_function: Format::BloomFilterHash.new(xxhash: Format::XxHash.new),
313
+ compression: Format::BloomFilterCompression.new(uncompressed: Format::BloomFilterUncompressed.new)
314
+ )
315
+ end
316
+ end
317
+
318
+ class Reader
319
+ # Internal (used by reads with where:): the bloom filter of a column chunk. +column+ is a
320
+ # dotted path ("a.b"), an Array path or a Schema::Column. Returns a BloomFilter, or nil when
321
+ # the chunk has none (or one of an unknown kind).
322
+ #
323
+ # @param row_group_index [Integer] index of the row group
324
+ # @param column [String, Array<String, Symbol>, Symbol, Schema::Column] the leaf column
325
+ # @return [BloomFilter, nil] the filter, or nil when there is none or it is unsupported
326
+ # @raise [IndexError] when there is no such row group
327
+ # @raise [ArgumentError] when there is no such column
328
+ # @raise [FormatError] when the filter is truncated or its header cannot be decoded
329
+ def bloom_filter(row_group_index, column)
330
+ col = bloom_filter_column(column)
331
+ rg = row_groups.fetch(row_group_index) { raise IndexError, "No row group #{row_group_index}" }
332
+ meta = rg.columns.fetch(col.index).meta_data
333
+ offset = meta&.bloom_filter_offset
334
+ return nil unless offset
335
+ length = meta.bloom_filter_length
336
+ @io.seek(offset)
337
+ if length
338
+ buf = @io.read(length)
339
+ raise FormatError, "Truncated bloom filter" if buf.nil? || buf.bytesize < length
340
+ return BloomFilter.decode(buf, column: col)
341
+ end
342
+ # Without a length (older writers), read the header first, then the bitset it announces
343
+ head = @io.read(256) || "".b
344
+ reader = Thrift::Reader.new(head)
345
+ header = reader.read_struct(Format::BloomFilterHeader)
346
+ return nil unless BloomFilter.supported_header?(header)
347
+ @io.seek(offset + reader.pos)
348
+ bitset = @io.read(header.num_bytes)
349
+ raise FormatError, "Truncated bloom filter" if bitset.nil? || bitset.bytesize != header.num_bytes
350
+ BloomFilter.new(bitset: bitset, column: col)
351
+ rescue Thrift::Error => e
352
+ raise FormatError, "Corrupt bloom filter header for #{col.dotted_path}: #{e.message}"
353
+ end
354
+
355
+ private
356
+
357
+ # Resolves the +column+ argument of #bloom_filter to a leaf column
358
+ #
359
+ # @param column [String, Array<String, Symbol>, Symbol, Schema::Column] dotted path, path
360
+ # Array or column
361
+ # @return [Schema::Column] the column
362
+ # @raise [ArgumentError] when the schema has no such column
363
+ def bloom_filter_column(column)
364
+ return column if column.is_a?(Schema::Column)
365
+ col = schema.column(column.is_a?(Array) ? column.map(&:to_s) : column.to_s)
366
+ raise ArgumentError, "No column #{column.inspect}" unless col
367
+ col
368
+ end
369
+ end
370
+ end
@@ -17,8 +17,11 @@ module Herringbone
17
17
  # After this many values, give up on the dictionary if more than half of them are distinct
18
18
  CARDINALITY_CHECK_AT = 4096
19
19
 
20
+ # Whether String#append_as_bytes (Ruby 3.4+) is available for appending without re-encoding
20
21
  APPEND_AS_BYTES = "".respond_to?(:append_as_bytes)
21
22
 
23
+ # @param width [Integer, nil] byte length of each value for FIXED_LEN_BYTE_ARRAY, nil for BYTE_ARRAY
24
+ # @param dictionary [Boolean] start in dictionary mode; false appends raw bytes from the start
22
25
  def initialize(width: nil, dictionary: true)
23
26
  @width = width
24
27
  @bytes = String.new(encoding: Encoding::BINARY)
@@ -31,14 +34,21 @@ module Herringbone
31
34
  end
32
35
  end
33
36
 
37
+ # @return [Integer] number of values held
34
38
  def size
35
39
  @indices ? @indices.size : @count
36
40
  end
37
41
 
42
+ # @return [Boolean] true when no values are held
38
43
  def empty? = size.zero?
39
44
 
45
+ # @return [Boolean] true while still dictionary-encoding (not yet switched to raw bytes)
40
46
  def dictionary? = !@indices.nil?
41
47
 
48
+ # Appends a value, switching to raw bytes when the dictionary limits are exceeded.
49
+ # @param value [String] bytes of the value; for FIXED_LEN_BYTE_ARRAY it must be +width+ bytes
50
+ # long (not checked here)
51
+ # @return [ByteValues] self
42
52
  def <<(value)
43
53
  if @indices
44
54
  index = @dictionary[value]
@@ -61,16 +71,20 @@ module Herringbone
61
71
  end
62
72
 
63
73
  # Removes the last value
74
+ # @return [void]
64
75
  def pop
65
76
  truncate(size - 1) unless size.zero?
66
77
  end
67
78
 
68
79
  # Supports the `slice!(n..)` form used to roll back a failed row
80
+ # @param range [Range] endless range; values from +range.begin+ on are dropped
81
+ # @return [void]
69
82
  def slice!(range)
70
83
  truncate(range.begin)
71
84
  end
72
85
 
73
86
  # Approximate memory held, used to size row groups
87
+ # @return [Integer] estimated bytes
74
88
  def memory_bytes
75
89
  if @indices
76
90
  # Each distinct value is a String object plus a Hash entry
@@ -82,6 +96,11 @@ module Herringbone
82
96
 
83
97
  # Returns [:dictionary, values, indices] when the column is worth dictionary-encoding,
84
98
  # otherwise [:plain, values]
99
+ #
100
+ # Dictionary encoding is kept unless there are more than 16 values and more than about half
101
+ # of them are distinct.
102
+ # @return [Array(Symbol, Array<String>, Array<Integer>), Array(Symbol, Array<String>)]
103
+ # distinct values and an index per value, or all values in order
85
104
  def materialize
86
105
  if @indices
87
106
  keys = @dictionary.keys
@@ -93,6 +112,9 @@ module Herringbone
93
112
 
94
113
  private
95
114
 
115
+ # Appends +value+ in raw-bytes mode, recording its length for BYTE_ARRAY.
116
+ # @param value [String] bytes of the value, in any encoding
117
+ # @return [ByteValues] self
96
118
  def append_bytes(value)
97
119
  if value.encoding == Encoding::BINARY || value.ascii_only?
98
120
  @bytes << value
@@ -106,6 +128,8 @@ module Herringbone
106
128
  self
107
129
  end
108
130
 
131
+ # Leaves dictionary mode, re-appending every value held so far as raw bytes.
132
+ # @return [void]
109
133
  def switch_to_bytes
110
134
  keys = @dictionary.keys
111
135
  indices = @indices
@@ -113,6 +137,9 @@ module Herringbone
113
137
  indices.each { |i| append_bytes(keys[i]) }
114
138
  end
115
139
 
140
+ # Drops all values after the first +n+.
141
+ # @param n [Integer] number of values to keep
142
+ # @return [void]
116
143
  def truncate(n)
117
144
  if @indices
118
145
  @indices.slice!(n..)
@@ -128,6 +155,8 @@ module Herringbone
128
155
  end
129
156
  end
130
157
 
158
+ # Splits the raw bytes back into one binary String per value.
159
+ # @return [Array<String>]
131
160
  def strings
132
161
  if @lengths
133
162
  pos = 0
@@ -5,17 +5,26 @@ module Herringbone
5
5
  # Pure-Ruby LZ4: raw block format (Parquet LZ4_RAW), Hadoop-framed blocks
6
6
  # (Parquet's deprecated LZ4) and a decoder for the LZ4 frame format.
7
7
  module LZ4
8
+ # Raised for corrupt, truncated or unsupported LZ4 input.
8
9
  class Error < StandardError; end
9
10
 
11
+ # Shortest match a sequence can encode; the token's match length is stored minus this.
10
12
  MIN_MATCH = 4
11
13
  LAST_LITERALS = 5 # the last 5 bytes of a block are always literals
12
14
  MFLIMIT = 12 # the last match must start at least 12 bytes before the end
15
+ # Largest backward distance a 2-byte match offset can express.
13
16
  MAX_OFFSET = 65_535
17
+ # Size of the compressor's hash table, in bits (16384 entries).
14
18
  HASH_LOG = 14
19
+ # Right shift that keeps the top HASH_LOG bits of the 32-bit multiplicative hash.
15
20
  HASH_SHIFT = 32 - HASH_LOG
21
+ # Controls how fast the compressor skips ahead through input without matches (as in the reference).
16
22
  SKIP_STRENGTH = 6
23
+ # Magic number at the start of an LZ4 frame (little-endian).
17
24
  FRAME_MAGIC = 0x184D2204
25
+ # Hadoop block header: big-endian 32-bit uncompressed size and compressed size.
18
26
  HADOOP_PREFIX = 8
27
+ # Shorthand for Encoding::BINARY.
19
28
  BINARY = Encoding::BINARY
20
29
  # String#unpack1 accepts offset: since Ruby 3.1
21
30
  UNPACK_OFFSET = begin
@@ -28,6 +37,10 @@ module Herringbone
28
37
  module_function
29
38
 
30
39
  # Decompress a raw LZ4 block that must expand to exactly uncompressed_size bytes.
40
+ # @param input [String] raw LZ4 block
41
+ # @param uncompressed_size [Integer] exact decompressed size, from the page header
42
+ # @return [String] decompressed bytes in ASCII-8BIT
43
+ # @raise [Error] if the block is corrupt or does not decode to +uncompressed_size+ bytes
31
44
  def decompress_block(input, uncompressed_size)
32
45
  src = binary(input)
33
46
  out = String.new(capacity: uncompressed_size, encoding: BINARY)
@@ -40,6 +53,10 @@ module Herringbone
40
53
 
41
54
  # Parquet LZ4 (codec 5). Tries Hadoop framing, then the LZ4 frame format,
42
55
  # then a bare raw block (Arrow falls back hadoop -> raw; some writers emitted frames).
56
+ # @param input [String] compressed page data
57
+ # @param uncompressed_size [Integer] exact decompressed size, from the page header
58
+ # @return [String] decompressed bytes in ASCII-8BIT
59
+ # @raise [Error] if none of the three layouts decodes
43
60
  def decompress_hadoop(input, uncompressed_size)
44
61
  src = binary(input)
45
62
  result = try_hadoop(src, uncompressed_size)
@@ -58,6 +75,10 @@ module Herringbone
58
75
 
59
76
  # Decode LZ4 frame format data (one or more frames, skippable frames ignored).
60
77
  # Checksums are skipped, not verified.
78
+ # @param input [String] one or more concatenated LZ4 frames
79
+ # @param max_size [Integer] upper bound on the decompressed size
80
+ # @return [String] decompressed bytes in ASCII-8BIT, possibly shorter than +max_size+
81
+ # @raise [Error] on bad magic, truncation, unsupported frame features or output over +max_size+
61
82
  def decompress_frame(input, max_size)
62
83
  src = binary(input)
63
84
  n = src.bytesize
@@ -79,6 +100,8 @@ module Herringbone
79
100
  end
80
101
 
81
102
  # Compress into a single raw LZ4 block.
103
+ # @param input [String] bytes to compress
104
+ # @return [String] raw LZ4 block in ASCII-8BIT
82
105
  def compress_block(input)
83
106
  src = binary(input)
84
107
  n = src.bytesize
@@ -121,7 +144,7 @@ module Herringbone
121
144
  # Emit the sequence: token, literal run, offset, extended match length
122
145
  lit_len = ip - anchor
123
146
  ml = len - MIN_MATCH
124
- out << (((lit_len < 15 ? lit_len : 15) << 4) | (ml < 15 ? ml : 15))
147
+ out << ((((lit_len < 15) ? lit_len : 15) << 4) | ((ml < 15) ? ml : 15))
125
148
  write_length(out, lit_len - 15) if lit_len >= 15
126
149
  out << src.byteslice(anchor, lit_len) if lit_len > 0
127
150
  offset = ip - ref
@@ -144,6 +167,8 @@ module Herringbone
144
167
  end
145
168
 
146
169
  # Single Hadoop-framed block: [BE uncompressed size][BE compressed size][raw block]
170
+ # @param input [String] bytes to compress
171
+ # @return [String] Hadoop-framed LZ4 data in ASCII-8BIT
147
172
  def compress_hadoop(input)
148
173
  src = binary(input)
149
174
  block = compress_block(src)
@@ -152,18 +177,31 @@ module Herringbone
152
177
 
153
178
  # -- internals --
154
179
 
180
+ # @param str [String] input in any encoding
181
+ # @return [String] +str+ itself if already binary, otherwise a binary copy
155
182
  def binary(str)
156
- str.encoding == BINARY ? str : str.b
183
+ (str.encoding == BINARY) ? str : str.b
157
184
  end
158
185
 
186
+ # @param src [String] binary input
187
+ # @param i [Integer] byte offset
188
+ # @return [Integer] unsigned little-endian 32-bit value at +i+
159
189
  def le32(src, i)
160
190
  src.byteslice(i, 4).unpack1("V")
161
191
  end
162
192
 
193
+ # Same as le32 but via getbyte, for Rubies without unpack1(offset:).
194
+ # @param src [String] binary input
195
+ # @param i [Integer] byte offset
196
+ # @return [Integer] unsigned little-endian 32-bit value at +i+
163
197
  def u32(src, i)
164
198
  src.getbyte(i) | (src.getbyte(i + 1) << 8) | (src.getbyte(i + 2) << 16) | (src.getbyte(i + 3) << 24)
165
199
  end
166
200
 
201
+ # Appends the extension of a literal or match length: 255-bytes followed by the remainder.
202
+ # @param out [String] binary output buffer, appended to
203
+ # @param len [Integer] length minus the 15 already stored in the token
204
+ # @return [String] +out+
167
205
  def write_length(out, len)
168
206
  if len >= 255
169
207
  out << ("\xFF".b * (len / 255))
@@ -172,14 +210,27 @@ module Herringbone
172
210
  out << len
173
211
  end
174
212
 
213
+ # Appends the final, literals-only sequence that terminates every block.
214
+ # @param out [String] binary output buffer, appended to
215
+ # @param src [String] binary input
216
+ # @param anchor [Integer] offset of the first pending literal in +src+
217
+ # @param lit_len [Integer] number of trailing literal bytes (may be zero)
218
+ # @return [String, nil] +out+, or nil when there are no literals
175
219
  def emit_last_literals(out, src, anchor, lit_len)
176
- out << ((lit_len < 15 ? lit_len : 15) << 4)
220
+ out << (((lit_len < 15) ? lit_len : 15) << 4)
177
221
  write_length(out, lit_len - 15) if lit_len >= 15
178
222
  out << src.byteslice(anchor, lit_len) if lit_len > 0
179
223
  end
180
224
 
181
225
  # Decode one raw block from src[ip...iend], appending to out (which may already
182
226
  # hold earlier data that matches can reference). out may not grow beyond limit.
227
+ # @param src [String] binary input
228
+ # @param ip [Integer] offset of the block in +src+
229
+ # @param iend [Integer] offset just past the block
230
+ # @param out [String] binary output buffer, appended to
231
+ # @param limit [Integer] maximum total byte size of +out+
232
+ # @return [Integer] input offset where decoding stopped
233
+ # @raise [Error] if the block is truncated, has an invalid offset or overflows +limit+
183
234
  def decode_block(src, ip, iend, out, limit)
184
235
  while ip < iend
185
236
  token = src.getbyte(ip)
@@ -237,6 +288,9 @@ module Herringbone
237
288
  end
238
289
 
239
290
  # Arrow-compatible Hadoop frame parsing; returns nil if the data does not fit the framing.
291
+ # @param src [String] binary input
292
+ # @param uncompressed_size [Integer] exact decompressed size expected over all blocks
293
+ # @return [String, nil] decompressed bytes, or nil if +src+ is not valid Hadoop-framed LZ4
240
294
  def try_hadoop(src, uncompressed_size)
241
295
  n = src.bytesize
242
296
  return nil if n < HADOOP_PREFIX
@@ -262,6 +316,15 @@ module Herringbone
262
316
  out
263
317
  end
264
318
 
319
+ # Decodes the descriptor and data blocks of one LZ4 frame (after its magic number),
320
+ # skipping content size and checksums.
321
+ # @param src [String] binary input
322
+ # @param ip [Integer] offset of the frame descriptor (just past the magic)
323
+ # @param n [Integer] byte size of +src+
324
+ # @param out [String] binary output buffer, appended to
325
+ # @param limit [Integer] maximum total byte size of +out+
326
+ # @return [Integer] offset just past the frame
327
+ # @raise [Error] on truncation, an unsupported version or dictionary, or output over +limit+
265
328
  def decode_frame(src, ip, n, out, limit)
266
329
  raise Error, "Truncated LZ4 frame descriptor" if ip + 3 > n
267
330
  flg = src.getbyte(ip)