herringbone 0.1.0 → 0.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,296 @@
1
+ # frozen_string_literal: true
2
+
3
+ module Herringbone
4
+ class Reader
5
+ # Decodes the pages of one column chunk, one page at a time. Pages are read from the IO as
6
+ # they are needed (a page header, then its body), so memory use is bounded by the page size
7
+ # rather than the size of the chunk.
8
+ #
9
+ # reader = ColumnChunkReader.new(io, chunk, column)
10
+ # while (page = reader.next_page)
11
+ # defs, reps, values = page # levels are nil when the column's max level is 0
12
+ # end
13
+ #
14
+ # The declared total_compressed_size of the chunk is not relied on (some old writers
15
+ # under-report it): pages are read until the chunk's num_values have been seen.
16
+ class ColumnChunkReader
17
+ T = Format::Type
18
+ E = Format::Encoding
19
+
20
+ # Bytes read past a page body, so that the next page header usually needs no extra read
21
+ READ_AHEAD = 64 * 1024
22
+ HEADER_GUESS = 1024
23
+ MAX_HEADER = 64 * 1024 * 1024
24
+
25
+ attr_reader :column
26
+
27
+ # With +lazy+, values of data pages that are not dictionary-encoded are returned as
28
+ # physical values, and #page_converter is what still has to be applied to them. This keeps
29
+ # a decoded page small (Integers instead of Time or BigDecimal objects) when only a slice
30
+ # of it is needed at a time.
31
+ attr_reader :page_converter
32
+
33
+ def initialize(io, chunk, column, converter: column.converter, lazy: false)
34
+ @io = io
35
+ @chunk = chunk
36
+ @column = column
37
+ @meta = chunk.meta_data or raise UnsupportedError, "Column chunk without metadata (encrypted?)"
38
+ raise UnsupportedError, "Column chunks in external files are not supported" if chunk.file_path
39
+ @max_def = column.max_definition_level
40
+ @max_rep = column.max_repetition_level
41
+ @converter = converter
42
+ @lazy = lazy
43
+ start = @meta.data_page_offset
44
+ dict = @meta.dictionary_page_offset
45
+ # Some writers store 0 when there is no dictionary page
46
+ start = dict if dict && dict.positive? && dict < start
47
+ @pos = start
48
+ @start = start
49
+ @locations = nil # OffsetIndex page locations, when jumping between pages
50
+ @page_number = nil
51
+ @total = @meta.num_values
52
+ @seen = 0
53
+ @buf = "".b
54
+ @buf_pos = start
55
+ end
56
+
57
+ # All pages concatenated: [definition_levels, repetition_levels, values]
58
+ def read
59
+ defs = @max_def.positive? ? [] : nil
60
+ reps = @max_rep.positive? ? [] : nil
61
+ values = []
62
+ while (page = next_page)
63
+ d, r, v = page
64
+ defs&.concat(d)
65
+ reps&.concat(r)
66
+ values.concat(v)
67
+ end
68
+ [defs, reps, values]
69
+ end
70
+
71
+ # The next data page as [defs, reps, values], or nil after the last one
72
+ def next_page
73
+ page = next_stream or return nil
74
+ n = page.remaining
75
+ defs, reps = page.read_levels(n)
76
+ non_null = defs ? defs.count(@max_def) : n
77
+ values = page.read_values(non_null)
78
+ if (conv = page.converter)
79
+ if @lazy
80
+ @page_converter = conv
81
+ else
82
+ values.map!(&conv)
83
+ end
84
+ else
85
+ @page_converter = nil
86
+ end
87
+ [defs, reps, values]
88
+ end
89
+
90
+ # The next data page as a PageStream::Page that decodes its levels and values on demand,
91
+ # or nil after the last one. Values come out physical; apply Page#converter to them.
92
+ def next_stream
93
+ while more_pages?
94
+ header, body = read_page
95
+ case header.type
96
+ when Format::PageType::DICTIONARY_PAGE
97
+ read_dictionary(header, body)
98
+ when Format::PageType::DATA_PAGE
99
+ page = data_page_v1(header, body)
100
+ @seen += header.data_page_header.num_values
101
+ @page_number += 1 if @page_number
102
+ return page
103
+ when Format::PageType::DATA_PAGE_V2
104
+ page = data_page_v2(header, body)
105
+ @seen += header.data_page_header_v2.num_values
106
+ @page_number += 1 if @page_number
107
+ return page
108
+ end
109
+ end
110
+ nil
111
+ rescue Thrift::Error => e
112
+ raise FormatError, "Corrupt page header in #{@column.dotted_path}: #{e.message}"
113
+ end
114
+
115
+ # The chunk's data page locations from its OffsetIndex (enables #jump_to_page)
116
+ attr_accessor :locations
117
+
118
+ # Continues reading at data page +index+ of the OffsetIndex. The dictionary page (which
119
+ # the OffsetIndex does not list) is read first if it has not been yet.
120
+ def jump_to_page(index)
121
+ raise ArgumentError, "No OffsetIndex for #{@column.dotted_path}" unless @locations
122
+ load_dictionary
123
+ @pos = @locations.fetch(index).offset
124
+ @page_number = index
125
+ end
126
+
127
+ # Whether all of the chunk's values have been returned
128
+ def done? = @seen >= @total
129
+
130
+ def seen = @seen
131
+ def total = @total
132
+
133
+ private
134
+
135
+ def more_pages?
136
+ @page_number ? @page_number < @locations.size : @seen < @total
137
+ end
138
+
139
+ # Reads the dictionary page at the start of the chunk, if there is one
140
+ def load_dictionary
141
+ return if @dictionary || @dictionary_checked
142
+ @dictionary_checked = true
143
+ first_data = @locations.first&.offset
144
+ return if first_data.nil? || @start >= first_data
145
+ saved = @pos
146
+ @pos = @start
147
+ header, body = read_page
148
+ read_dictionary(header, body) if header.type == Format::PageType::DICTIONARY_PAGE
149
+ @pos = saved
150
+ end
151
+
152
+ # Reads the page header at @pos and the page body after it
153
+ def read_page
154
+ # With an OffsetIndex the page's size (header included) is known, so read exactly that
155
+ loc = @page_number && @locations[@page_number]
156
+ exact = loc && loc.offset == @pos && loc.compressed_page_size.positive?
157
+ want = exact ? loc.compressed_page_size : HEADER_GUESS
158
+ begin
159
+ buf, off = window(@pos, want)
160
+ header, hend = Format::PageHeader.decode(buf, off)
161
+ rescue Thrift::Error
162
+ # The header may be larger than guessed (long statistics) or be cut off by EOF
163
+ raise if buf.bytesize - off < want || want >= MAX_HEADER
164
+ want *= 4
165
+ retry
166
+ end
167
+ hlen = hend - off
168
+ size = header.compressed_page_size
169
+ raise FormatError, "Negative page size in #{@column.dotted_path}" if size.nil? || size.negative?
170
+ buf, off = window(@pos + hlen, size + (exact ? 0 : READ_AHEAD), size)
171
+ if buf.bytesize - off < size
172
+ raise FormatError, "Column #{@column.dotted_path}: page overruns the file (read #{@seen} of #{@total} values)"
173
+ end
174
+ body = off.zero? && buf.bytesize == size ? buf : buf.byteslice(off, size)
175
+ @pos += hlen + size
176
+ [header, body]
177
+ end
178
+
179
+ # Returns [buffer, offset] where buffer[offset..] holds at least +need+ bytes from file
180
+ # position +pos+ (fewer only at EOF), reading +len+ bytes when the buffer does not cover it.
181
+ def window(pos, len, need = len)
182
+ off = pos - @buf_pos
183
+ return [@buf, off] if off >= 0 && off + need <= @buf.bytesize
184
+ @io.seek(pos)
185
+ @buf = @io.read(len) || "".b
186
+ @buf.force_encoding(Encoding::BINARY)
187
+ @buf_pos = pos
188
+ [@buf, 0]
189
+ end
190
+
191
+ def decompress(body, size)
192
+ Compression.decompress(@meta.codec, body, size)
193
+ rescue UnsupportedError => e
194
+ raise e, "#{e.message} (column #{@column.dotted_path})"
195
+ end
196
+
197
+ def read_dictionary(header, body)
198
+ dh = header.dictionary_page_header
199
+ data = decompress(body, header.uncompressed_page_size)
200
+ vals, = Encodings::Plain.decode(data, 0, dh.num_values, @column.type, @column.type_length)
201
+ vals.map!(&@converter) if @converter
202
+ @dictionary = vals
203
+ end
204
+
205
+ def data_page_v1(header, body)
206
+ dh = header.data_page_header
207
+ n = dh.num_values
208
+ data = decompress(body, header.uncompressed_page_size)
209
+ pos = 0
210
+ reps = defs = nil
211
+ reps, pos = level_decoder(data, pos, dh.repetition_level_encoding, @max_rep, n) if @max_rep.positive?
212
+ defs, pos = level_decoder(data, pos, dh.definition_level_encoding, @max_def, n) if @max_def.positive?
213
+ values, conv = value_decoder(data, pos, dh.encoding)
214
+ PageStream::Page.new(n, defs, reps, values, conv)
215
+ end
216
+
217
+ def data_page_v2(header, body)
218
+ dh = header.data_page_header_v2
219
+ n = dh.num_values
220
+ rep_len = dh.repetition_levels_byte_length
221
+ def_len = dh.definition_levels_byte_length
222
+ reps = defs = nil
223
+ reps = PageStream::HybridDecoder.new(body, 0, rep_len, RLE_WIDTH[@max_rep]) if @max_rep.positive?
224
+ defs = PageStream::HybridDecoder.new(body, rep_len, rep_len + def_len, RLE_WIDTH[@max_def]) if @max_def.positive?
225
+ data = body.byteslice(rep_len + def_len, body.bytesize - rep_len - def_len)
226
+ if dh.is_compressed != false
227
+ data = decompress(data, header.uncompressed_page_size - rep_len - def_len)
228
+ end
229
+ values, conv = value_decoder(data, 0, dh.encoding)
230
+ PageStream::Page.new(n, defs, reps, values, conv)
231
+ end
232
+
233
+ RLE_WIDTH = Hash.new { |h, k| h[k] = k.bit_length }
234
+
235
+ # [decoder, position after the levels]
236
+ def level_decoder(data, pos, encoding, max, n)
237
+ width = RLE_WIDTH[max]
238
+ case encoding
239
+ when E::RLE
240
+ len = data.byteslice(pos, 4)&.unpack1("V") or raise FormatError, "Truncated levels"
241
+ start = pos + 4
242
+ [PageStream::HybridDecoder.new(data, start, start + len, width), start + len]
243
+ when E::BIT_PACKED
244
+ levels = Encodings::RLE.decode_legacy_bit_packed(data, pos, width, n)
245
+ [PageStream::ArrayDecoder.new(levels), pos + (n * width + 7) / 8]
246
+ else
247
+ raise UnsupportedError, "Unsupported level encoding #{E::NAMES[encoding] || encoding}"
248
+ end
249
+ end
250
+
251
+ FIXED_FORMATS = {
252
+ T::INT32 => ["l<", 4], T::INT64 => ["q<", 8], T::FLOAT => ["e", 4], T::DOUBLE => ["E", 8]
253
+ }.freeze
254
+
255
+ # [value decoder, converter still to apply to its values (nil for dictionary pages)]
256
+ def value_decoder(data, pos, encoding)
257
+ type = @column.type
258
+ decoder = case encoding
259
+ when E::PLAIN
260
+ case type
261
+ when T::BOOLEAN then PageStream::BooleanDecoder.new(data, pos)
262
+ when T::INT96 then PageStream::Int96Decoder.new(data, pos)
263
+ when T::BYTE_ARRAY then PageStream::ByteArrayDecoder.new(data, pos)
264
+ when T::FIXED_LEN_BYTE_ARRAY then PageStream::FixedBytesDecoder.new(data, pos, @column.type_length)
265
+ else PageStream::FixedDecoder.new(data, pos, *FIXED_FORMATS.fetch(type))
266
+ end
267
+ when E::PLAIN_DICTIONARY, E::RLE_DICTIONARY
268
+ raise FormatError, "Dictionary-encoded page without a dictionary in #{@column.dotted_path}" unless @dictionary
269
+ # Dictionary values are converted once, when the dictionary page is read
270
+ return [PageStream::DictionaryDecoder.new(data, pos, @dictionary, @column.dotted_path), nil]
271
+ when E::RLE
272
+ raise UnsupportedError, "RLE value encoding is only supported for BOOLEAN" unless type == T::BOOLEAN
273
+ PageStream::RleBooleanDecoder.new(data, pos)
274
+ when E::DELTA_BINARY_PACKED
275
+ bits = type == T::INT32 ? 32 : 64
276
+ PageStream::ArrayDecoder.new(Encodings::Delta.decode_binary_packed(data, pos, bits).first)
277
+ when E::DELTA_LENGTH_BYTE_ARRAY
278
+ PageStream::DeltaLengthDecoder.new(data, pos)
279
+ when E::DELTA_BYTE_ARRAY
280
+ PageStream::DeltaByteArrayDecoder.new(data, pos)
281
+ when E::BYTE_STREAM_SPLIT
282
+ width = case type
283
+ when T::INT32, T::FLOAT then 4
284
+ when T::INT64, T::DOUBLE then 8
285
+ when T::FIXED_LEN_BYTE_ARRAY then @column.type_length
286
+ else raise UnsupportedError, "BYTE_STREAM_SPLIT is not valid for #{T::NAMES[type]}"
287
+ end
288
+ PageStream::ByteStreamSplitDecoder.new(data, pos, width, type, @column.type_length)
289
+ else
290
+ raise UnsupportedError, "Unsupported encoding #{E::NAMES[encoding] || encoding}"
291
+ end
292
+ [decoder, @converter]
293
+ end
294
+ end
295
+ end
296
+ end
@@ -0,0 +1,181 @@
1
+ # frozen_string_literal: true
2
+
3
+ module Herringbone
4
+ class Reader
5
+ # Walks the pages of one leaf column chunk and hands out the entries (levels and values)
6
+ # of the next +k+ rows. Pages are decoded incrementally (see PageStream), so only the
7
+ # entries of the requested rows (plus a small lookahead of levels for repeated columns)
8
+ # become Ruby objects. A row starts at an entry with repetition level 0 and may continue
9
+ # over any number of following pages.
10
+ class ColumnCursor
11
+ # Levels decoded ahead at a time when looking for row starts in repeated columns
12
+ LOOKAHEAD = 4096
13
+
14
+ def initialize(chunk_reader)
15
+ @src = chunk_reader
16
+ col = chunk_reader.column
17
+ @path = col.dotted_path
18
+ @max_def = col.max_definition_level
19
+ @repeated = col.max_repetition_level.positive?
20
+ @page = nil
21
+ # Repeated columns: levels decoded from the current page but not handed out yet
22
+ @bd = @br = nil
23
+ @bi = 0
24
+ @started = false
25
+ @row = 0 # rows handed out or skipped so far
26
+ @page_idx = -1 # index of the current data page within the chunk
27
+ end
28
+
29
+ # Rows handed out or skipped so far
30
+ attr_reader :row
31
+
32
+ # Moves forward to row +target+ of the chunk (0-based). With an OffsetIndex, pages before
33
+ # the one holding +target+ are not read at all; otherwise rows are skipped page by page,
34
+ # decoding levels but not building values.
35
+ def seek(target)
36
+ raise ArgumentError, "Cannot seek backwards (at row #{@row}, asked for #{target})" if target < @row
37
+ return if target == @row
38
+ locs = @src.locations
39
+ if locs
40
+ j = locs.bsearch_index { |loc| loc.first_row_index > target }
41
+ j = (j || locs.size) - 1
42
+ if j > @page_idx
43
+ @src.jump_to_page(j)
44
+ @page = nil
45
+ @bd = @br = nil
46
+ @bi = 0
47
+ @page_idx = j - 1
48
+ @row = locs[j].first_row_index
49
+ @started = true # pages listed in an OffsetIndex start at row boundaries
50
+ end
51
+ end
52
+ skip(target - @row)
53
+ end
54
+
55
+ # Moves past the next +k+ rows without building their values
56
+ def skip(k)
57
+ return if k <= 0
58
+ @repeated ? take_repeated(k, false) : take_flat(k, false)
59
+ @row += k
60
+ nil
61
+ end
62
+
63
+ # [definition_levels, repetition_levels, values] of the next +k+ rows. Levels are nil
64
+ # when the column's max level is 0.
65
+ def take(k)
66
+ pieces = @repeated ? take_repeated(k) : take_flat(k)
67
+ @row += k
68
+ return pieces.first if pieces.size == 1
69
+ defs = @max_def.positive? ? [] : nil
70
+ reps = @repeated ? [] : nil
71
+ vals = []
72
+ pieces.each do |d, r, v|
73
+ defs&.concat(d)
74
+ reps&.concat(r)
75
+ vals.concat(v)
76
+ end
77
+ [defs, reps, vals]
78
+ end
79
+
80
+ private
81
+
82
+ def take_flat(k, keep = true)
83
+ pieces = []
84
+ while k > 0
85
+ load_page! while @page.nil? || @page.remaining.zero?
86
+ t = @page.remaining
87
+ t = k if k < t
88
+ defs, = @page.read_levels(t)
89
+ nv = defs ? defs.count(@max_def) : t
90
+ if keep
91
+ pieces << [defs, nil, values(nv)]
92
+ else
93
+ @page.skip_values(nv)
94
+ end
95
+ k -= t
96
+ end
97
+ pieces
98
+ end
99
+
100
+ # Collects entries until +k+ rows have started and the next row start (or the end of the
101
+ # column) is reached. Values are read from a page before moving on to the next one.
102
+ def take_repeated(k, keep = true)
103
+ pieces = []
104
+ rows = 0
105
+ defs = @max_def.positive? ? [] : nil
106
+ reps = []
107
+ while true
108
+ if @bi >= @br.to_a.size
109
+ if @page && @page.remaining.positive?
110
+ @bd, @br = @page.read_levels(LOOKAHEAD)
111
+ @bi = 0
112
+ else
113
+ flush(pieces, defs, reps, keep)
114
+ defs = @max_def.positive? ? [] : nil
115
+ reps = []
116
+ break unless load_page
117
+ next
118
+ end
119
+ end
120
+ unless @started
121
+ raise FormatError, "Column #{@path}: first page does not start at a row boundary" unless @br[@bi].zero?
122
+ @started = true
123
+ end
124
+ br = @br
125
+ i = @bi
126
+ n = br.size
127
+ done = false
128
+ while i < n
129
+ if br[i].zero?
130
+ if rows == k
131
+ done = true
132
+ break
133
+ end
134
+ rows += 1
135
+ end
136
+ i += 1
137
+ end
138
+ if i > @bi
139
+ reps.concat(br[@bi, i - @bi])
140
+ defs&.concat(@bd[@bi, i - @bi])
141
+ @bi = i
142
+ end
143
+ break if done
144
+ end
145
+ flush(pieces, defs, reps, keep)
146
+ pieces
147
+ end
148
+
149
+ # Reads (or skips) the values belonging to the collected entries of the current page
150
+ def flush(pieces, defs, reps, keep)
151
+ return if reps.empty?
152
+ nv = defs ? defs.count(@max_def) : reps.size
153
+ if keep
154
+ pieces << [defs, reps, values(nv)]
155
+ else
156
+ @page.skip_values(nv)
157
+ end
158
+ end
159
+
160
+ def values(n)
161
+ vals = @page.read_values(n)
162
+ conv = @page.converter
163
+ conv ? vals.map!(&conv) : vals
164
+ end
165
+
166
+ def load_page!
167
+ return if load_page
168
+ raise FormatError, "Column #{@path}: ran out of pages after #{@src.seen} of #{@src.total} values"
169
+ end
170
+
171
+ # Moves to the next data page; false at the end of the chunk
172
+ def load_page
173
+ @page = @src.next_stream or return false
174
+ @page_idx += 1
175
+ @bd = @br = nil
176
+ @bi = 0
177
+ true
178
+ end
179
+ end
180
+ end
181
+ end