herringbone 0.1.0 → 0.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/README.md +146 -74
- data/bin/herringbone +38 -26
- data/lib/herringbone/active_record.rb +10 -13
- data/lib/herringbone/bloom_filter.rb +270 -0
- data/lib/herringbone/compression.rb +73 -16
- data/lib/herringbone/encodings/delta.rb +7 -7
- data/lib/herringbone/encodings/plain.rb +7 -7
- data/lib/herringbone/encodings/rle.rb +2 -2
- data/lib/herringbone/format.rb +53 -0
- data/lib/herringbone/inspector.rb +1388 -0
- data/lib/herringbone/reader/column_chunk_reader.rb +296 -0
- data/lib/herringbone/reader/column_cursor.rb +181 -0
- data/lib/herringbone/reader/page_stream.rb +339 -0
- data/lib/herringbone/reader/scan.rb +256 -0
- data/lib/herringbone/reader.rb +294 -272
- data/lib/herringbone/schema.rb +21 -76
- data/lib/herringbone/types.rb +4 -4
- data/lib/herringbone/version.rb +1 -1
- data/lib/herringbone/visualizer.rb +1096 -0
- data/lib/herringbone/writer.rb +258 -90
- data/lib/herringbone/xxhash.rb +319 -0
- data/lib/herringbone.rb +36 -9
- metadata +12 -32
|
@@ -0,0 +1,296 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Herringbone
|
|
4
|
+
class Reader
|
|
5
|
+
# Decodes the pages of one column chunk, one page at a time. Pages are read from the IO as
|
|
6
|
+
# they are needed (a page header, then its body), so memory use is bounded by the page size
|
|
7
|
+
# rather than the size of the chunk.
|
|
8
|
+
#
|
|
9
|
+
# reader = ColumnChunkReader.new(io, chunk, column)
|
|
10
|
+
# while (page = reader.next_page)
|
|
11
|
+
# defs, reps, values = page # levels are nil when the column's max level is 0
|
|
12
|
+
# end
|
|
13
|
+
#
|
|
14
|
+
# The declared total_compressed_size of the chunk is not relied on (some old writers
|
|
15
|
+
# under-report it): pages are read until the chunk's num_values have been seen.
|
|
16
|
+
class ColumnChunkReader
|
|
17
|
+
T = Format::Type
|
|
18
|
+
E = Format::Encoding
|
|
19
|
+
|
|
20
|
+
# Bytes read past a page body, so that the next page header usually needs no extra read
|
|
21
|
+
READ_AHEAD = 64 * 1024
|
|
22
|
+
HEADER_GUESS = 1024
|
|
23
|
+
MAX_HEADER = 64 * 1024 * 1024
|
|
24
|
+
|
|
25
|
+
attr_reader :column
|
|
26
|
+
|
|
27
|
+
# With +lazy+, values of data pages that are not dictionary-encoded are returned as
|
|
28
|
+
# physical values, and #page_converter is what still has to be applied to them. This keeps
|
|
29
|
+
# a decoded page small (Integers instead of Time or BigDecimal objects) when only a slice
|
|
30
|
+
# of it is needed at a time.
|
|
31
|
+
attr_reader :page_converter
|
|
32
|
+
|
|
33
|
+
def initialize(io, chunk, column, converter: column.converter, lazy: false)
|
|
34
|
+
@io = io
|
|
35
|
+
@chunk = chunk
|
|
36
|
+
@column = column
|
|
37
|
+
@meta = chunk.meta_data or raise UnsupportedError, "Column chunk without metadata (encrypted?)"
|
|
38
|
+
raise UnsupportedError, "Column chunks in external files are not supported" if chunk.file_path
|
|
39
|
+
@max_def = column.max_definition_level
|
|
40
|
+
@max_rep = column.max_repetition_level
|
|
41
|
+
@converter = converter
|
|
42
|
+
@lazy = lazy
|
|
43
|
+
start = @meta.data_page_offset
|
|
44
|
+
dict = @meta.dictionary_page_offset
|
|
45
|
+
# Some writers store 0 when there is no dictionary page
|
|
46
|
+
start = dict if dict && dict.positive? && dict < start
|
|
47
|
+
@pos = start
|
|
48
|
+
@start = start
|
|
49
|
+
@locations = nil # OffsetIndex page locations, when jumping between pages
|
|
50
|
+
@page_number = nil
|
|
51
|
+
@total = @meta.num_values
|
|
52
|
+
@seen = 0
|
|
53
|
+
@buf = "".b
|
|
54
|
+
@buf_pos = start
|
|
55
|
+
end
|
|
56
|
+
|
|
57
|
+
# All pages concatenated: [definition_levels, repetition_levels, values]
|
|
58
|
+
def read
|
|
59
|
+
defs = @max_def.positive? ? [] : nil
|
|
60
|
+
reps = @max_rep.positive? ? [] : nil
|
|
61
|
+
values = []
|
|
62
|
+
while (page = next_page)
|
|
63
|
+
d, r, v = page
|
|
64
|
+
defs&.concat(d)
|
|
65
|
+
reps&.concat(r)
|
|
66
|
+
values.concat(v)
|
|
67
|
+
end
|
|
68
|
+
[defs, reps, values]
|
|
69
|
+
end
|
|
70
|
+
|
|
71
|
+
# The next data page as [defs, reps, values], or nil after the last one
|
|
72
|
+
def next_page
|
|
73
|
+
page = next_stream or return nil
|
|
74
|
+
n = page.remaining
|
|
75
|
+
defs, reps = page.read_levels(n)
|
|
76
|
+
non_null = defs ? defs.count(@max_def) : n
|
|
77
|
+
values = page.read_values(non_null)
|
|
78
|
+
if (conv = page.converter)
|
|
79
|
+
if @lazy
|
|
80
|
+
@page_converter = conv
|
|
81
|
+
else
|
|
82
|
+
values.map!(&conv)
|
|
83
|
+
end
|
|
84
|
+
else
|
|
85
|
+
@page_converter = nil
|
|
86
|
+
end
|
|
87
|
+
[defs, reps, values]
|
|
88
|
+
end
|
|
89
|
+
|
|
90
|
+
# The next data page as a PageStream::Page that decodes its levels and values on demand,
|
|
91
|
+
# or nil after the last one. Values come out physical; apply Page#converter to them.
|
|
92
|
+
def next_stream
|
|
93
|
+
while more_pages?
|
|
94
|
+
header, body = read_page
|
|
95
|
+
case header.type
|
|
96
|
+
when Format::PageType::DICTIONARY_PAGE
|
|
97
|
+
read_dictionary(header, body)
|
|
98
|
+
when Format::PageType::DATA_PAGE
|
|
99
|
+
page = data_page_v1(header, body)
|
|
100
|
+
@seen += header.data_page_header.num_values
|
|
101
|
+
@page_number += 1 if @page_number
|
|
102
|
+
return page
|
|
103
|
+
when Format::PageType::DATA_PAGE_V2
|
|
104
|
+
page = data_page_v2(header, body)
|
|
105
|
+
@seen += header.data_page_header_v2.num_values
|
|
106
|
+
@page_number += 1 if @page_number
|
|
107
|
+
return page
|
|
108
|
+
end
|
|
109
|
+
end
|
|
110
|
+
nil
|
|
111
|
+
rescue Thrift::Error => e
|
|
112
|
+
raise FormatError, "Corrupt page header in #{@column.dotted_path}: #{e.message}"
|
|
113
|
+
end
|
|
114
|
+
|
|
115
|
+
# The chunk's data page locations from its OffsetIndex (enables #jump_to_page)
|
|
116
|
+
attr_accessor :locations
|
|
117
|
+
|
|
118
|
+
# Continues reading at data page +index+ of the OffsetIndex. The dictionary page (which
|
|
119
|
+
# the OffsetIndex does not list) is read first if it has not been yet.
|
|
120
|
+
def jump_to_page(index)
|
|
121
|
+
raise ArgumentError, "No OffsetIndex for #{@column.dotted_path}" unless @locations
|
|
122
|
+
load_dictionary
|
|
123
|
+
@pos = @locations.fetch(index).offset
|
|
124
|
+
@page_number = index
|
|
125
|
+
end
|
|
126
|
+
|
|
127
|
+
# Whether all of the chunk's values have been returned
|
|
128
|
+
def done? = @seen >= @total
|
|
129
|
+
|
|
130
|
+
def seen = @seen
|
|
131
|
+
def total = @total
|
|
132
|
+
|
|
133
|
+
private
|
|
134
|
+
|
|
135
|
+
def more_pages?
|
|
136
|
+
@page_number ? @page_number < @locations.size : @seen < @total
|
|
137
|
+
end
|
|
138
|
+
|
|
139
|
+
# Reads the dictionary page at the start of the chunk, if there is one
|
|
140
|
+
def load_dictionary
|
|
141
|
+
return if @dictionary || @dictionary_checked
|
|
142
|
+
@dictionary_checked = true
|
|
143
|
+
first_data = @locations.first&.offset
|
|
144
|
+
return if first_data.nil? || @start >= first_data
|
|
145
|
+
saved = @pos
|
|
146
|
+
@pos = @start
|
|
147
|
+
header, body = read_page
|
|
148
|
+
read_dictionary(header, body) if header.type == Format::PageType::DICTIONARY_PAGE
|
|
149
|
+
@pos = saved
|
|
150
|
+
end
|
|
151
|
+
|
|
152
|
+
# Reads the page header at @pos and the page body after it
|
|
153
|
+
def read_page
|
|
154
|
+
# With an OffsetIndex the page's size (header included) is known, so read exactly that
|
|
155
|
+
loc = @page_number && @locations[@page_number]
|
|
156
|
+
exact = loc && loc.offset == @pos && loc.compressed_page_size.positive?
|
|
157
|
+
want = exact ? loc.compressed_page_size : HEADER_GUESS
|
|
158
|
+
begin
|
|
159
|
+
buf, off = window(@pos, want)
|
|
160
|
+
header, hend = Format::PageHeader.decode(buf, off)
|
|
161
|
+
rescue Thrift::Error
|
|
162
|
+
# The header may be larger than guessed (long statistics) or be cut off by EOF
|
|
163
|
+
raise if buf.bytesize - off < want || want >= MAX_HEADER
|
|
164
|
+
want *= 4
|
|
165
|
+
retry
|
|
166
|
+
end
|
|
167
|
+
hlen = hend - off
|
|
168
|
+
size = header.compressed_page_size
|
|
169
|
+
raise FormatError, "Negative page size in #{@column.dotted_path}" if size.nil? || size.negative?
|
|
170
|
+
buf, off = window(@pos + hlen, size + (exact ? 0 : READ_AHEAD), size)
|
|
171
|
+
if buf.bytesize - off < size
|
|
172
|
+
raise FormatError, "Column #{@column.dotted_path}: page overruns the file (read #{@seen} of #{@total} values)"
|
|
173
|
+
end
|
|
174
|
+
body = off.zero? && buf.bytesize == size ? buf : buf.byteslice(off, size)
|
|
175
|
+
@pos += hlen + size
|
|
176
|
+
[header, body]
|
|
177
|
+
end
|
|
178
|
+
|
|
179
|
+
# Returns [buffer, offset] where buffer[offset..] holds at least +need+ bytes from file
|
|
180
|
+
# position +pos+ (fewer only at EOF), reading +len+ bytes when the buffer does not cover it.
|
|
181
|
+
def window(pos, len, need = len)
|
|
182
|
+
off = pos - @buf_pos
|
|
183
|
+
return [@buf, off] if off >= 0 && off + need <= @buf.bytesize
|
|
184
|
+
@io.seek(pos)
|
|
185
|
+
@buf = @io.read(len) || "".b
|
|
186
|
+
@buf.force_encoding(Encoding::BINARY)
|
|
187
|
+
@buf_pos = pos
|
|
188
|
+
[@buf, 0]
|
|
189
|
+
end
|
|
190
|
+
|
|
191
|
+
def decompress(body, size)
|
|
192
|
+
Compression.decompress(@meta.codec, body, size)
|
|
193
|
+
rescue UnsupportedError => e
|
|
194
|
+
raise e, "#{e.message} (column #{@column.dotted_path})"
|
|
195
|
+
end
|
|
196
|
+
|
|
197
|
+
def read_dictionary(header, body)
|
|
198
|
+
dh = header.dictionary_page_header
|
|
199
|
+
data = decompress(body, header.uncompressed_page_size)
|
|
200
|
+
vals, = Encodings::Plain.decode(data, 0, dh.num_values, @column.type, @column.type_length)
|
|
201
|
+
vals.map!(&@converter) if @converter
|
|
202
|
+
@dictionary = vals
|
|
203
|
+
end
|
|
204
|
+
|
|
205
|
+
def data_page_v1(header, body)
|
|
206
|
+
dh = header.data_page_header
|
|
207
|
+
n = dh.num_values
|
|
208
|
+
data = decompress(body, header.uncompressed_page_size)
|
|
209
|
+
pos = 0
|
|
210
|
+
reps = defs = nil
|
|
211
|
+
reps, pos = level_decoder(data, pos, dh.repetition_level_encoding, @max_rep, n) if @max_rep.positive?
|
|
212
|
+
defs, pos = level_decoder(data, pos, dh.definition_level_encoding, @max_def, n) if @max_def.positive?
|
|
213
|
+
values, conv = value_decoder(data, pos, dh.encoding)
|
|
214
|
+
PageStream::Page.new(n, defs, reps, values, conv)
|
|
215
|
+
end
|
|
216
|
+
|
|
217
|
+
def data_page_v2(header, body)
|
|
218
|
+
dh = header.data_page_header_v2
|
|
219
|
+
n = dh.num_values
|
|
220
|
+
rep_len = dh.repetition_levels_byte_length
|
|
221
|
+
def_len = dh.definition_levels_byte_length
|
|
222
|
+
reps = defs = nil
|
|
223
|
+
reps = PageStream::HybridDecoder.new(body, 0, rep_len, RLE_WIDTH[@max_rep]) if @max_rep.positive?
|
|
224
|
+
defs = PageStream::HybridDecoder.new(body, rep_len, rep_len + def_len, RLE_WIDTH[@max_def]) if @max_def.positive?
|
|
225
|
+
data = body.byteslice(rep_len + def_len, body.bytesize - rep_len - def_len)
|
|
226
|
+
if dh.is_compressed != false
|
|
227
|
+
data = decompress(data, header.uncompressed_page_size - rep_len - def_len)
|
|
228
|
+
end
|
|
229
|
+
values, conv = value_decoder(data, 0, dh.encoding)
|
|
230
|
+
PageStream::Page.new(n, defs, reps, values, conv)
|
|
231
|
+
end
|
|
232
|
+
|
|
233
|
+
RLE_WIDTH = Hash.new { |h, k| h[k] = k.bit_length }
|
|
234
|
+
|
|
235
|
+
# [decoder, position after the levels]
|
|
236
|
+
def level_decoder(data, pos, encoding, max, n)
|
|
237
|
+
width = RLE_WIDTH[max]
|
|
238
|
+
case encoding
|
|
239
|
+
when E::RLE
|
|
240
|
+
len = data.byteslice(pos, 4)&.unpack1("V") or raise FormatError, "Truncated levels"
|
|
241
|
+
start = pos + 4
|
|
242
|
+
[PageStream::HybridDecoder.new(data, start, start + len, width), start + len]
|
|
243
|
+
when E::BIT_PACKED
|
|
244
|
+
levels = Encodings::RLE.decode_legacy_bit_packed(data, pos, width, n)
|
|
245
|
+
[PageStream::ArrayDecoder.new(levels), pos + (n * width + 7) / 8]
|
|
246
|
+
else
|
|
247
|
+
raise UnsupportedError, "Unsupported level encoding #{E::NAMES[encoding] || encoding}"
|
|
248
|
+
end
|
|
249
|
+
end
|
|
250
|
+
|
|
251
|
+
FIXED_FORMATS = {
|
|
252
|
+
T::INT32 => ["l<", 4], T::INT64 => ["q<", 8], T::FLOAT => ["e", 4], T::DOUBLE => ["E", 8]
|
|
253
|
+
}.freeze
|
|
254
|
+
|
|
255
|
+
# [value decoder, converter still to apply to its values (nil for dictionary pages)]
|
|
256
|
+
def value_decoder(data, pos, encoding)
|
|
257
|
+
type = @column.type
|
|
258
|
+
decoder = case encoding
|
|
259
|
+
when E::PLAIN
|
|
260
|
+
case type
|
|
261
|
+
when T::BOOLEAN then PageStream::BooleanDecoder.new(data, pos)
|
|
262
|
+
when T::INT96 then PageStream::Int96Decoder.new(data, pos)
|
|
263
|
+
when T::BYTE_ARRAY then PageStream::ByteArrayDecoder.new(data, pos)
|
|
264
|
+
when T::FIXED_LEN_BYTE_ARRAY then PageStream::FixedBytesDecoder.new(data, pos, @column.type_length)
|
|
265
|
+
else PageStream::FixedDecoder.new(data, pos, *FIXED_FORMATS.fetch(type))
|
|
266
|
+
end
|
|
267
|
+
when E::PLAIN_DICTIONARY, E::RLE_DICTIONARY
|
|
268
|
+
raise FormatError, "Dictionary-encoded page without a dictionary in #{@column.dotted_path}" unless @dictionary
|
|
269
|
+
# Dictionary values are converted once, when the dictionary page is read
|
|
270
|
+
return [PageStream::DictionaryDecoder.new(data, pos, @dictionary, @column.dotted_path), nil]
|
|
271
|
+
when E::RLE
|
|
272
|
+
raise UnsupportedError, "RLE value encoding is only supported for BOOLEAN" unless type == T::BOOLEAN
|
|
273
|
+
PageStream::RleBooleanDecoder.new(data, pos)
|
|
274
|
+
when E::DELTA_BINARY_PACKED
|
|
275
|
+
bits = type == T::INT32 ? 32 : 64
|
|
276
|
+
PageStream::ArrayDecoder.new(Encodings::Delta.decode_binary_packed(data, pos, bits).first)
|
|
277
|
+
when E::DELTA_LENGTH_BYTE_ARRAY
|
|
278
|
+
PageStream::DeltaLengthDecoder.new(data, pos)
|
|
279
|
+
when E::DELTA_BYTE_ARRAY
|
|
280
|
+
PageStream::DeltaByteArrayDecoder.new(data, pos)
|
|
281
|
+
when E::BYTE_STREAM_SPLIT
|
|
282
|
+
width = case type
|
|
283
|
+
when T::INT32, T::FLOAT then 4
|
|
284
|
+
when T::INT64, T::DOUBLE then 8
|
|
285
|
+
when T::FIXED_LEN_BYTE_ARRAY then @column.type_length
|
|
286
|
+
else raise UnsupportedError, "BYTE_STREAM_SPLIT is not valid for #{T::NAMES[type]}"
|
|
287
|
+
end
|
|
288
|
+
PageStream::ByteStreamSplitDecoder.new(data, pos, width, type, @column.type_length)
|
|
289
|
+
else
|
|
290
|
+
raise UnsupportedError, "Unsupported encoding #{E::NAMES[encoding] || encoding}"
|
|
291
|
+
end
|
|
292
|
+
[decoder, @converter]
|
|
293
|
+
end
|
|
294
|
+
end
|
|
295
|
+
end
|
|
296
|
+
end
|
|
@@ -0,0 +1,181 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Herringbone
|
|
4
|
+
class Reader
|
|
5
|
+
# Walks the pages of one leaf column chunk and hands out the entries (levels and values)
|
|
6
|
+
# of the next +k+ rows. Pages are decoded incrementally (see PageStream), so only the
|
|
7
|
+
# entries of the requested rows (plus a small lookahead of levels for repeated columns)
|
|
8
|
+
# become Ruby objects. A row starts at an entry with repetition level 0 and may continue
|
|
9
|
+
# over any number of following pages.
|
|
10
|
+
class ColumnCursor
|
|
11
|
+
# Levels decoded ahead at a time when looking for row starts in repeated columns
|
|
12
|
+
LOOKAHEAD = 4096
|
|
13
|
+
|
|
14
|
+
def initialize(chunk_reader)
|
|
15
|
+
@src = chunk_reader
|
|
16
|
+
col = chunk_reader.column
|
|
17
|
+
@path = col.dotted_path
|
|
18
|
+
@max_def = col.max_definition_level
|
|
19
|
+
@repeated = col.max_repetition_level.positive?
|
|
20
|
+
@page = nil
|
|
21
|
+
# Repeated columns: levels decoded from the current page but not handed out yet
|
|
22
|
+
@bd = @br = nil
|
|
23
|
+
@bi = 0
|
|
24
|
+
@started = false
|
|
25
|
+
@row = 0 # rows handed out or skipped so far
|
|
26
|
+
@page_idx = -1 # index of the current data page within the chunk
|
|
27
|
+
end
|
|
28
|
+
|
|
29
|
+
# Rows handed out or skipped so far
|
|
30
|
+
attr_reader :row
|
|
31
|
+
|
|
32
|
+
# Moves forward to row +target+ of the chunk (0-based). With an OffsetIndex, pages before
|
|
33
|
+
# the one holding +target+ are not read at all; otherwise rows are skipped page by page,
|
|
34
|
+
# decoding levels but not building values.
|
|
35
|
+
def seek(target)
|
|
36
|
+
raise ArgumentError, "Cannot seek backwards (at row #{@row}, asked for #{target})" if target < @row
|
|
37
|
+
return if target == @row
|
|
38
|
+
locs = @src.locations
|
|
39
|
+
if locs
|
|
40
|
+
j = locs.bsearch_index { |loc| loc.first_row_index > target }
|
|
41
|
+
j = (j || locs.size) - 1
|
|
42
|
+
if j > @page_idx
|
|
43
|
+
@src.jump_to_page(j)
|
|
44
|
+
@page = nil
|
|
45
|
+
@bd = @br = nil
|
|
46
|
+
@bi = 0
|
|
47
|
+
@page_idx = j - 1
|
|
48
|
+
@row = locs[j].first_row_index
|
|
49
|
+
@started = true # pages listed in an OffsetIndex start at row boundaries
|
|
50
|
+
end
|
|
51
|
+
end
|
|
52
|
+
skip(target - @row)
|
|
53
|
+
end
|
|
54
|
+
|
|
55
|
+
# Moves past the next +k+ rows without building their values
|
|
56
|
+
def skip(k)
|
|
57
|
+
return if k <= 0
|
|
58
|
+
@repeated ? take_repeated(k, false) : take_flat(k, false)
|
|
59
|
+
@row += k
|
|
60
|
+
nil
|
|
61
|
+
end
|
|
62
|
+
|
|
63
|
+
# [definition_levels, repetition_levels, values] of the next +k+ rows. Levels are nil
|
|
64
|
+
# when the column's max level is 0.
|
|
65
|
+
def take(k)
|
|
66
|
+
pieces = @repeated ? take_repeated(k) : take_flat(k)
|
|
67
|
+
@row += k
|
|
68
|
+
return pieces.first if pieces.size == 1
|
|
69
|
+
defs = @max_def.positive? ? [] : nil
|
|
70
|
+
reps = @repeated ? [] : nil
|
|
71
|
+
vals = []
|
|
72
|
+
pieces.each do |d, r, v|
|
|
73
|
+
defs&.concat(d)
|
|
74
|
+
reps&.concat(r)
|
|
75
|
+
vals.concat(v)
|
|
76
|
+
end
|
|
77
|
+
[defs, reps, vals]
|
|
78
|
+
end
|
|
79
|
+
|
|
80
|
+
private
|
|
81
|
+
|
|
82
|
+
def take_flat(k, keep = true)
|
|
83
|
+
pieces = []
|
|
84
|
+
while k > 0
|
|
85
|
+
load_page! while @page.nil? || @page.remaining.zero?
|
|
86
|
+
t = @page.remaining
|
|
87
|
+
t = k if k < t
|
|
88
|
+
defs, = @page.read_levels(t)
|
|
89
|
+
nv = defs ? defs.count(@max_def) : t
|
|
90
|
+
if keep
|
|
91
|
+
pieces << [defs, nil, values(nv)]
|
|
92
|
+
else
|
|
93
|
+
@page.skip_values(nv)
|
|
94
|
+
end
|
|
95
|
+
k -= t
|
|
96
|
+
end
|
|
97
|
+
pieces
|
|
98
|
+
end
|
|
99
|
+
|
|
100
|
+
# Collects entries until +k+ rows have started and the next row start (or the end of the
|
|
101
|
+
# column) is reached. Values are read from a page before moving on to the next one.
|
|
102
|
+
def take_repeated(k, keep = true)
|
|
103
|
+
pieces = []
|
|
104
|
+
rows = 0
|
|
105
|
+
defs = @max_def.positive? ? [] : nil
|
|
106
|
+
reps = []
|
|
107
|
+
while true
|
|
108
|
+
if @bi >= @br.to_a.size
|
|
109
|
+
if @page && @page.remaining.positive?
|
|
110
|
+
@bd, @br = @page.read_levels(LOOKAHEAD)
|
|
111
|
+
@bi = 0
|
|
112
|
+
else
|
|
113
|
+
flush(pieces, defs, reps, keep)
|
|
114
|
+
defs = @max_def.positive? ? [] : nil
|
|
115
|
+
reps = []
|
|
116
|
+
break unless load_page
|
|
117
|
+
next
|
|
118
|
+
end
|
|
119
|
+
end
|
|
120
|
+
unless @started
|
|
121
|
+
raise FormatError, "Column #{@path}: first page does not start at a row boundary" unless @br[@bi].zero?
|
|
122
|
+
@started = true
|
|
123
|
+
end
|
|
124
|
+
br = @br
|
|
125
|
+
i = @bi
|
|
126
|
+
n = br.size
|
|
127
|
+
done = false
|
|
128
|
+
while i < n
|
|
129
|
+
if br[i].zero?
|
|
130
|
+
if rows == k
|
|
131
|
+
done = true
|
|
132
|
+
break
|
|
133
|
+
end
|
|
134
|
+
rows += 1
|
|
135
|
+
end
|
|
136
|
+
i += 1
|
|
137
|
+
end
|
|
138
|
+
if i > @bi
|
|
139
|
+
reps.concat(br[@bi, i - @bi])
|
|
140
|
+
defs&.concat(@bd[@bi, i - @bi])
|
|
141
|
+
@bi = i
|
|
142
|
+
end
|
|
143
|
+
break if done
|
|
144
|
+
end
|
|
145
|
+
flush(pieces, defs, reps, keep)
|
|
146
|
+
pieces
|
|
147
|
+
end
|
|
148
|
+
|
|
149
|
+
# Reads (or skips) the values belonging to the collected entries of the current page
|
|
150
|
+
def flush(pieces, defs, reps, keep)
|
|
151
|
+
return if reps.empty?
|
|
152
|
+
nv = defs ? defs.count(@max_def) : reps.size
|
|
153
|
+
if keep
|
|
154
|
+
pieces << [defs, reps, values(nv)]
|
|
155
|
+
else
|
|
156
|
+
@page.skip_values(nv)
|
|
157
|
+
end
|
|
158
|
+
end
|
|
159
|
+
|
|
160
|
+
def values(n)
|
|
161
|
+
vals = @page.read_values(n)
|
|
162
|
+
conv = @page.converter
|
|
163
|
+
conv ? vals.map!(&conv) : vals
|
|
164
|
+
end
|
|
165
|
+
|
|
166
|
+
def load_page!
|
|
167
|
+
return if load_page
|
|
168
|
+
raise FormatError, "Column #{@path}: ran out of pages after #{@src.seen} of #{@src.total} values"
|
|
169
|
+
end
|
|
170
|
+
|
|
171
|
+
# Moves to the next data page; false at the end of the chunk
|
|
172
|
+
def load_page
|
|
173
|
+
@page = @src.next_stream or return false
|
|
174
|
+
@page_idx += 1
|
|
175
|
+
@bd = @br = nil
|
|
176
|
+
@bi = 0
|
|
177
|
+
true
|
|
178
|
+
end
|
|
179
|
+
end
|
|
180
|
+
end
|
|
181
|
+
end
|