herringbone 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +7 -0
- data/LICENSE +21 -0
- data/README.md +209 -0
- data/bin/herringbone +54 -0
- data/lib/herringbone/active_record.rb +131 -0
- data/lib/herringbone/byte_values.rb +144 -0
- data/lib/herringbone/codecs/lz4.rb +305 -0
- data/lib/herringbone/codecs/snappy.rb +375 -0
- data/lib/herringbone/compression.rb +75 -0
- data/lib/herringbone/encodings/delta.rb +153 -0
- data/lib/herringbone/encodings/plain.rb +87 -0
- data/lib/herringbone/encodings/rle.rb +203 -0
- data/lib/herringbone/format.rb +309 -0
- data/lib/herringbone/io_buffer_support.rb +25 -0
- data/lib/herringbone/reader.rb +469 -0
- data/lib/herringbone/schema.rb +551 -0
- data/lib/herringbone/thrift.rb +334 -0
- data/lib/herringbone/types.rb +454 -0
- data/lib/herringbone/version.rb +5 -0
- data/lib/herringbone/writer.rb +634 -0
- data/lib/herringbone.rb +45 -0
- metadata +104 -0
|
@@ -0,0 +1,305 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Herringbone
|
|
4
|
+
module Codecs
|
|
5
|
+
# Pure-Ruby LZ4: raw block format (Parquet LZ4_RAW), Hadoop-framed blocks
|
|
6
|
+
# (Parquet's deprecated LZ4) and a decoder for the LZ4 frame format.
|
|
7
|
+
module LZ4
|
|
8
|
+
class Error < StandardError; end
|
|
9
|
+
|
|
10
|
+
MIN_MATCH = 4
|
|
11
|
+
LAST_LITERALS = 5 # the last 5 bytes of a block are always literals
|
|
12
|
+
MFLIMIT = 12 # the last match must start at least 12 bytes before the end
|
|
13
|
+
MAX_OFFSET = 65_535
|
|
14
|
+
HASH_LOG = 14
|
|
15
|
+
HASH_SHIFT = 32 - HASH_LOG
|
|
16
|
+
SKIP_STRENGTH = 6
|
|
17
|
+
FRAME_MAGIC = 0x184D2204
|
|
18
|
+
HADOOP_PREFIX = 8
|
|
19
|
+
BINARY = Encoding::BINARY
|
|
20
|
+
# String#unpack1 accepts offset: since Ruby 3.1
|
|
21
|
+
UNPACK_OFFSET = begin
|
|
22
|
+
"\x00\x00\x00\x00".unpack1("V", offset: 0)
|
|
23
|
+
true
|
|
24
|
+
rescue ArgumentError
|
|
25
|
+
false
|
|
26
|
+
end
|
|
27
|
+
|
|
28
|
+
module_function
|
|
29
|
+
|
|
30
|
+
# Decompress a raw LZ4 block that must expand to exactly uncompressed_size bytes.
|
|
31
|
+
def decompress_block(input, uncompressed_size)
|
|
32
|
+
src = binary(input)
|
|
33
|
+
out = String.new(capacity: uncompressed_size, encoding: BINARY)
|
|
34
|
+
decode_block(src, 0, src.bytesize, out, uncompressed_size)
|
|
35
|
+
unless out.bytesize == uncompressed_size
|
|
36
|
+
raise Error, "LZ4 block decoded to #{out.bytesize} bytes, expected #{uncompressed_size}"
|
|
37
|
+
end
|
|
38
|
+
out
|
|
39
|
+
end
|
|
40
|
+
|
|
41
|
+
# Parquet LZ4 (codec 5). Tries Hadoop framing, then the LZ4 frame format,
|
|
42
|
+
# then a bare raw block (Arrow falls back hadoop -> raw; some writers emitted frames).
|
|
43
|
+
def decompress_hadoop(input, uncompressed_size)
|
|
44
|
+
src = binary(input)
|
|
45
|
+
result = try_hadoop(src, uncompressed_size)
|
|
46
|
+
return result if result
|
|
47
|
+
|
|
48
|
+
if src.bytesize >= 4 && src.unpack1("V") == FRAME_MAGIC
|
|
49
|
+
begin
|
|
50
|
+
out = decompress_frame(src, uncompressed_size)
|
|
51
|
+
return out if out.bytesize == uncompressed_size
|
|
52
|
+
rescue Error
|
|
53
|
+
# fall through to raw block
|
|
54
|
+
end
|
|
55
|
+
end
|
|
56
|
+
decompress_block(src, uncompressed_size)
|
|
57
|
+
end
|
|
58
|
+
|
|
59
|
+
# Decode LZ4 frame format data (one or more frames, skippable frames ignored).
|
|
60
|
+
# Checksums are skipped, not verified.
|
|
61
|
+
def decompress_frame(input, max_size)
|
|
62
|
+
src = binary(input)
|
|
63
|
+
n = src.bytesize
|
|
64
|
+
out = String.new(capacity: max_size, encoding: BINARY)
|
|
65
|
+
ip = 0
|
|
66
|
+
while ip < n
|
|
67
|
+
raise Error, "Truncated LZ4 frame header" if ip + 4 > n
|
|
68
|
+
magic = le32(src, ip)
|
|
69
|
+
if (magic & 0xFFFFFFF0) == 0x184D2A50 # skippable frame
|
|
70
|
+
raise Error, "Truncated skippable frame" if ip + 8 > n
|
|
71
|
+
ip += 8 + le32(src, ip + 4)
|
|
72
|
+
next
|
|
73
|
+
end
|
|
74
|
+
raise Error, "Bad LZ4 frame magic 0x#{magic.to_s(16)}" unless magic == FRAME_MAGIC
|
|
75
|
+
ip = decode_frame(src, ip + 4, n, out, max_size)
|
|
76
|
+
end
|
|
77
|
+
raise Error, "Truncated skippable frame" if ip > n
|
|
78
|
+
out
|
|
79
|
+
end
|
|
80
|
+
|
|
81
|
+
# Compress into a single raw LZ4 block.
|
|
82
|
+
def compress_block(input)
|
|
83
|
+
src = binary(input)
|
|
84
|
+
n = src.bytesize
|
|
85
|
+
out = String.new(capacity: n + (n / 255) + 16, encoding: BINARY)
|
|
86
|
+
anchor = 0
|
|
87
|
+
|
|
88
|
+
if n > MFLIMIT
|
|
89
|
+
table = Array.new(1 << HASH_LOG, -1)
|
|
90
|
+
mflimit = n - MFLIMIT
|
|
91
|
+
match_limit = n - LAST_LITERALS
|
|
92
|
+
ip = 0
|
|
93
|
+
while ip < mflimit
|
|
94
|
+
seq = UNPACK_OFFSET ? src.unpack1("V", offset: ip) : u32(src, ip)
|
|
95
|
+
h = ((seq * 40_503) & 0xFFFFFFFF) >> HASH_SHIFT
|
|
96
|
+
ref = table[h]
|
|
97
|
+
table[h] = ip
|
|
98
|
+
|
|
99
|
+
if ref < 0 || ip - ref > MAX_OFFSET ||
|
|
100
|
+
seq != (UNPACK_OFFSET ? src.unpack1("V", offset: ref) : u32(src, ref))
|
|
101
|
+
ip += 1 + ((ip - anchor) >> SKIP_STRENGTH)
|
|
102
|
+
next
|
|
103
|
+
end
|
|
104
|
+
|
|
105
|
+
# Extend the match backwards into pending literals
|
|
106
|
+
while ip > anchor && ref > 0 && src.getbyte(ip - 1) == src.getbyte(ref - 1)
|
|
107
|
+
ip -= 1
|
|
108
|
+
ref -= 1
|
|
109
|
+
end
|
|
110
|
+
|
|
111
|
+
# Extend forwards, 8 bytes at a time where possible
|
|
112
|
+
len = MIN_MATCH
|
|
113
|
+
if UNPACK_OFFSET
|
|
114
|
+
while ip + len + 8 <= match_limit &&
|
|
115
|
+
src.unpack1("Q<", offset: ip + len) == src.unpack1("Q<", offset: ref + len)
|
|
116
|
+
len += 8
|
|
117
|
+
end
|
|
118
|
+
end
|
|
119
|
+
len += 1 while ip + len < match_limit && src.getbyte(ip + len) == src.getbyte(ref + len)
|
|
120
|
+
|
|
121
|
+
# Emit the sequence: token, literal run, offset, extended match length
|
|
122
|
+
lit_len = ip - anchor
|
|
123
|
+
ml = len - MIN_MATCH
|
|
124
|
+
out << (((lit_len < 15 ? lit_len : 15) << 4) | (ml < 15 ? ml : 15))
|
|
125
|
+
write_length(out, lit_len - 15) if lit_len >= 15
|
|
126
|
+
out << src.byteslice(anchor, lit_len) if lit_len > 0
|
|
127
|
+
offset = ip - ref
|
|
128
|
+
out << (offset & 0xFF) << (offset >> 8)
|
|
129
|
+
write_length(out, ml - 15) if ml >= 15
|
|
130
|
+
|
|
131
|
+
ip += len
|
|
132
|
+
anchor = ip
|
|
133
|
+
break if ip >= mflimit
|
|
134
|
+
|
|
135
|
+
# Seed the table with a position inside the match for better follow-up matches
|
|
136
|
+
p2 = ip - 2
|
|
137
|
+
s2 = UNPACK_OFFSET ? src.unpack1("V", offset: p2) : u32(src, p2)
|
|
138
|
+
table[((s2 * 40_503) & 0xFFFFFFFF) >> HASH_SHIFT] = p2
|
|
139
|
+
end
|
|
140
|
+
end
|
|
141
|
+
|
|
142
|
+
emit_last_literals(out, src, anchor, n - anchor)
|
|
143
|
+
out
|
|
144
|
+
end
|
|
145
|
+
|
|
146
|
+
# Single Hadoop-framed block: [BE uncompressed size][BE compressed size][raw block]
|
|
147
|
+
def compress_hadoop(input)
|
|
148
|
+
src = binary(input)
|
|
149
|
+
block = compress_block(src)
|
|
150
|
+
[src.bytesize, block.bytesize].pack("NN") << block
|
|
151
|
+
end
|
|
152
|
+
|
|
153
|
+
# -- internals --
|
|
154
|
+
|
|
155
|
+
def binary(str)
|
|
156
|
+
str.encoding == BINARY ? str : str.b
|
|
157
|
+
end
|
|
158
|
+
|
|
159
|
+
def le32(src, i)
|
|
160
|
+
src.byteslice(i, 4).unpack1("V")
|
|
161
|
+
end
|
|
162
|
+
|
|
163
|
+
def u32(src, i)
|
|
164
|
+
src.getbyte(i) | (src.getbyte(i + 1) << 8) | (src.getbyte(i + 2) << 16) | (src.getbyte(i + 3) << 24)
|
|
165
|
+
end
|
|
166
|
+
|
|
167
|
+
def write_length(out, len)
|
|
168
|
+
if len >= 255
|
|
169
|
+
out << ("\xFF".b * (len / 255))
|
|
170
|
+
len %= 255
|
|
171
|
+
end
|
|
172
|
+
out << len
|
|
173
|
+
end
|
|
174
|
+
|
|
175
|
+
def emit_last_literals(out, src, anchor, lit_len)
|
|
176
|
+
out << ((lit_len < 15 ? lit_len : 15) << 4)
|
|
177
|
+
write_length(out, lit_len - 15) if lit_len >= 15
|
|
178
|
+
out << src.byteslice(anchor, lit_len) if lit_len > 0
|
|
179
|
+
end
|
|
180
|
+
|
|
181
|
+
# Decode one raw block from src[ip...iend], appending to out (which may already
|
|
182
|
+
# hold earlier data that matches can reference). out may not grow beyond limit.
|
|
183
|
+
def decode_block(src, ip, iend, out, limit)
|
|
184
|
+
while ip < iend
|
|
185
|
+
token = src.getbyte(ip)
|
|
186
|
+
ip += 1
|
|
187
|
+
|
|
188
|
+
lit = token >> 4
|
|
189
|
+
if lit == 15
|
|
190
|
+
while true
|
|
191
|
+
raise Error, "Truncated LZ4 literal length" if ip >= iend
|
|
192
|
+
b = src.getbyte(ip)
|
|
193
|
+
ip += 1
|
|
194
|
+
lit += b
|
|
195
|
+
break if b != 255
|
|
196
|
+
end
|
|
197
|
+
end
|
|
198
|
+
if lit > 0
|
|
199
|
+
raise Error, "LZ4 literals run past end of input" if ip + lit > iend
|
|
200
|
+
raise Error, "LZ4 output exceeds expected size" if out.bytesize + lit > limit
|
|
201
|
+
out << src.byteslice(ip, lit)
|
|
202
|
+
ip += lit
|
|
203
|
+
end
|
|
204
|
+
break if ip == iend # last sequence carries literals only
|
|
205
|
+
|
|
206
|
+
raise Error, "Truncated LZ4 match offset" if ip + 2 > iend
|
|
207
|
+
offset = src.getbyte(ip) | (src.getbyte(ip + 1) << 8)
|
|
208
|
+
ip += 2
|
|
209
|
+
olen = out.bytesize
|
|
210
|
+
raise Error, "Invalid LZ4 match offset #{offset}" if offset == 0 || offset > olen
|
|
211
|
+
|
|
212
|
+
mlen = token & 15
|
|
213
|
+
if mlen == 15
|
|
214
|
+
while true
|
|
215
|
+
raise Error, "Truncated LZ4 match length" if ip >= iend
|
|
216
|
+
b = src.getbyte(ip)
|
|
217
|
+
ip += 1
|
|
218
|
+
mlen += b
|
|
219
|
+
break if b != 255
|
|
220
|
+
end
|
|
221
|
+
end
|
|
222
|
+
mlen += MIN_MATCH
|
|
223
|
+
raise Error, "LZ4 output exceeds expected size" if olen + mlen > limit
|
|
224
|
+
|
|
225
|
+
pos = olen - offset
|
|
226
|
+
if mlen <= offset
|
|
227
|
+
out << out.byteslice(pos, mlen)
|
|
228
|
+
else
|
|
229
|
+
# Overlapping match: the pattern of length `offset` repeats
|
|
230
|
+
pattern = out.byteslice(pos, offset)
|
|
231
|
+
reps, rem = mlen.divmod(offset)
|
|
232
|
+
out << (pattern * reps)
|
|
233
|
+
out << pattern.byteslice(0, rem) if rem > 0
|
|
234
|
+
end
|
|
235
|
+
end
|
|
236
|
+
ip
|
|
237
|
+
end
|
|
238
|
+
|
|
239
|
+
# Arrow-compatible Hadoop frame parsing; returns nil if the data does not fit the framing.
|
|
240
|
+
def try_hadoop(src, uncompressed_size)
|
|
241
|
+
n = src.bytesize
|
|
242
|
+
return nil if n < HADOOP_PREFIX
|
|
243
|
+
|
|
244
|
+
out = String.new(capacity: uncompressed_size, encoding: BINARY)
|
|
245
|
+
ip = 0
|
|
246
|
+
while n - ip >= HADOOP_PREFIX
|
|
247
|
+
expected, csize = src.byteslice(ip, HADOOP_PREFIX).unpack("NN")
|
|
248
|
+
ip += HADOOP_PREFIX
|
|
249
|
+
return nil if csize > n - ip
|
|
250
|
+
return nil if out.bytesize + expected > uncompressed_size
|
|
251
|
+
|
|
252
|
+
target = out.bytesize + expected
|
|
253
|
+
begin
|
|
254
|
+
decode_block(src, ip, ip + csize, out, target)
|
|
255
|
+
rescue Error
|
|
256
|
+
return nil
|
|
257
|
+
end
|
|
258
|
+
return nil unless out.bytesize == target
|
|
259
|
+
ip += csize
|
|
260
|
+
end
|
|
261
|
+
return nil unless ip == n && out.bytesize == uncompressed_size
|
|
262
|
+
out
|
|
263
|
+
end
|
|
264
|
+
|
|
265
|
+
def decode_frame(src, ip, n, out, limit)
|
|
266
|
+
raise Error, "Truncated LZ4 frame descriptor" if ip + 3 > n
|
|
267
|
+
flg = src.getbyte(ip)
|
|
268
|
+
raise Error, "Unsupported LZ4 frame version" unless (flg >> 6) == 1
|
|
269
|
+
raise Error, "LZ4 frames with dictionaries are not supported" if flg & 0x01 != 0
|
|
270
|
+
block_checksum = flg & 0x10 != 0
|
|
271
|
+
content_size = flg & 0x08 != 0
|
|
272
|
+
content_checksum = flg & 0x04 != 0
|
|
273
|
+
ip += 2 # FLG, BD
|
|
274
|
+
ip += 8 if content_size
|
|
275
|
+
ip += 1 # header checksum
|
|
276
|
+
raise Error, "Truncated LZ4 frame descriptor" if ip > n
|
|
277
|
+
|
|
278
|
+
while true
|
|
279
|
+
raise Error, "Truncated LZ4 frame block header" if ip + 4 > n
|
|
280
|
+
bsize = le32(src, ip)
|
|
281
|
+
ip += 4
|
|
282
|
+
break if bsize == 0 # EndMark
|
|
283
|
+
|
|
284
|
+
uncompressed = bsize & 0x80000000 != 0
|
|
285
|
+
bsize &= 0x7FFFFFFF
|
|
286
|
+
raise Error, "LZ4 frame block runs past end of input" if ip + bsize > n
|
|
287
|
+
if uncompressed
|
|
288
|
+
raise Error, "LZ4 output exceeds expected size" if out.bytesize + bsize > limit
|
|
289
|
+
out << src.byteslice(ip, bsize)
|
|
290
|
+
else
|
|
291
|
+
decode_block(src, ip, ip + bsize, out, limit)
|
|
292
|
+
end
|
|
293
|
+
ip += bsize
|
|
294
|
+
ip += 4 if block_checksum
|
|
295
|
+
end
|
|
296
|
+
ip += 4 if content_checksum
|
|
297
|
+
raise Error, "Truncated LZ4 frame" if ip > n
|
|
298
|
+
ip
|
|
299
|
+
end
|
|
300
|
+
|
|
301
|
+
private_class_method :binary, :le32, :u32, :write_length, :emit_last_literals,
|
|
302
|
+
:decode_block, :try_hadoop, :decode_frame
|
|
303
|
+
end
|
|
304
|
+
end
|
|
305
|
+
end
|
|
@@ -0,0 +1,375 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Herringbone
|
|
4
|
+
module Codecs
|
|
5
|
+
# Pure-Ruby implementation of the raw Snappy block format (as used by Parquet),
|
|
6
|
+
# see https://github.com/google/snappy/blob/main/format_description.txt
|
|
7
|
+
#
|
|
8
|
+
# 32-bit loads are done with four getbyte calls rather than unpack1(offset:) to stay
|
|
9
|
+
# compatible with Ruby 3.0 - the speed difference on MRI is marginal.
|
|
10
|
+
module Snappy
|
|
11
|
+
class Error < StandardError; end
|
|
12
|
+
|
|
13
|
+
BLOCK_SIZE = 1 << 16
|
|
14
|
+
HASH_BITS = 14
|
|
15
|
+
HASH_SHIFT = 32 - HASH_BITS
|
|
16
|
+
HASH_MUL = 0x1e35a7bd
|
|
17
|
+
INPUT_MARGIN = 15 # bytes at the block end never searched for matches, as in the reference
|
|
18
|
+
MAX_UNCOMPRESSED = (1 << 32) - 1
|
|
19
|
+
|
|
20
|
+
module_function
|
|
21
|
+
|
|
22
|
+
# @param input [String] raw snappy block
|
|
23
|
+
# @return [String] decompressed bytes in ASCII-8BIT
|
|
24
|
+
def decompress(input)
|
|
25
|
+
src = input.encoding == Encoding::BINARY ? input : input.b
|
|
26
|
+
return decompress_io_buffer(src) if IOBufferSupport::AVAILABLE
|
|
27
|
+
decompress_string(src)
|
|
28
|
+
end
|
|
29
|
+
|
|
30
|
+
# Decompresses into a preallocated IO::Buffer: copies do not allocate intermediate Strings.
|
|
31
|
+
def decompress_io_buffer(src)
|
|
32
|
+
n = src.bytesize
|
|
33
|
+
expected, pos = read_varint(src, n)
|
|
34
|
+
return "".b if expected.zero? && pos == n
|
|
35
|
+
out = IO::Buffer.new([expected, 1].max)
|
|
36
|
+
inbuf = IO::Buffer.for(src)
|
|
37
|
+
op = 0
|
|
38
|
+
while pos < n
|
|
39
|
+
tag = src.getbyte(pos)
|
|
40
|
+
pos += 1
|
|
41
|
+
kind = tag & 3
|
|
42
|
+
if kind == 0
|
|
43
|
+
len = tag >> 2
|
|
44
|
+
if len >= 60
|
|
45
|
+
nbytes = len - 59
|
|
46
|
+
raise Error, "truncated literal length" if pos + nbytes > n
|
|
47
|
+
len = 0
|
|
48
|
+
nbytes.times { |i| len |= src.getbyte(pos + i) << (8 * i) }
|
|
49
|
+
pos += nbytes
|
|
50
|
+
end
|
|
51
|
+
len += 1
|
|
52
|
+
raise Error, "literal overruns input" if pos + len > n
|
|
53
|
+
raise Error, "output exceeds declared length" if op + len > expected
|
|
54
|
+
out.copy(inbuf, op, len, pos)
|
|
55
|
+
op += len
|
|
56
|
+
pos += len
|
|
57
|
+
next
|
|
58
|
+
elsif kind == 1
|
|
59
|
+
raise Error, "truncated copy" if pos >= n
|
|
60
|
+
len = ((tag >> 2) & 7) + 4
|
|
61
|
+
offset = ((tag >> 5) << 8) | src.getbyte(pos)
|
|
62
|
+
pos += 1
|
|
63
|
+
elsif kind == 2
|
|
64
|
+
raise Error, "truncated copy" if pos + 2 > n
|
|
65
|
+
len = (tag >> 2) + 1
|
|
66
|
+
offset = src.getbyte(pos) | (src.getbyte(pos + 1) << 8)
|
|
67
|
+
pos += 2
|
|
68
|
+
else
|
|
69
|
+
raise Error, "truncated copy" if pos + 4 > n
|
|
70
|
+
len = (tag >> 2) + 1
|
|
71
|
+
offset = src.getbyte(pos) | (src.getbyte(pos + 1) << 8) |
|
|
72
|
+
(src.getbyte(pos + 2) << 16) | (src.getbyte(pos + 3) << 24)
|
|
73
|
+
pos += 4
|
|
74
|
+
end
|
|
75
|
+
raise Error, "invalid copy offset #{offset}" if offset == 0 || offset > op
|
|
76
|
+
raise Error, "output exceeds declared length" if op + len > expected
|
|
77
|
+
if offset >= len
|
|
78
|
+
out.copy(out, op, len, op - offset)
|
|
79
|
+
else
|
|
80
|
+
# Overlapping copy of a pattern with period `offset`: each copy doubles the
|
|
81
|
+
# replicated region, and never overlaps its own source.
|
|
82
|
+
from = op - offset
|
|
83
|
+
done = 0
|
|
84
|
+
step = offset
|
|
85
|
+
while done < len
|
|
86
|
+
chunk = len - done
|
|
87
|
+
chunk = step if chunk > step
|
|
88
|
+
out.copy(out, op + done, chunk, from)
|
|
89
|
+
done += chunk
|
|
90
|
+
step <<= 1
|
|
91
|
+
end
|
|
92
|
+
end
|
|
93
|
+
op += len
|
|
94
|
+
end
|
|
95
|
+
raise Error, "decompressed #{op} bytes, expected #{expected}" if op != expected
|
|
96
|
+
out.get_string(0, expected)
|
|
97
|
+
ensure
|
|
98
|
+
inbuf&.free
|
|
99
|
+
end
|
|
100
|
+
|
|
101
|
+
def decompress_string(src)
|
|
102
|
+
n = src.bytesize
|
|
103
|
+
expected, pos = read_varint(src, n)
|
|
104
|
+
out = String.new(capacity: expected, encoding: Encoding::BINARY)
|
|
105
|
+
|
|
106
|
+
while pos < n
|
|
107
|
+
tag = src.getbyte(pos)
|
|
108
|
+
pos += 1
|
|
109
|
+
|
|
110
|
+
case tag & 3
|
|
111
|
+
when 0 # literal
|
|
112
|
+
len = tag >> 2
|
|
113
|
+
if len >= 60
|
|
114
|
+
nbytes = len - 59
|
|
115
|
+
raise Error, "truncated literal length" if pos + nbytes > n
|
|
116
|
+
len = 0
|
|
117
|
+
nbytes.times { |i| len |= src.getbyte(pos + i) << (8 * i) }
|
|
118
|
+
pos += nbytes
|
|
119
|
+
end
|
|
120
|
+
len += 1
|
|
121
|
+
raise Error, "literal overruns input" if pos + len > n
|
|
122
|
+
raise Error, "output exceeds declared length" if out.bytesize + len > expected
|
|
123
|
+
out << src.byteslice(pos, len)
|
|
124
|
+
pos += len
|
|
125
|
+
next
|
|
126
|
+
when 1 # copy, 1-byte offset
|
|
127
|
+
raise Error, "truncated copy" if pos >= n
|
|
128
|
+
len = ((tag >> 2) & 7) + 4
|
|
129
|
+
offset = ((tag >> 5) << 8) | src.getbyte(pos)
|
|
130
|
+
pos += 1
|
|
131
|
+
when 2 # copy, 2-byte offset
|
|
132
|
+
raise Error, "truncated copy" if pos + 2 > n
|
|
133
|
+
len = (tag >> 2) + 1
|
|
134
|
+
offset = src.getbyte(pos) | (src.getbyte(pos + 1) << 8)
|
|
135
|
+
pos += 2
|
|
136
|
+
else # copy, 4-byte offset
|
|
137
|
+
raise Error, "truncated copy" if pos + 4 > n
|
|
138
|
+
len = (tag >> 2) + 1
|
|
139
|
+
offset = src.getbyte(pos) | (src.getbyte(pos + 1) << 8) |
|
|
140
|
+
(src.getbyte(pos + 2) << 16) | (src.getbyte(pos + 3) << 24)
|
|
141
|
+
pos += 4
|
|
142
|
+
end
|
|
143
|
+
|
|
144
|
+
produced = out.bytesize
|
|
145
|
+
raise Error, "invalid copy offset #{offset}" if offset == 0 || offset > produced
|
|
146
|
+
raise Error, "output exceeds declared length" if produced + len > expected
|
|
147
|
+
|
|
148
|
+
if offset >= len
|
|
149
|
+
out << out.byteslice(produced - offset, len)
|
|
150
|
+
else
|
|
151
|
+
# Overlapping copy: the source is a pattern with period `offset`. Double it
|
|
152
|
+
# until it covers `len` (keeps whole periods, so it stays in phase).
|
|
153
|
+
pattern = out.byteslice(produced - offset, offset)
|
|
154
|
+
pattern << pattern while pattern.bytesize < len
|
|
155
|
+
out << pattern.byteslice(0, len)
|
|
156
|
+
end
|
|
157
|
+
end
|
|
158
|
+
|
|
159
|
+
raise Error, "decompressed #{out.bytesize} bytes, expected #{expected}" if out.bytesize != expected
|
|
160
|
+
out
|
|
161
|
+
end
|
|
162
|
+
|
|
163
|
+
# @param input [String] bytes to compress
|
|
164
|
+
# @return [String] raw snappy block in ASCII-8BIT
|
|
165
|
+
def compress(input)
|
|
166
|
+
src = input.encoding == Encoding::BINARY ? input : input.b
|
|
167
|
+
n = src.bytesize
|
|
168
|
+
raise Error, "input too large for snappy" if n > MAX_UNCOMPRESSED
|
|
169
|
+
|
|
170
|
+
out = String.new(capacity: 32 + n + n / 6, encoding: Encoding::BINARY)
|
|
171
|
+
write_varint(out, n)
|
|
172
|
+
table = Array.new(1 << HASH_BITS, 0)
|
|
173
|
+
start = 0
|
|
174
|
+
while start < n
|
|
175
|
+
len = n - start
|
|
176
|
+
len = BLOCK_SIZE if len > BLOCK_SIZE
|
|
177
|
+
table.fill(0)
|
|
178
|
+
compress_block(src, start, len, out, table)
|
|
179
|
+
start += len
|
|
180
|
+
end
|
|
181
|
+
out
|
|
182
|
+
end
|
|
183
|
+
|
|
184
|
+
def read_varint(src, n)
|
|
185
|
+
value = 0
|
|
186
|
+
shift = 0
|
|
187
|
+
pos = 0
|
|
188
|
+
while true
|
|
189
|
+
raise Error, "truncated length preamble" if pos >= n
|
|
190
|
+
byte = src.getbyte(pos)
|
|
191
|
+
pos += 1
|
|
192
|
+
value |= (byte & 0x7f) << shift
|
|
193
|
+
break if byte < 0x80
|
|
194
|
+
shift += 7
|
|
195
|
+
raise Error, "length preamble too long" if shift > 28
|
|
196
|
+
end
|
|
197
|
+
raise Error, "declared length too large" if value > MAX_UNCOMPRESSED
|
|
198
|
+
[value, pos]
|
|
199
|
+
end
|
|
200
|
+
|
|
201
|
+
def write_varint(out, value)
|
|
202
|
+
while value >= 0x80
|
|
203
|
+
out << ((value & 0x7f) | 0x80)
|
|
204
|
+
value >>= 7
|
|
205
|
+
end
|
|
206
|
+
out << value
|
|
207
|
+
end
|
|
208
|
+
|
|
209
|
+
# Mirrors CompressFragment from the reference implementation. Table entries are
|
|
210
|
+
# positions relative to `base`; matches never cross the block boundary.
|
|
211
|
+
# The 4-byte little-endian word at every position of the block is unpacked up front
|
|
212
|
+
# (in C, via String#unpack) so hashing and match checks are single Array lookups.
|
|
213
|
+
def compress_block(src, base, len, out, table)
|
|
214
|
+
ip_end = base + len
|
|
215
|
+
next_emit = base
|
|
216
|
+
|
|
217
|
+
if len >= INPUT_MARGIN
|
|
218
|
+
words = block_words(src, base, len)
|
|
219
|
+
ip_limit = ip_end - INPUT_MARGIN
|
|
220
|
+
ip = base + 1
|
|
221
|
+
next_hash = ((words[1][0] * HASH_MUL) & 0xffffffff) >> HASH_SHIFT
|
|
222
|
+
done = false
|
|
223
|
+
|
|
224
|
+
until done
|
|
225
|
+
# Scan for a 4-byte match, skipping faster the longer we find nothing.
|
|
226
|
+
skip = 32
|
|
227
|
+
next_ip = ip
|
|
228
|
+
candidate = nil
|
|
229
|
+
while true
|
|
230
|
+
ip = next_ip
|
|
231
|
+
h = next_hash
|
|
232
|
+
next_ip = ip + (skip >> 5)
|
|
233
|
+
skip += 1
|
|
234
|
+
if next_ip > ip_limit
|
|
235
|
+
done = true
|
|
236
|
+
break
|
|
237
|
+
end
|
|
238
|
+
rel_next = next_ip - base
|
|
239
|
+
next_hash = ((words[rel_next & 3][rel_next >> 2] * HASH_MUL) & 0xffffffff) >> HASH_SHIFT
|
|
240
|
+
rel = table[h]
|
|
241
|
+
candidate = base + rel
|
|
242
|
+
table[h] = ip - base
|
|
243
|
+
rel_ip = ip - base
|
|
244
|
+
break if words[rel_ip & 3][rel_ip >> 2] == words[rel & 3][rel >> 2]
|
|
245
|
+
end
|
|
246
|
+
break if done
|
|
247
|
+
|
|
248
|
+
emit_literal(out, src, next_emit, ip - next_emit)
|
|
249
|
+
|
|
250
|
+
# Emit copies for as long as the position right after a copy matches again.
|
|
251
|
+
while true
|
|
252
|
+
matched = 4 + match_length_words(words, src, base, candidate + 4, ip + 4, ip_end)
|
|
253
|
+
emit_copy(out, ip - candidate, matched)
|
|
254
|
+
ip += matched
|
|
255
|
+
next_emit = ip
|
|
256
|
+
if ip >= ip_limit
|
|
257
|
+
done = true
|
|
258
|
+
break
|
|
259
|
+
end
|
|
260
|
+
rel_ip = ip - base
|
|
261
|
+
prev = rel_ip - 1
|
|
262
|
+
table[((words[prev & 3][prev >> 2] * HASH_MUL) & 0xffffffff) >> HASH_SHIFT] = prev
|
|
263
|
+
cur = words[rel_ip & 3][rel_ip >> 2]
|
|
264
|
+
h = ((cur * HASH_MUL) & 0xffffffff) >> HASH_SHIFT
|
|
265
|
+
rel = table[h]
|
|
266
|
+
candidate = base + rel
|
|
267
|
+
table[h] = rel_ip
|
|
268
|
+
break unless cur == words[rel & 3][rel >> 2]
|
|
269
|
+
end
|
|
270
|
+
break if done
|
|
271
|
+
|
|
272
|
+
ip += 1
|
|
273
|
+
rel_ip = ip - base
|
|
274
|
+
next_hash = ((words[rel_ip & 3][rel_ip >> 2] * HASH_MUL) & 0xffffffff) >> HASH_SHIFT
|
|
275
|
+
end
|
|
276
|
+
end
|
|
277
|
+
|
|
278
|
+
emit_literal(out, src, next_emit, ip_end - next_emit) if next_emit < ip_end
|
|
279
|
+
end
|
|
280
|
+
|
|
281
|
+
# words[k][j] is the 4-byte little-endian value at base + 4 * j + k, so the word at
|
|
282
|
+
# relative position i is words[i & 3][i >> 2]
|
|
283
|
+
def block_words(src, base, len)
|
|
284
|
+
Array.new(4) { |k| src.byteslice(base + k, len - k).unpack("V*") }
|
|
285
|
+
end
|
|
286
|
+
|
|
287
|
+
# Like match_length, but compares 4 bytes at a time using the unpacked block words,
|
|
288
|
+
# which avoids allocating substrings for the (common) short matches.
|
|
289
|
+
def match_length_words(words, src, base, s1, s2, limit)
|
|
290
|
+
start = s2
|
|
291
|
+
r1 = s1 - base
|
|
292
|
+
r2 = s2 - base
|
|
293
|
+
last_word = limit - base - 4
|
|
294
|
+
while r2 <= last_word && words[r1 & 3][r1 >> 2] == words[r2 & 3][r2 >> 2]
|
|
295
|
+
r1 += 4
|
|
296
|
+
r2 += 4
|
|
297
|
+
return (base + r2 - start) + match_length(src, base + r1, base + r2, limit) if r2 - (start - base) >= 64
|
|
298
|
+
end
|
|
299
|
+
s1 = base + r1
|
|
300
|
+
s2 = base + r2
|
|
301
|
+
while s2 < limit && src.getbyte(s1) == src.getbyte(s2)
|
|
302
|
+
s1 += 1
|
|
303
|
+
s2 += 1
|
|
304
|
+
end
|
|
305
|
+
s2 - start
|
|
306
|
+
end
|
|
307
|
+
|
|
308
|
+
# Number of equal bytes at s1 and s2 (s1 < s2), not reading past limit. Gallops
|
|
309
|
+
# with byteslice comparisons (memcmp) to avoid per-byte loops on long matches.
|
|
310
|
+
def match_length(src, s1, s2, limit)
|
|
311
|
+
start = s2
|
|
312
|
+
return 0 if s2 >= limit || src.getbyte(s1) != src.getbyte(s2)
|
|
313
|
+
|
|
314
|
+
step = 8
|
|
315
|
+
growing = true
|
|
316
|
+
while step > 0
|
|
317
|
+
if s2 + step <= limit && src.byteslice(s1, step) == src.byteslice(s2, step)
|
|
318
|
+
s1 += step
|
|
319
|
+
s2 += step
|
|
320
|
+
if growing
|
|
321
|
+
step <<= 1 if step < 4096
|
|
322
|
+
else
|
|
323
|
+
step >>= 1
|
|
324
|
+
end
|
|
325
|
+
else
|
|
326
|
+
growing = false
|
|
327
|
+
step >>= 1
|
|
328
|
+
end
|
|
329
|
+
end
|
|
330
|
+
s2 - start
|
|
331
|
+
end
|
|
332
|
+
|
|
333
|
+
def emit_literal(out, src, pos, len)
|
|
334
|
+
return if len == 0
|
|
335
|
+
n = len - 1
|
|
336
|
+
if n < 60
|
|
337
|
+
out << (n << 2)
|
|
338
|
+
elsif n < 0x100
|
|
339
|
+
out << (60 << 2) << n
|
|
340
|
+
elsif n < 0x10000
|
|
341
|
+
out << (61 << 2) << (n & 0xff) << (n >> 8)
|
|
342
|
+
elsif n < 0x1000000
|
|
343
|
+
out << (62 << 2) << (n & 0xff) << ((n >> 8) & 0xff) << (n >> 16)
|
|
344
|
+
else
|
|
345
|
+
out << (63 << 2) << (n & 0xff) << ((n >> 8) & 0xff) << ((n >> 16) & 0xff) << (n >> 24)
|
|
346
|
+
end
|
|
347
|
+
out << src.byteslice(pos, len)
|
|
348
|
+
end
|
|
349
|
+
|
|
350
|
+
# Offsets are always < 64KB (matches stay within a block), so 4-byte offsets are never needed.
|
|
351
|
+
def emit_copy(out, offset, len)
|
|
352
|
+
while len >= 68
|
|
353
|
+
emit_copy_upto64(out, offset, 64)
|
|
354
|
+
len -= 64
|
|
355
|
+
end
|
|
356
|
+
if len > 64
|
|
357
|
+
emit_copy_upto64(out, offset, 60)
|
|
358
|
+
len -= 60
|
|
359
|
+
end
|
|
360
|
+
emit_copy_upto64(out, offset, len)
|
|
361
|
+
end
|
|
362
|
+
|
|
363
|
+
def emit_copy_upto64(out, offset, len)
|
|
364
|
+
if len < 12 && offset < 2048
|
|
365
|
+
out << (1 | ((len - 4) << 2) | ((offset >> 8) << 5)) << (offset & 0xff)
|
|
366
|
+
else
|
|
367
|
+
out << (2 | ((len - 1) << 2)) << (offset & 0xff) << (offset >> 8)
|
|
368
|
+
end
|
|
369
|
+
end
|
|
370
|
+
|
|
371
|
+
private_class_method :decompress_io_buffer, :decompress_string, :read_varint, :write_varint, :compress_block, :block_words, :match_length_words,
|
|
372
|
+
:match_length, :emit_literal, :emit_copy, :emit_copy_upto64
|
|
373
|
+
end
|
|
374
|
+
end
|
|
375
|
+
end
|