omnizip 0.3.13 → 0.3.15
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/lib/omnizip/algorithm.rb +36 -1
- data/lib/omnizip/algorithms/bzip2.rb +1 -3
- data/lib/omnizip/algorithms/deflate.rb +1 -3
- data/lib/omnizip/algorithms/deflate64.rb +2 -2
- data/lib/omnizip/algorithms/lzma/optimal_encoder.rb +7 -8
- data/lib/omnizip/algorithms/lzma.rb +1 -1
- data/lib/omnizip/algorithms/lzma2/lzma2_chunk.rb +5 -1
- data/lib/omnizip/algorithms/lzma2.rb +4 -12
- data/lib/omnizip/algorithms/zstandard/constants.rb +38 -25
- data/lib/omnizip/algorithms/zstandard/decoder.rb +101 -136
- data/lib/omnizip/algorithms/zstandard/encoder.rb +81 -138
- data/lib/omnizip/algorithms/zstandard/frame/block.rb +17 -0
- data/lib/omnizip/algorithms/zstandard/frame/header.rb +42 -29
- data/lib/omnizip/algorithms/zstandard/fse/bitstream.rb +114 -117
- data/lib/omnizip/algorithms/zstandard/fse/encoder.rb +434 -222
- data/lib/omnizip/algorithms/zstandard/fse/interleaved.rb +92 -0
- data/lib/omnizip/algorithms/zstandard/fse/table.rb +98 -176
- data/lib/omnizip/algorithms/zstandard/fse/table_description.rb +203 -0
- data/lib/omnizip/algorithms/zstandard/fse.rb +21 -2
- data/lib/omnizip/algorithms/zstandard/huffman.rb +195 -188
- data/lib/omnizip/algorithms/zstandard/huffman_encoder.rb +232 -255
- data/lib/omnizip/algorithms/zstandard/literals.rb +170 -103
- data/lib/omnizip/algorithms/zstandard/literals_encoder.rb +47 -199
- data/lib/omnizip/algorithms/zstandard/sequences.rb +260 -252
- data/lib/omnizip/algorithms/zstandard/xxhash.rb +159 -0
- data/lib/omnizip/algorithms/zstandard.rb +20 -24
- data/lib/omnizip/algorithms.rb +14 -1
- data/lib/omnizip/buffer.rb +1 -2
- data/lib/omnizip/commands/archive_list_command.rb +9 -7
- data/lib/omnizip/convenience.rb +1 -2
- data/lib/omnizip/entry.rb +44 -0
- data/lib/omnizip/extraction/selective_extractor.rb +2 -10
- data/lib/omnizip/filter_pipeline.rb +1 -1
- data/lib/omnizip/filter_registry.rb +18 -16
- data/lib/omnizip/filters/bcj2.rb +1 -1
- data/lib/omnizip/filters/bcj_arm.rb +1 -1
- data/lib/omnizip/filters/bcj_arm64.rb +1 -1
- data/lib/omnizip/filters/bcj_ia64.rb +1 -1
- data/lib/omnizip/filters/bcj_ppc.rb +1 -1
- data/lib/omnizip/filters/bcj_sparc.rb +1 -1
- data/lib/omnizip/filters/bcj_x86.rb +1 -1
- data/lib/omnizip/filters.rb +0 -2
- data/lib/omnizip/formats/bzip2_file.rb +0 -7
- data/lib/omnizip/formats/cpio/entry.rb +6 -0
- data/lib/omnizip/formats/cpio.rb +0 -6
- data/lib/omnizip/formats/gzip.rb +0 -7
- data/lib/omnizip/formats/iso/directory_record.rb +7 -0
- data/lib/omnizip/formats/iso.rb +0 -5
- data/lib/omnizip/formats/lzip.rb +0 -7
- data/lib/omnizip/formats/lzma_alone.rb +0 -6
- data/lib/omnizip/formats/msi/entry.rb +6 -0
- data/lib/omnizip/formats/msi.rb +0 -9
- data/lib/omnizip/formats/ole/dirent.rb +6 -0
- data/lib/omnizip/formats/ole.rb +0 -9
- data/lib/omnizip/formats/rar.rb +0 -6
- data/lib/omnizip/formats/rar3/reader.rb +7 -0
- data/lib/omnizip/formats/rar5/reader.rb +7 -0
- data/lib/omnizip/formats/rpm/entry.rb +6 -0
- data/lib/omnizip/formats/seven_zip/models/file_entry.rb +7 -0
- data/lib/omnizip/formats/seven_zip.rb +0 -6
- data/lib/omnizip/formats/tar/entry.rb +6 -0
- data/lib/omnizip/formats/tar.rb +2 -10
- data/lib/omnizip/formats/xar/entry.rb +6 -0
- data/lib/omnizip/formats/xar.rb +0 -6
- data/lib/omnizip/formats/zip.rb +0 -6
- data/lib/omnizip/implementations/seven_zip/lzma2/encoder.rb +33 -31
- data/lib/omnizip/implementations/xz_utils/lzma2/encoder.rb +177 -163
- data/lib/omnizip/implementations.rb +18 -4
- data/lib/omnizip/io/source.rb +3 -0
- data/lib/omnizip/metadata/entry_metadata.rb +7 -0
- data/lib/omnizip/parallel/engine.rb +48 -0
- data/lib/omnizip/parallel/parallel_compressor.rb +5 -16
- data/lib/omnizip/parallel/parallel_extractor.rb +5 -16
- data/lib/omnizip/parallel.rb +1 -0
- data/lib/omnizip/profile/profile_registry.rb +1 -2
- data/lib/omnizip/registry.rb +0 -1
- data/lib/omnizip/rubyzip_compat.rb +0 -1
- data/lib/omnizip/temp/temp_file.rb +1 -2
- data/lib/omnizip/version.rb +1 -1
- data/lib/omnizip/zip/entry.rb +7 -0
- data/lib/omnizip/zip/file.rb +4 -8
- data/lib/omnizip.rb +1 -1
- metadata +7 -6
- data/lib/omnizip/filters/filter_base.rb +0 -6
- data/lib/omnizip/filters/registration.rb +0 -22
- data/lib/omnizip/filters/registry.rb +0 -111
- data/lib/omnizip/format_registry.rb +0 -100
|
@@ -23,150 +23,217 @@
|
|
|
23
23
|
module Omnizip
|
|
24
24
|
module Algorithms
|
|
25
25
|
class Zstandard
|
|
26
|
-
# Literals section decoder (RFC 8878
|
|
26
|
+
# Literals section decoder (RFC 8878 §3.1.1.3.1).
|
|
27
27
|
#
|
|
28
|
-
#
|
|
29
|
-
#
|
|
28
|
+
# Header byte 0 layout (bits numbered from the LSB, matching the C
|
|
29
|
+
# reference `litEncType = istart[0] & 3`):
|
|
30
|
+
#
|
|
31
|
+
# bits 0-1 Literals_Block_Type (0=Raw, 1=RLE, 2=Compressed,
|
|
32
|
+
# 3=Treeless)
|
|
33
|
+
# bits 2-3 Size_Format selector
|
|
34
|
+
#
|
|
35
|
+
# Raw / RLE size formats:
|
|
36
|
+
# 00, 10 -> 1-byte header, regen = byte0 >> 3
|
|
37
|
+
# 01 -> 2-byte header, regen = LE16 >> 4
|
|
38
|
+
# 11 -> 3-byte header, regen = LE24 >> 4
|
|
39
|
+
#
|
|
40
|
+
# Compressed / Treeless size formats:
|
|
41
|
+
# 00 -> 3-byte header, single stream, 10-bit sizes
|
|
42
|
+
# 01 -> 3-byte header, 4 streams, 10-bit sizes
|
|
43
|
+
# 10 -> 4-byte header, 4 streams, 14-bit sizes
|
|
44
|
+
# 11 -> 5-byte header, 4 streams, 18-bit sizes
|
|
30
45
|
class LiteralsDecoder
|
|
31
46
|
include Constants
|
|
32
47
|
|
|
33
|
-
# @return [String]
|
|
48
|
+
# @return [String] decoded literal bytes
|
|
34
49
|
attr_reader :literals
|
|
35
50
|
|
|
36
|
-
# @return [Huffman, nil]
|
|
51
|
+
# @return [Huffman, nil] table from a Compressed block, for the
|
|
52
|
+
# next Treeless block in the same frame
|
|
37
53
|
attr_reader :huffman_table
|
|
38
54
|
|
|
39
|
-
#
|
|
40
|
-
|
|
41
|
-
# @param input [IO] Input stream positioned at literals section
|
|
42
|
-
# @param previous_table [Huffman, nil] Previous Huffman table (for treeless)
|
|
43
|
-
# @return [LiteralsDecoder] Decoder with decoded literals
|
|
44
|
-
def self.decode(input, previous_table = nil)
|
|
45
|
-
decoder = new(input, previous_table)
|
|
46
|
-
decoder.decode_section
|
|
47
|
-
decoder
|
|
48
|
-
end
|
|
55
|
+
# @return [Integer] bytes consumed from the input
|
|
56
|
+
attr_reader :consumed
|
|
49
57
|
|
|
50
|
-
#
|
|
58
|
+
# Decode the literals section at the head of `input`.
|
|
51
59
|
#
|
|
52
|
-
# @param input [
|
|
53
|
-
# @param previous_table [Huffman, nil]
|
|
54
|
-
|
|
55
|
-
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
@huffman_table = previous_table
|
|
59
|
-
end
|
|
60
|
+
# @param input [String]
|
|
61
|
+
# @param previous_table [Huffman, nil] table from the previous
|
|
62
|
+
# compressed literals block (required for Treeless)
|
|
63
|
+
# @return [LiteralsDecoder]
|
|
64
|
+
def self.decode(input, previous_table = nil)
|
|
65
|
+
raise Omnizip::DecompressionError, "empty literals section" if input.empty?
|
|
60
66
|
|
|
61
|
-
|
|
62
|
-
#
|
|
63
|
-
# @return [void]
|
|
64
|
-
def decode_section
|
|
65
|
-
# Read literals header (1-3 bytes)
|
|
66
|
-
header1 = @input.read(1).ord
|
|
67
|
-
block_type = (header1 >> 6) & 0x03
|
|
67
|
+
block_type = input.getbyte(0) & 0x03
|
|
68
68
|
|
|
69
69
|
case block_type
|
|
70
|
-
when LITERALS_BLOCK_RAW
|
|
71
|
-
|
|
72
|
-
when LITERALS_BLOCK_RLE
|
|
73
|
-
decode_rle(header1)
|
|
70
|
+
when LITERALS_BLOCK_RAW then decode_raw(input)
|
|
71
|
+
when LITERALS_BLOCK_RLE then decode_rle(input)
|
|
74
72
|
when LITERALS_BLOCK_COMPRESSED
|
|
75
|
-
decode_compressed(
|
|
73
|
+
decode_compressed(input, previous_table, false)
|
|
76
74
|
when LITERALS_BLOCK_TREELESS
|
|
77
|
-
|
|
75
|
+
decode_compressed(input, previous_table, true)
|
|
78
76
|
end
|
|
79
77
|
end
|
|
80
78
|
|
|
81
|
-
|
|
79
|
+
def self.decode_raw(input)
|
|
80
|
+
regen_size, header_size = size_format_raw_rle(input)
|
|
81
|
+
end_pos = header_size + regen_size
|
|
82
|
+
if input.bytesize < end_pos
|
|
83
|
+
raise Omnizip::DecompressionError,
|
|
84
|
+
"truncated raw literals: need #{end_pos} bytes"
|
|
85
|
+
end
|
|
82
86
|
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
# Size format: 5-bit or 12-bit or 20-bit
|
|
86
|
-
size = header1 & 0x1F
|
|
87
|
+
new(input.byteslice(header_size, regen_size), nil, end_pos)
|
|
88
|
+
end
|
|
87
89
|
|
|
88
|
-
|
|
89
|
-
|
|
90
|
-
|
|
91
|
-
|
|
90
|
+
def self.decode_rle(input)
|
|
91
|
+
regen_size, header_size = size_format_raw_rle(input)
|
|
92
|
+
if input.bytesize < header_size + 1
|
|
93
|
+
raise Omnizip::DecompressionError,
|
|
94
|
+
"truncated RLE literals: missing repeated byte"
|
|
92
95
|
end
|
|
93
96
|
|
|
94
|
-
|
|
97
|
+
byte = input.getbyte(header_size)
|
|
98
|
+
new(byte.chr * regen_size, nil, header_size + 1)
|
|
95
99
|
end
|
|
96
100
|
|
|
97
|
-
#
|
|
98
|
-
|
|
99
|
-
|
|
100
|
-
|
|
101
|
-
|
|
102
|
-
|
|
103
|
-
|
|
104
|
-
header2 = @input.read(2).unpack1("v")
|
|
105
|
-
size = header2 + 31
|
|
101
|
+
# rubocop:disable Metrics/AbcSize
|
|
102
|
+
# rubocop:disable-next Metrics/MethodLength
|
|
103
|
+
def self.decode_compressed(input, previous_table, is_repeat)
|
|
104
|
+
if is_repeat && previous_table.nil?
|
|
105
|
+
raise Omnizip::DecompressionError,
|
|
106
|
+
"treeless literals block requires a previous compressed " \
|
|
107
|
+
"block in the same frame"
|
|
106
108
|
end
|
|
107
109
|
|
|
108
|
-
|
|
109
|
-
byte = @input.read(1)
|
|
110
|
-
@literals = byte * size
|
|
111
|
-
end
|
|
110
|
+
lhl_code = (input.getbyte(0) >> 2) & 0x03
|
|
112
111
|
|
|
113
|
-
|
|
114
|
-
|
|
115
|
-
|
|
116
|
-
|
|
117
|
-
|
|
118
|
-
|
|
119
|
-
|
|
120
|
-
|
|
121
|
-
|
|
122
|
-
|
|
123
|
-
|
|
124
|
-
|
|
125
|
-
|
|
126
|
-
|
|
112
|
+
min_header = [3, 3, 4, 5][lhl_code]
|
|
113
|
+
if input.bytesize < min_header
|
|
114
|
+
raise Omnizip::DecompressionError,
|
|
115
|
+
"truncated compressed literals header: need #{min_header} bytes"
|
|
116
|
+
end
|
|
117
|
+
|
|
118
|
+
lit_size, lit_c_size, lh_size, single_stream =
|
|
119
|
+
case lhl_code
|
|
120
|
+
when 0, 1
|
|
121
|
+
lhc = input.getbyte(0) |
|
|
122
|
+
(input.getbyte(1) << 8) | (input.getbyte(2) << 16)
|
|
123
|
+
[(lhc >> 4) & 0x3FF, (lhc >> 14) & 0x3FF, 3, lhl_code.zero?]
|
|
124
|
+
when 2
|
|
125
|
+
lhc = input.unpack1("V") & 0xFFFFFFFF
|
|
126
|
+
[(lhc >> 4) & 0x3FFF, lhc >> 18, 4, false]
|
|
127
127
|
else
|
|
128
|
-
|
|
129
|
-
|
|
130
|
-
|
|
131
|
-
4
|
|
128
|
+
low = input.byteslice(0, 4).unpack1("V")
|
|
129
|
+
high = input.getbyte(4)
|
|
130
|
+
lhc = low | (high << 32)
|
|
131
|
+
[(lhc >> 4) & 0x3FFFF, lhc >> 22, 5, false]
|
|
132
132
|
end
|
|
133
|
-
end
|
|
134
133
|
|
|
135
|
-
|
|
134
|
+
needed = lh_size + lit_c_size
|
|
135
|
+
if input.bytesize < needed
|
|
136
|
+
raise Omnizip::DecompressionError,
|
|
137
|
+
"truncated compressed literals: need #{needed} bytes"
|
|
138
|
+
end
|
|
139
|
+
compressed = input.byteslice(lh_size, lit_c_size)
|
|
136
140
|
|
|
137
|
-
|
|
138
|
-
|
|
139
|
-
|
|
141
|
+
table_bytes = 0
|
|
142
|
+
table = previous_table
|
|
143
|
+
unless is_repeat
|
|
144
|
+
table, table_bytes = HuffmanTableReader.read(compressed)
|
|
145
|
+
end
|
|
140
146
|
|
|
141
|
-
|
|
142
|
-
|
|
147
|
+
literal_data = compressed.byteslice(table_bytes..)
|
|
148
|
+
literals = if single_stream
|
|
149
|
+
decode_single_stream(table, literal_data, lit_size)
|
|
150
|
+
else
|
|
151
|
+
decode_four_stream(table, literal_data, lit_size)
|
|
152
|
+
end
|
|
143
153
|
|
|
144
|
-
|
|
145
|
-
# This is a simplified implementation
|
|
146
|
-
@literals = @input.read(regenerated_size)
|
|
154
|
+
new(literals, is_repeat ? nil : table, needed)
|
|
147
155
|
end
|
|
156
|
+
# rubocop:enable Metrics/AbcSize
|
|
157
|
+
|
|
158
|
+
# rubocop:disable-next Metrics/AbcSize
|
|
159
|
+
def self.size_format_raw_rle(input)
|
|
160
|
+
lhl_code = (input.getbyte(0) >> 2) & 0x03
|
|
161
|
+
header_size = [1, 2, 1, 3][lhl_code]
|
|
162
|
+
if input.bytesize < header_size
|
|
163
|
+
raise Omnizip::DecompressionError,
|
|
164
|
+
"truncated Raw/RLE literals header: need #{header_size} bytes"
|
|
165
|
+
end
|
|
148
166
|
|
|
149
|
-
|
|
150
|
-
|
|
151
|
-
|
|
152
|
-
|
|
167
|
+
case lhl_code
|
|
168
|
+
when 0, 2
|
|
169
|
+
[input.getbyte(0) >> 3, 1]
|
|
170
|
+
when 1
|
|
171
|
+
[input.unpack1("v") >> 4, 2]
|
|
172
|
+
else
|
|
173
|
+
lhc = input.getbyte(0) | (input.getbyte(1) << 8) |
|
|
174
|
+
(input.getbyte(2) << 16)
|
|
175
|
+
[lhc >> 4, 3]
|
|
176
|
+
end
|
|
177
|
+
end
|
|
153
178
|
|
|
154
|
-
|
|
155
|
-
|
|
156
|
-
|
|
179
|
+
# One reverse bitstream decoded into lit_size symbols.
|
|
180
|
+
def self.decode_single_stream(table, src, lit_size)
|
|
181
|
+
bs = FSE::BitStream.new(src)
|
|
182
|
+
out = String.new(encoding: Encoding::BINARY)
|
|
183
|
+
lit_size.times do
|
|
184
|
+
out << table.decode(bs).chr
|
|
185
|
+
bs.reload
|
|
157
186
|
end
|
|
187
|
+
out
|
|
188
|
+
end
|
|
158
189
|
|
|
159
|
-
|
|
190
|
+
# Jump table + four independent reverse bitstreams, each decoding
|
|
191
|
+
# ~1/4 of the literals (RFC 8878 §3.1.1.3.1.4).
|
|
192
|
+
# rubocop:disable Metrics/AbcSize
|
|
193
|
+
def self.decode_four_stream(table, src, lit_size)
|
|
194
|
+
if src.bytesize < 10
|
|
195
|
+
raise Omnizip::DecompressionError,
|
|
196
|
+
"4-stream literals too short (need at least 10 bytes)"
|
|
197
|
+
end
|
|
198
|
+
# rubocop:enable Metrics/AbcSize
|
|
199
|
+
|
|
200
|
+
length1 = src.unpack1("v")
|
|
201
|
+
length2 = src.byteslice(2, 2).unpack1("v")
|
|
202
|
+
length3 = src.byteslice(4, 2).unpack1("v")
|
|
203
|
+
total = src.bytesize - 6
|
|
204
|
+
if length1 + length2 + length3 > total
|
|
205
|
+
raise Omnizip::DecompressionError,
|
|
206
|
+
"4-stream jump table sizes exceed total stream size"
|
|
207
|
+
end
|
|
160
208
|
|
|
161
|
-
|
|
162
|
-
|
|
163
|
-
|
|
164
|
-
|
|
165
|
-
|
|
209
|
+
streams = src.byteslice(6..)
|
|
210
|
+
seg1 = streams.byteslice(0, length1)
|
|
211
|
+
seg2 = streams.byteslice(length1, length2)
|
|
212
|
+
seg3 = streams.byteslice(length1 + length2, length3)
|
|
213
|
+
seg4 = streams.byteslice((length1 + length2 + length3)..)
|
|
214
|
+
|
|
215
|
+
segment_size = (lit_size + 3) / 4
|
|
216
|
+
bounds = [0,
|
|
217
|
+
segment_size,
|
|
218
|
+
segment_size * 2,
|
|
219
|
+
segment_size * 3,
|
|
220
|
+
lit_size]
|
|
221
|
+
|
|
222
|
+
out = String.new(encoding: Encoding::BINARY)
|
|
223
|
+
[seg1, seg2, seg3, seg4].each_with_index do |seg, i|
|
|
224
|
+
bs = FSE::BitStream.new(seg)
|
|
225
|
+
(bounds[i]...bounds[i + 1]).each do
|
|
226
|
+
out << table.decode(bs).chr
|
|
227
|
+
bs.reload
|
|
228
|
+
end
|
|
166
229
|
end
|
|
230
|
+
out
|
|
231
|
+
end
|
|
167
232
|
|
|
168
|
-
|
|
169
|
-
@literals =
|
|
233
|
+
def initialize(literals, huffman_table, consumed)
|
|
234
|
+
@literals = literals
|
|
235
|
+
@huffman_table = huffman_table
|
|
236
|
+
@consumed = consumed
|
|
170
237
|
end
|
|
171
238
|
end
|
|
172
239
|
end
|
|
@@ -23,225 +23,73 @@
|
|
|
23
23
|
module Omnizip
|
|
24
24
|
module Algorithms
|
|
25
25
|
class Zstandard
|
|
26
|
-
# Literals Section
|
|
27
|
-
#
|
|
28
|
-
|
|
29
|
-
# Supports raw, RLE, and Huffman-compressed literals.
|
|
30
|
-
class LiteralsEncoder
|
|
26
|
+
# Literals Section encoder (RFC 8878 §3.1.1.3.1): picks Raw, RLE
|
|
27
|
+
# or Huffman-compressed, whichever is smallest.
|
|
28
|
+
module LiteralsEncoder
|
|
31
29
|
include Constants
|
|
32
30
|
|
|
33
|
-
|
|
34
|
-
attr_reader :huffman_encoder
|
|
31
|
+
module_function
|
|
35
32
|
|
|
36
|
-
# Encode literals section
|
|
33
|
+
# Encode a literals section for `literals`.
|
|
37
34
|
#
|
|
38
|
-
# @param literals [String]
|
|
39
|
-
# @
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
|
|
50
|
-
# @param previous_huffman [HuffmanEncoder, nil] Previous Huffman encoder
|
|
51
|
-
# @param use_compression [Boolean] Whether to use compression
|
|
52
|
-
def initialize(literals, previous_huffman = nil, use_compression = true)
|
|
53
|
-
@literals = literals.to_s.dup.force_encoding(Encoding::BINARY)
|
|
54
|
-
@previous_huffman = previous_huffman
|
|
55
|
-
@use_compression = use_compression
|
|
56
|
-
@huffman_encoder = nil
|
|
57
|
-
end
|
|
58
|
-
|
|
59
|
-
# Encode the literals section
|
|
60
|
-
#
|
|
61
|
-
# @return [String] Encoded section
|
|
62
|
-
def encode_section
|
|
63
|
-
return encode_empty if @literals.empty?
|
|
64
|
-
|
|
65
|
-
# Choose encoding method based on data characteristics
|
|
66
|
-
if rle_efficient?
|
|
67
|
-
encode_rle
|
|
68
|
-
elsif @use_compression && huffman_efficient?
|
|
69
|
-
encode_huffman
|
|
70
|
-
else
|
|
71
|
-
encode_raw
|
|
35
|
+
# @param literals [String]
|
|
36
|
+
# @return [String] the full section (header + content)
|
|
37
|
+
def encode(literals)
|
|
38
|
+
return encode_raw(literals) if literals.bytesize < 16
|
|
39
|
+
|
|
40
|
+
return encode_rle(literals) if rle?(literals)
|
|
41
|
+
|
|
42
|
+
compressed = nil
|
|
43
|
+
begin
|
|
44
|
+
compressed = HuffmanEncoder.encode_literals(literals)
|
|
45
|
+
rescue Omnizip::CompressionError
|
|
46
|
+
compressed = nil
|
|
72
47
|
end
|
|
73
|
-
end
|
|
74
|
-
|
|
75
|
-
private
|
|
76
|
-
|
|
77
|
-
# Check if RLE encoding would be efficient
|
|
78
|
-
def rle_efficient?
|
|
79
|
-
return false if @literals.length < 3
|
|
80
|
-
|
|
81
|
-
# Check if all bytes are the same
|
|
82
|
-
first_byte = @literals.getbyte(0)
|
|
83
|
-
@literals.bytes.all?(first_byte)
|
|
84
|
-
end
|
|
85
|
-
|
|
86
|
-
# Check if Huffman encoding would be efficient
|
|
87
|
-
def huffman_efficient?
|
|
88
|
-
return false if @literals.length < 16
|
|
89
|
-
|
|
90
|
-
# Check if data has enough redundancy
|
|
91
|
-
entropy = calculate_entropy(@literals)
|
|
92
|
-
entropy < 7.5 # Less than 7.5 bits per byte suggests compressibility
|
|
93
|
-
end
|
|
94
|
-
|
|
95
|
-
# Calculate Shannon entropy of data
|
|
96
|
-
def calculate_entropy(data)
|
|
97
|
-
return 0 if data.empty?
|
|
98
|
-
|
|
99
|
-
# Count byte frequencies
|
|
100
|
-
freq = Array.new(256, 0)
|
|
101
|
-
data.each_byte { |b| freq[b] += 1 }
|
|
102
48
|
|
|
103
|
-
|
|
104
|
-
|
|
105
|
-
|
|
106
|
-
|
|
107
|
-
freq.each do |count|
|
|
108
|
-
next if count.zero?
|
|
109
|
-
|
|
110
|
-
prob = count / total
|
|
111
|
-
entropy -= prob * Math.log2(prob)
|
|
49
|
+
if compressed && compressed.bytesize < literals.bytesize
|
|
50
|
+
compressed
|
|
51
|
+
else
|
|
52
|
+
encode_raw(literals)
|
|
112
53
|
end
|
|
113
|
-
|
|
114
|
-
entropy
|
|
115
|
-
end
|
|
116
|
-
|
|
117
|
-
# Encode empty literals
|
|
118
|
-
def encode_empty
|
|
119
|
-
# Type 0 (raw), size 0
|
|
120
|
-
"\x00"
|
|
121
54
|
end
|
|
122
55
|
|
|
123
|
-
#
|
|
124
|
-
def encode_raw
|
|
125
|
-
|
|
126
|
-
|
|
127
|
-
header + @literals
|
|
56
|
+
# Raw literals block with the minimal size-format header.
|
|
57
|
+
def encode_raw(literals)
|
|
58
|
+
encode_raw_rle_header(LITERALS_BLOCK_RAW, literals.bytesize) +
|
|
59
|
+
literals
|
|
128
60
|
end
|
|
129
61
|
|
|
130
|
-
#
|
|
131
|
-
def encode_rle
|
|
132
|
-
|
|
133
|
-
|
|
134
|
-
|
|
135
|
-
header = encode_literals_header(LITERALS_BLOCK_RLE, size)
|
|
136
|
-
header + [byte].pack("C")
|
|
137
|
-
end
|
|
138
|
-
|
|
139
|
-
# Encode Huffman-compressed literals
|
|
140
|
-
def encode_huffman
|
|
141
|
-
size = @literals.bytesize
|
|
142
|
-
|
|
143
|
-
# Build Huffman tree from literals
|
|
144
|
-
@huffman_encoder = build_huffman_encoder(@literals)
|
|
145
|
-
|
|
146
|
-
if @huffman_encoder.nil?
|
|
147
|
-
# Fallback to raw if Huffman fails
|
|
148
|
-
return encode_raw
|
|
149
|
-
end
|
|
150
|
-
|
|
151
|
-
# Encode literals with Huffman
|
|
152
|
-
compressed = @huffman_encoder.encode(@literals)
|
|
153
|
-
|
|
154
|
-
# Check if compression is beneficial
|
|
155
|
-
# Need to account for header + table description overhead
|
|
156
|
-
table_desc = @huffman_encoder.encode_table_description
|
|
157
|
-
total_compressed_size = compressed.bytesize + table_desc.bytesize
|
|
158
|
-
|
|
159
|
-
if total_compressed_size >= size
|
|
160
|
-
# Not beneficial, use raw
|
|
161
|
-
@huffman_encoder = nil
|
|
162
|
-
return encode_raw
|
|
163
|
-
end
|
|
164
|
-
|
|
165
|
-
# Build header for LITERALS_BLOCK_COMPRESSED
|
|
166
|
-
# Type (2 bits) = 10, followed by regenerated size
|
|
167
|
-
header = encode_literals_header(LITERALS_BLOCK_COMPRESSED, size,
|
|
168
|
-
total_compressed_size)
|
|
169
|
-
|
|
170
|
-
# Build complete section: header + table_desc + compressed
|
|
171
|
-
header + table_desc + compressed
|
|
62
|
+
# RLE literals block: one byte, repeated.
|
|
63
|
+
def encode_rle(literals)
|
|
64
|
+
encode_raw_rle_header(LITERALS_BLOCK_RLE, literals.bytesize) +
|
|
65
|
+
literals.getbyte(0).chr
|
|
172
66
|
end
|
|
173
67
|
|
|
174
|
-
|
|
175
|
-
|
|
176
|
-
|
|
177
|
-
|
|
178
|
-
|
|
179
|
-
# - Compressed size (variable length, only for compressed type)
|
|
180
|
-
def encode_literals_header(type, regenerated_size,
|
|
181
|
-
compressed_size = nil)
|
|
182
|
-
# Encode regenerated size
|
|
183
|
-
if regenerated_size < 32
|
|
184
|
-
# 5-bit size: type(2) + size(5) + padding(1) = 8 bits
|
|
185
|
-
header_byte = (type << 6) | regenerated_size
|
|
186
|
-
header = [header_byte].pack("C")
|
|
187
|
-
elsif regenerated_size < 4096
|
|
188
|
-
# 12-bit size
|
|
189
|
-
header_byte = (type << 6) | 31
|
|
190
|
-
size_field = regenerated_size - 31
|
|
191
|
-
header = [header_byte, size_field & 0xFF,
|
|
192
|
-
(size_field >> 8) & 0xFF].pack("Cv")
|
|
193
|
-
else
|
|
194
|
-
# 20-bit size
|
|
195
|
-
header_byte = (type << 6) | 31
|
|
196
|
-
# Extended size format
|
|
197
|
-
header = [header_byte].pack("C")
|
|
198
|
-
header += encode_extended_size(regenerated_size - 31)
|
|
199
|
-
end
|
|
200
|
-
|
|
201
|
-
# Add compressed size for LITERALS_BLOCK_COMPRESSED
|
|
202
|
-
if type == LITERALS_BLOCK_COMPRESSED && compressed_size
|
|
203
|
-
header + encode_compressed_size(compressed_size)
|
|
204
|
-
else
|
|
205
|
-
header
|
|
206
|
-
end
|
|
207
|
-
end
|
|
68
|
+
def rle?(literals)
|
|
69
|
+
first = literals.getbyte(0)
|
|
70
|
+
idx = 1
|
|
71
|
+
while idx < literals.bytesize
|
|
72
|
+
return false if literals.getbyte(idx) != first
|
|
208
73
|
|
|
209
|
-
|
|
210
|
-
def encode_extended_size(size)
|
|
211
|
-
if size < 128
|
|
212
|
-
# Single byte
|
|
213
|
-
[size].pack("C")
|
|
214
|
-
elsif size < 16384
|
|
215
|
-
# Two bytes
|
|
216
|
-
[size | 0x80, (size >> 7) & 0x7F].pack("CC")
|
|
217
|
-
else
|
|
218
|
-
# Three bytes
|
|
219
|
-
[size | 0x80, (size >> 7) | 0x80, (size >> 14) & 0x7F].pack("CCC")
|
|
74
|
+
idx += 1
|
|
220
75
|
end
|
|
76
|
+
true
|
|
221
77
|
end
|
|
222
78
|
|
|
223
|
-
#
|
|
224
|
-
|
|
225
|
-
|
|
226
|
-
|
|
227
|
-
|
|
228
|
-
[
|
|
79
|
+
# 1/2/3-byte Raw/RLE header per the size-format rules. For the
|
|
80
|
+
# 2- and 3-byte forms the Size_Format bits (01 / 11) live in
|
|
81
|
+
# bits 2-3 and the size starts at bit 4.
|
|
82
|
+
def encode_raw_rle_header(type, size)
|
|
83
|
+
if size < 32
|
|
84
|
+
[type | (size << 3)].pack("C")
|
|
85
|
+
elsif size < 4096
|
|
86
|
+
lhc = type | 0x04 | (size << 4)
|
|
87
|
+
[lhc & 0xFF, (lhc >> 8) & 0xFF].pack("CC")
|
|
229
88
|
else
|
|
230
|
-
|
|
89
|
+
lhc = type | 0x0C | (size << 4)
|
|
90
|
+
[lhc & 0xFF, (lhc >> 8) & 0xFF, (lhc >> 16) & 0xFF].pack("CCC")
|
|
231
91
|
end
|
|
232
92
|
end
|
|
233
|
-
|
|
234
|
-
# Build Huffman encoder from data
|
|
235
|
-
def build_huffman_encoder(data)
|
|
236
|
-
return nil if data.nil? || data.empty?
|
|
237
|
-
|
|
238
|
-
# Count byte frequencies
|
|
239
|
-
freq = Array.new(256, 0)
|
|
240
|
-
data.each_byte { |b| freq[b] += 1 }
|
|
241
|
-
|
|
242
|
-
# Build Huffman encoder
|
|
243
|
-
HuffmanEncoder.build_from_frequencies(freq, HUFFMAN_MAX_BITS)
|
|
244
|
-
end
|
|
245
93
|
end
|
|
246
94
|
end
|
|
247
95
|
end
|