omnizip 0.3.13 → 0.3.15
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/lib/omnizip/algorithm.rb +36 -1
- data/lib/omnizip/algorithms/bzip2.rb +1 -3
- data/lib/omnizip/algorithms/deflate.rb +1 -3
- data/lib/omnizip/algorithms/deflate64.rb +2 -2
- data/lib/omnizip/algorithms/lzma/optimal_encoder.rb +7 -8
- data/lib/omnizip/algorithms/lzma.rb +1 -1
- data/lib/omnizip/algorithms/lzma2/lzma2_chunk.rb +5 -1
- data/lib/omnizip/algorithms/lzma2.rb +4 -12
- data/lib/omnizip/algorithms/zstandard/constants.rb +38 -25
- data/lib/omnizip/algorithms/zstandard/decoder.rb +101 -136
- data/lib/omnizip/algorithms/zstandard/encoder.rb +81 -138
- data/lib/omnizip/algorithms/zstandard/frame/block.rb +17 -0
- data/lib/omnizip/algorithms/zstandard/frame/header.rb +42 -29
- data/lib/omnizip/algorithms/zstandard/fse/bitstream.rb +114 -117
- data/lib/omnizip/algorithms/zstandard/fse/encoder.rb +434 -222
- data/lib/omnizip/algorithms/zstandard/fse/interleaved.rb +92 -0
- data/lib/omnizip/algorithms/zstandard/fse/table.rb +98 -176
- data/lib/omnizip/algorithms/zstandard/fse/table_description.rb +203 -0
- data/lib/omnizip/algorithms/zstandard/fse.rb +21 -2
- data/lib/omnizip/algorithms/zstandard/huffman.rb +195 -188
- data/lib/omnizip/algorithms/zstandard/huffman_encoder.rb +232 -255
- data/lib/omnizip/algorithms/zstandard/literals.rb +170 -103
- data/lib/omnizip/algorithms/zstandard/literals_encoder.rb +47 -199
- data/lib/omnizip/algorithms/zstandard/sequences.rb +260 -252
- data/lib/omnizip/algorithms/zstandard/xxhash.rb +159 -0
- data/lib/omnizip/algorithms/zstandard.rb +20 -24
- data/lib/omnizip/algorithms.rb +14 -1
- data/lib/omnizip/buffer.rb +1 -2
- data/lib/omnizip/commands/archive_list_command.rb +9 -7
- data/lib/omnizip/convenience.rb +1 -2
- data/lib/omnizip/entry.rb +44 -0
- data/lib/omnizip/extraction/selective_extractor.rb +2 -10
- data/lib/omnizip/filter_pipeline.rb +1 -1
- data/lib/omnizip/filter_registry.rb +18 -16
- data/lib/omnizip/filters/bcj2.rb +1 -1
- data/lib/omnizip/filters/bcj_arm.rb +1 -1
- data/lib/omnizip/filters/bcj_arm64.rb +1 -1
- data/lib/omnizip/filters/bcj_ia64.rb +1 -1
- data/lib/omnizip/filters/bcj_ppc.rb +1 -1
- data/lib/omnizip/filters/bcj_sparc.rb +1 -1
- data/lib/omnizip/filters/bcj_x86.rb +1 -1
- data/lib/omnizip/filters.rb +0 -2
- data/lib/omnizip/formats/bzip2_file.rb +0 -7
- data/lib/omnizip/formats/cpio/entry.rb +6 -0
- data/lib/omnizip/formats/cpio.rb +0 -6
- data/lib/omnizip/formats/gzip.rb +0 -7
- data/lib/omnizip/formats/iso/directory_record.rb +7 -0
- data/lib/omnizip/formats/iso.rb +0 -5
- data/lib/omnizip/formats/lzip.rb +0 -7
- data/lib/omnizip/formats/lzma_alone.rb +0 -6
- data/lib/omnizip/formats/msi/entry.rb +6 -0
- data/lib/omnizip/formats/msi.rb +0 -9
- data/lib/omnizip/formats/ole/dirent.rb +6 -0
- data/lib/omnizip/formats/ole.rb +0 -9
- data/lib/omnizip/formats/rar.rb +0 -6
- data/lib/omnizip/formats/rar3/reader.rb +7 -0
- data/lib/omnizip/formats/rar5/reader.rb +7 -0
- data/lib/omnizip/formats/rpm/entry.rb +6 -0
- data/lib/omnizip/formats/seven_zip/models/file_entry.rb +7 -0
- data/lib/omnizip/formats/seven_zip.rb +0 -6
- data/lib/omnizip/formats/tar/entry.rb +6 -0
- data/lib/omnizip/formats/tar.rb +2 -10
- data/lib/omnizip/formats/xar/entry.rb +6 -0
- data/lib/omnizip/formats/xar.rb +0 -6
- data/lib/omnizip/formats/zip.rb +0 -6
- data/lib/omnizip/implementations/seven_zip/lzma2/encoder.rb +33 -31
- data/lib/omnizip/implementations/xz_utils/lzma2/encoder.rb +177 -163
- data/lib/omnizip/implementations.rb +18 -4
- data/lib/omnizip/io/source.rb +3 -0
- data/lib/omnizip/metadata/entry_metadata.rb +7 -0
- data/lib/omnizip/parallel/engine.rb +48 -0
- data/lib/omnizip/parallel/parallel_compressor.rb +5 -16
- data/lib/omnizip/parallel/parallel_extractor.rb +5 -16
- data/lib/omnizip/parallel.rb +1 -0
- data/lib/omnizip/profile/profile_registry.rb +1 -2
- data/lib/omnizip/registry.rb +0 -1
- data/lib/omnizip/rubyzip_compat.rb +0 -1
- data/lib/omnizip/temp/temp_file.rb +1 -2
- data/lib/omnizip/version.rb +1 -1
- data/lib/omnizip/zip/entry.rb +7 -0
- data/lib/omnizip/zip/file.rb +4 -8
- data/lib/omnizip.rb +1 -1
- metadata +7 -6
- data/lib/omnizip/filters/filter_base.rb +0 -6
- data/lib/omnizip/filters/registration.rb +0 -22
- data/lib/omnizip/filters/registry.rb +0 -111
- data/lib/omnizip/format_registry.rb +0 -100
|
@@ -59,6 +59,9 @@ module Omnizip
|
|
|
59
59
|
UNCOMPRESSED_MAX = 1 << 21 # 2,097,152 bytes
|
|
60
60
|
# Maximum COMPRESSED size per chunk: 64KB
|
|
61
61
|
COMPRESSED_MAX = 1 << 16 # 65,536 bytes
|
|
62
|
+
# Uncompressed chunks store their size in a 16-bit field,
|
|
63
|
+
# so each uncompressed chunk is also limited to 64KB.
|
|
64
|
+
UNCOMPRESSED_CHUNK_MAX = 1 << 16
|
|
62
65
|
|
|
63
66
|
# Initialize the encoder
|
|
64
67
|
# @param options [Hash] Encoding options
|
|
@@ -96,7 +99,6 @@ module Omnizip
|
|
|
96
99
|
# CRITICAL: For XZ Utils compatibility, first chunk MUST reset the dictionary
|
|
97
100
|
# (matches XZ Utils behavior - see lzma2_encoder.c:334-336)
|
|
98
101
|
# need_dictionary_reset is set to true for the first compressed chunk
|
|
99
|
-
@need_properties = false # Properties will be written in first compressed chunk
|
|
100
102
|
@need_state_reset = false
|
|
101
103
|
@need_dictionary_reset = true # Always reset dictionary for first chunk (XZ Utils compatibility)
|
|
102
104
|
end
|
|
@@ -111,10 +113,8 @@ module Omnizip
|
|
|
111
113
|
|
|
112
114
|
# Write property byte if standalone mode (for .lz2 files)
|
|
113
115
|
# The property byte encodes dictionary size
|
|
114
|
-
# Formula: For power-of-2 sizes, d = 2 * (log2_size - 12)
|
|
115
116
|
if @standalone
|
|
116
|
-
|
|
117
|
-
output.putc(prop_byte)
|
|
117
|
+
output.putc(encode_dict_size(@dict_size))
|
|
118
118
|
end
|
|
119
119
|
|
|
120
120
|
input = StringIO.new(input_data)
|
|
@@ -125,12 +125,7 @@ module Omnizip
|
|
|
125
125
|
chunk_data = input.read(UNCOMPRESSED_MAX)
|
|
126
126
|
break if chunk_data.nil? || chunk_data.empty?
|
|
127
127
|
|
|
128
|
-
|
|
129
|
-
output.write(chunk.to_bytes)
|
|
130
|
-
|
|
131
|
-
@need_properties = false
|
|
132
|
-
@need_state_reset = false
|
|
133
|
-
@need_dictionary_reset = false
|
|
128
|
+
consume_chunk(chunk_data, output)
|
|
134
129
|
end
|
|
135
130
|
|
|
136
131
|
# End marker (0x00) is REQUIRED for all LZMA2 streams
|
|
@@ -151,60 +146,128 @@ module Omnizip
|
|
|
151
146
|
|
|
152
147
|
private
|
|
153
148
|
|
|
154
|
-
|
|
155
|
-
|
|
149
|
+
# Encode chunk_data (<= UNCOMPRESSED_MAX bytes), emitting one or
|
|
150
|
+
# more LZMA2 chunks.
|
|
151
|
+
#
|
|
152
|
+
# Multiple chunks are required when the compressed output reaches
|
|
153
|
+
# COMPRESSED_MAX (the compressed size is stored in a 16-bit field)
|
|
154
|
+
# or when compression is not beneficial and the data must be
|
|
155
|
+
# stored as uncompressed chunks (each capped at 64KB).
|
|
156
|
+
def consume_chunk(chunk_data, output)
|
|
157
|
+
@match_finder.feed(chunk_data)
|
|
158
|
+
@match_finder.skip(@match_finder.buffer.bytesize)
|
|
159
|
+
|
|
160
|
+
offset = 0
|
|
161
|
+
while offset < chunk_data.bytesize
|
|
162
|
+
slice = chunk_data.byteslice(offset, chunk_data.bytesize - offset)
|
|
163
|
+
|
|
164
|
+
# A state reset must apply BEFORE encoding the chunk that
|
|
165
|
+
# announces it (0xC0/0xE0 control byte), so that the encoder's
|
|
166
|
+
# models match the decoder's from the chunk's first bit.
|
|
167
|
+
reset_lzma_state! if @need_state_reset || @need_dictionary_reset
|
|
168
|
+
|
|
169
|
+
compressed, consumed = try_compress(slice)
|
|
170
|
+
|
|
171
|
+
use_compressed =
|
|
172
|
+
consumed.positive? &&
|
|
173
|
+
compressed.bytesize <= COMPRESSED_MAX &&
|
|
174
|
+
compressed.bytesize < (consumed == slice.bytesize ? slice.bytesize : consumed)
|
|
175
|
+
|
|
176
|
+
if use_compressed
|
|
177
|
+
uncompressed = chunk_data.byteslice(offset, consumed)
|
|
178
|
+
emit_compressed_chunk(output, uncompressed, compressed)
|
|
179
|
+
|
|
180
|
+
offset += consumed
|
|
181
|
+
@need_state_reset = false
|
|
182
|
+
else
|
|
183
|
+
# Compression didn't help (or hit the cap immediately):
|
|
184
|
+
# store the whole slice as uncompressed chunks.
|
|
185
|
+
emit_uncompressed_chunks(output, slice)
|
|
186
|
+
|
|
187
|
+
offset += slice.bytesize
|
|
188
|
+
# After an uncompressed chunk the LZMA state is lost, so the
|
|
189
|
+
# next compressed chunk must reset it (XZ Utils behavior,
|
|
190
|
+
# lzma2_encoder.c).
|
|
191
|
+
@need_state_reset = true
|
|
192
|
+
end
|
|
193
|
+
@need_dictionary_reset = false
|
|
194
|
+
end
|
|
195
|
+
|
|
196
|
+
# Update prev_byte for next chunk
|
|
197
|
+
if chunk_data.bytesize.positive?
|
|
198
|
+
@prev_byte = chunk_data.getbyte(chunk_data.bytesize - 1)
|
|
199
|
+
end
|
|
200
|
+
end
|
|
156
201
|
|
|
157
|
-
|
|
158
|
-
|
|
159
|
-
|
|
160
|
-
|
|
161
|
-
|
|
202
|
+
# Reset the LZMA state machine and probability models.
|
|
203
|
+
#
|
|
204
|
+
# MUST be called before encoding any chunk whose control byte
|
|
205
|
+
# announces a state reset (0xC0/0xE0), because the decoder will
|
|
206
|
+
# reinitialize its models on seeing that byte.
|
|
207
|
+
def reset_lzma_state!
|
|
208
|
+
@state = Omnizip::Algorithms::LZMA::LZMAState.new(0)
|
|
209
|
+
@models = Omnizip::Algorithms::LZMA::XzProbabilityModels.new(@lc,
|
|
210
|
+
@lp,
|
|
211
|
+
@pb)
|
|
212
|
+
@prev_byte = 0
|
|
213
|
+
end
|
|
214
|
+
|
|
215
|
+
# Emit a compressed LZMA2 chunk.
|
|
216
|
+
def emit_compressed_chunk(output, uncompressed, compressed)
|
|
217
|
+
# For compressed chunks, properties encode lc/lp/pb:
|
|
218
|
+
# (pb * 5 + lp) * 9 + lc
|
|
219
|
+
chunk_properties = (((@pb * 5) + @lp) * 9) + @lc
|
|
220
|
+
|
|
221
|
+
# A dictionary reset implies a state reset; props accompany
|
|
222
|
+
# every state reset (control 0xE0 or 0xC0).
|
|
223
|
+
state_reset = @need_state_reset || @need_dictionary_reset
|
|
224
|
+
|
|
225
|
+
chunk = Omnizip::Algorithms::LZMA2::LZMA2Chunk.new(
|
|
226
|
+
chunk_type: :compressed,
|
|
227
|
+
uncompressed_data: uncompressed,
|
|
228
|
+
compressed_data: compressed,
|
|
229
|
+
compressed_size: compressed.bytesize,
|
|
230
|
+
properties: chunk_properties,
|
|
231
|
+
need_dict_reset: @need_dictionary_reset,
|
|
232
|
+
need_state_reset: state_reset,
|
|
233
|
+
need_props: state_reset,
|
|
234
|
+
)
|
|
235
|
+
output.write(chunk.to_bytes)
|
|
162
236
|
|
|
163
|
-
|
|
164
|
-
|
|
237
|
+
# Update dictionary with the chunk data
|
|
238
|
+
@dictionary.append(uncompressed)
|
|
239
|
+
end
|
|
240
|
+
|
|
241
|
+
# Emit uncompressed LZMA2 chunks for data (splitting at the
|
|
242
|
+
# 64KB per-chunk limit imposed by the 16-bit size field).
|
|
243
|
+
def emit_uncompressed_chunks(output, data)
|
|
244
|
+
offset = 0
|
|
245
|
+
while offset < data.bytesize
|
|
246
|
+
slice = data.byteslice(offset, UNCOMPRESSED_CHUNK_MAX)
|
|
165
247
|
chunk = Omnizip::Algorithms::LZMA2::LZMA2Chunk.new(
|
|
166
248
|
chunk_type: :uncompressed,
|
|
167
|
-
uncompressed_data:
|
|
249
|
+
uncompressed_data: slice,
|
|
168
250
|
compressed_data: "",
|
|
169
251
|
need_dict_reset: @need_dictionary_reset,
|
|
170
252
|
need_state_reset: false,
|
|
171
253
|
need_props: false,
|
|
172
254
|
)
|
|
173
|
-
|
|
174
|
-
|
|
175
|
-
@need_state_reset = true
|
|
176
|
-
else
|
|
177
|
-
# Use compressed chunk (compression helped)
|
|
178
|
-
# For compressed chunks, properties encode lc/lp/pb:
|
|
179
|
-
# (pb * 5 + lp) * 9 + lc
|
|
180
|
-
chunk_properties = (((@pb * 5) + @lp) * 9) + @lc
|
|
181
|
-
# CRITICAL: need_props must be TRUE when we're providing properties!
|
|
182
|
-
# This tells the chunk to encode properties in the control byte
|
|
183
|
-
# CRITICAL: compressed_size includes ALL bytes (LZMA data + flush bytes)
|
|
184
|
-
# The flush bytes are part of the range encoder output and must be included
|
|
185
|
-
chunk = Omnizip::Algorithms::LZMA2::LZMA2Chunk.new(
|
|
186
|
-
chunk_type: :compressed,
|
|
187
|
-
uncompressed_data: uncompressed_data,
|
|
188
|
-
compressed_data: compressed,
|
|
189
|
-
compressed_size: compressed.bytesize, # Full size including flush bytes
|
|
190
|
-
properties: chunk_properties,
|
|
191
|
-
need_dict_reset: @need_dictionary_reset,
|
|
192
|
-
need_state_reset: @need_state_reset,
|
|
193
|
-
need_props: true, # Always true for compressed chunks with properties
|
|
194
|
-
)
|
|
195
|
-
end
|
|
196
|
-
|
|
197
|
-
# Update dictionary with the chunk data (done once per chunk)
|
|
198
|
-
@dictionary.append(uncompressed_data)
|
|
255
|
+
output.write(chunk.to_bytes)
|
|
256
|
+
@need_dictionary_reset = false
|
|
199
257
|
|
|
200
|
-
|
|
201
|
-
|
|
202
|
-
@prev_byte = uncompressed_data.getbyte(uncompressed_data.bytesize - 1)
|
|
258
|
+
offset += slice.bytesize
|
|
259
|
+
@dictionary.append(slice)
|
|
203
260
|
end
|
|
204
|
-
|
|
205
|
-
chunk
|
|
206
261
|
end
|
|
207
262
|
|
|
263
|
+
# Compress data starting at the current dictionary position.
|
|
264
|
+
#
|
|
265
|
+
# Stops early (returning fewer bytes than data.bytesize) when the
|
|
266
|
+
# compressed output is about to exceed COMPRESSED_MAX, leaving the
|
|
267
|
+
# range encoder at a clean item boundary.
|
|
268
|
+
#
|
|
269
|
+
# @return [Array<String, Integer>] compressed bytes and number of
|
|
270
|
+
# input bytes consumed
|
|
208
271
|
def try_compress(data)
|
|
209
272
|
# Create output buffer to capture compressed data
|
|
210
273
|
output_buffer = StringIO.new
|
|
@@ -213,39 +276,26 @@ module Omnizip
|
|
|
213
276
|
# Create range encoder (direct XZ Utils port)
|
|
214
277
|
encoder = Omnizip::Algorithms::LZMA::XzRangeEncoder.new(output_buffer)
|
|
215
278
|
|
|
216
|
-
#
|
|
217
|
-
#
|
|
218
|
-
@match_finder.feed(data)
|
|
219
|
-
|
|
220
|
-
# CRITICAL: Initialize hash table for positions BEFORE encoding starts
|
|
221
|
-
# This ensures that matches can be found for repeated data patterns
|
|
222
|
-
# Matches XZ Utils lzma_encoder.c: mf_skip() behavior
|
|
223
|
-
# We skip to position (start_pos + data.bytesize - MATCH_LEN_MAX),
|
|
224
|
-
# but ensure we don't go negative for small inputs
|
|
225
|
-
match_len_max = 2 # Minimum match length in LZMA2
|
|
226
|
-
end_pos = [
|
|
227
|
-
@dictionary.buffer.bytesize + data.bytesize - match_len_max, 0
|
|
228
|
-
].max
|
|
229
|
-
@match_finder.skip(end_pos)
|
|
230
|
-
|
|
231
|
-
# Position in match finder's buffer for encoding
|
|
232
|
-
# Start after the data we just fed
|
|
279
|
+
# The whole chunk has already been fed to the match finder by
|
|
280
|
+
# consume_chunk; positions map from the dictionary position.
|
|
233
281
|
start_pos = @dictionary.buffer.bytesize
|
|
234
282
|
|
|
235
|
-
#
|
|
236
|
-
|
|
283
|
+
# Headroom for pending symbols plus the final flush
|
|
284
|
+
size_cap = COMPRESSED_MAX - 64
|
|
237
285
|
|
|
238
286
|
pos = 0
|
|
239
287
|
while pos < data.bytesize
|
|
240
|
-
#
|
|
241
|
-
#
|
|
242
|
-
#
|
|
243
|
-
|
|
244
|
-
|
|
245
|
-
|
|
288
|
+
# Drain before every item: a single worst-case match (long
|
|
289
|
+
# length + dist slot >= 14 with up to 30 footer bits) queues
|
|
290
|
+
# ~48 symbols, and RC_SYMBOLS_MAX is 53, so the queue MUST be
|
|
291
|
+
# empty when an item starts or `bit` raises "Symbol buffer
|
|
292
|
+
# overflow" (issue #26).
|
|
293
|
+
encode_queued_symbols(encoder, output_buffer)
|
|
294
|
+
|
|
295
|
+
# Enforce the 64KB compressed-chunk cap at an item boundary
|
|
296
|
+
break if pos.positive? && output_buffer.size >= size_cap
|
|
246
297
|
|
|
247
298
|
# Position in match finder's buffer for encoding
|
|
248
|
-
# Start after the data we just fed
|
|
249
299
|
match_pos = start_pos + pos
|
|
250
300
|
|
|
251
301
|
# Get optimal encoding choice
|
|
@@ -265,21 +315,21 @@ module Omnizip
|
|
|
265
315
|
# CRITICAL: Use UINT32_MAX to check for literal (not distance.zero?)
|
|
266
316
|
# because distance=0 means repeated match rep0, not literal!
|
|
267
317
|
if distance == UINT32_MAX || length == 1
|
|
268
|
-
# Encode literal
|
|
269
|
-
#
|
|
270
|
-
|
|
318
|
+
# Encode literal. The position argument must be GLOBAL
|
|
319
|
+
# (dictionary position), because the decoder keeps counting
|
|
320
|
+
# bytes across chunk boundaries for pos_state.
|
|
321
|
+
encode_literal(data.getbyte(pos), encoder, match_pos)
|
|
271
322
|
pos += 1
|
|
272
323
|
elsif distance < REPS
|
|
273
324
|
# Encode repeated match (distance is 0-3 for rep0-rep3)
|
|
274
|
-
|
|
275
|
-
|
|
325
|
+
encode_repeated_match(distance, length, encoder, match_pos,
|
|
326
|
+
match_pos)
|
|
276
327
|
pos += length
|
|
277
328
|
else
|
|
278
329
|
# Encode normal match (distance is actual_distance + REPS)
|
|
279
330
|
actual_distance = distance - REPS
|
|
280
|
-
|
|
281
|
-
|
|
282
|
-
data)
|
|
331
|
+
encode_match(actual_distance, length, encoder, match_pos,
|
|
332
|
+
match_pos, data)
|
|
283
333
|
pos += length
|
|
284
334
|
end
|
|
285
335
|
end
|
|
@@ -295,46 +345,12 @@ module Omnizip
|
|
|
295
345
|
# This will write additional bytes to output_buffer
|
|
296
346
|
encode_queued_symbols(encoder, output_buffer)
|
|
297
347
|
|
|
298
|
-
|
|
299
|
-
full_output = output_buffer.string
|
|
300
|
-
|
|
301
|
-
puts "[DEBUG] try_compress: full_output.size=#{full_output.bytesize}, encoder.out_total=#{encoder.out_total}" if ENV["DEBUG_FLUSH"]
|
|
302
|
-
|
|
303
|
-
# Return all bytes (flush bytes are part of the LZMA data)
|
|
304
|
-
full_output
|
|
305
|
-
end
|
|
306
|
-
|
|
307
|
-
# Encode queued symbols to output
|
|
308
|
-
# rubocop:disable Style/CollectionQuerying
|
|
309
|
-
def encode_queued_symbols(encoder, output)
|
|
310
|
-
return if encoder.count.zero?
|
|
311
|
-
|
|
312
|
-
# Encode symbols to buffer
|
|
313
|
-
encoder.encode_symbols(temp_buffer, out_pos, 10000)
|
|
314
|
-
|
|
315
|
-
# Track size before encoding
|
|
316
|
-
size_before = output.size
|
|
317
|
-
|
|
318
|
-
# Encode symbols to buffer
|
|
319
|
-
encoder.encode_symbols(temp_buffer, out_pos, 10000)
|
|
320
|
-
|
|
321
|
-
# Write to output stream
|
|
322
|
-
if out_pos.value.positive?
|
|
323
|
-
# Use StringCompat.byteslice for Ruby 3.0-3.1 compatibility
|
|
324
|
-
# Ruby's [] operator has a bug with null bytes that can return extra bytes
|
|
325
|
-
# See: https://bugs.ruby-lang.org/issues/15985
|
|
326
|
-
output.write(StringCompat.byteslice(temp_buffer, 0,
|
|
327
|
-
out_pos.value))
|
|
328
|
-
end
|
|
329
|
-
|
|
330
|
-
# Return the number of bytes written
|
|
331
|
-
output.size - size_before
|
|
348
|
+
[output_buffer.string, pos]
|
|
332
349
|
end
|
|
333
350
|
|
|
334
351
|
# Encode queued symbols to output
|
|
335
|
-
# rubocop:disable Style/CollectionQuerying
|
|
336
352
|
def encode_queued_symbols(encoder, output)
|
|
337
|
-
return if encoder.
|
|
353
|
+
return if encoder.none?
|
|
338
354
|
|
|
339
355
|
# Create temporary buffer for encoding
|
|
340
356
|
temp_buffer = "\0" * 10000
|
|
@@ -360,6 +376,8 @@ module Omnizip
|
|
|
360
376
|
end
|
|
361
377
|
|
|
362
378
|
# Encode literal byte
|
|
379
|
+
# pos is the GLOBAL position (dictionary position) because the
|
|
380
|
+
# decoder counts bytes continuously across chunks.
|
|
363
381
|
def encode_literal(symbol, encoder, pos)
|
|
364
382
|
pos_state = pos & ((1 << @pb) - 1)
|
|
365
383
|
|
|
@@ -382,8 +400,7 @@ module Omnizip
|
|
|
382
400
|
# Matched literal (compare with match byte at rep0)
|
|
383
401
|
# XZ Utils: mf->buffer[mf->read_pos - coder->reps[0] - 1 - mf->read_ahead]
|
|
384
402
|
# We don't use read_ahead, so it's 0
|
|
385
|
-
|
|
386
|
-
match_byte_pos = match_pos - @state.reps[0] - 1
|
|
403
|
+
match_byte_pos = pos - @state.reps[0] - 1
|
|
387
404
|
match_byte = @match_finder.buffer.getbyte(match_byte_pos) if match_byte_pos >= 0 && match_byte_pos < @match_finder.buffer.bytesize
|
|
388
405
|
|
|
389
406
|
# If match_byte is nil (shouldn't happen in normal operation),
|
|
@@ -417,8 +434,9 @@ _input_data)
|
|
|
417
434
|
encoder.queue_bit(prob_is_rep, 0)
|
|
418
435
|
|
|
419
436
|
# CRITICAL: Update state BEFORE encoding length/distance (XZ Utils order)
|
|
420
|
-
# This also updates reps
|
|
421
|
-
|
|
437
|
+
# This also updates reps. Reps are stored 0-based (distance - 1),
|
|
438
|
+
# matching XZ Utils and the decoder's rep0 convention.
|
|
439
|
+
@state.update_match!(distance - 1)
|
|
422
440
|
|
|
423
441
|
# Encode length - uses NEW state value
|
|
424
442
|
encode_match_length(length, pos_state, encoder)
|
|
@@ -466,27 +484,15 @@ _input_data)
|
|
|
466
484
|
|
|
467
485
|
prob_is_rep2 = @models.is_rep2[@state.value]
|
|
468
486
|
encoder.queue_bit(prob_is_rep2, rep - 2)
|
|
469
|
-
|
|
470
|
-
if rep == 3
|
|
471
|
-
# Update reps[3] = reps[2] before updating reps[2]
|
|
472
|
-
@state.reps[3] = @state.reps[2]
|
|
473
|
-
end
|
|
474
|
-
|
|
475
|
-
# Update reps[2] = reps[1]
|
|
476
|
-
@state.reps[2] = @state.reps[1]
|
|
477
487
|
end
|
|
478
488
|
|
|
479
|
-
#
|
|
480
|
-
|
|
481
|
-
|
|
482
|
-
#
|
|
489
|
+
# XZ Utils rep_match: capture the selected distance BEFORE
|
|
490
|
+
# rotating, then shift reps down (lzma_encoder.c):
|
|
491
|
+
# distance = reps[rep];
|
|
492
|
+
# for (i = rep; i > 0; --i) reps[i] = reps[i-1];
|
|
493
|
+
# reps[0] = distance;
|
|
483
494
|
distance = @state.reps[rep]
|
|
484
|
-
|
|
485
|
-
# Defensive check: distance should never be nil
|
|
486
|
-
if distance.nil?
|
|
487
|
-
raise "Distance is nil for rep #{rep}, reps=#{@state.reps.inspect}"
|
|
488
|
-
end
|
|
489
|
-
|
|
495
|
+
rep.downto(1) { |i| @state.reps[i] = @state.reps[i - 1] }
|
|
490
496
|
@state.reps[0] = distance
|
|
491
497
|
end
|
|
492
498
|
|
|
@@ -494,15 +500,15 @@ _input_data)
|
|
|
494
500
|
if length == 1
|
|
495
501
|
@state.update_short_rep!
|
|
496
502
|
else
|
|
497
|
-
# Encode length
|
|
498
|
-
|
|
503
|
+
# Encode length using the REP length coder (rep matches have
|
|
504
|
+
# their own length models, decoder's @rep_length_coder)
|
|
505
|
+
encode_match_length(length, pos_state, encoder,
|
|
506
|
+
@models.rep_len_encoder)
|
|
499
507
|
@state.update_long_rep!
|
|
500
508
|
end
|
|
501
509
|
|
|
502
|
-
# Update prev_byte (last byte of match)
|
|
503
|
-
|
|
504
|
-
# But after updating reps above, reps[0] now contains the distance
|
|
505
|
-
last_byte_pos = match_pos - @state.reps[0] + length - 1
|
|
510
|
+
# Update prev_byte (last byte of the match region in the stream)
|
|
511
|
+
last_byte_pos = match_pos + length - 1
|
|
506
512
|
@prev_byte = @match_finder.buffer.getbyte(last_byte_pos) if last_byte_pos >= 0 && last_byte_pos < @match_finder.buffer.bytesize
|
|
507
513
|
end
|
|
508
514
|
|
|
@@ -513,12 +519,13 @@ _input_data)
|
|
|
513
519
|
# (((((pos) << 8) + (prev_byte)) & (literal_mask)) << (lc)))
|
|
514
520
|
# where literal_mask = (0x100 << lp) - (0x100 >> lc)
|
|
515
521
|
#
|
|
522
|
+
# The factor of 3 gives each context a 0x300-sized subcoder and
|
|
523
|
+
# MUST match the decoder's base_offset = 3 * (lit_state << lc).
|
|
524
|
+
#
|
|
516
525
|
# Returns the flat index into the literal probability array.
|
|
517
|
-
# The literal array is now a flat array (matching XZ Utils) with
|
|
518
|
-
# size 0x300 << (lc + lp), not a 2D array.
|
|
519
526
|
def get_literal_state(pos, prev_byte)
|
|
520
527
|
literal_mask = (0x100 << @lp) - (0x100 >> @lc)
|
|
521
|
-
((((pos << 8) + prev_byte) & literal_mask) << @lc)
|
|
528
|
+
3 * ((((pos << 8) + prev_byte) & literal_mask) << @lc)
|
|
522
529
|
end
|
|
523
530
|
|
|
524
531
|
# Get byte from dictionary at distance back
|
|
@@ -570,35 +577,38 @@ encoder)
|
|
|
570
577
|
end
|
|
571
578
|
|
|
572
579
|
# Encode match length
|
|
573
|
-
|
|
580
|
+
# len_encoder is the length model set to use: match_len_encoder for
|
|
581
|
+
# normal matches, rep_len_encoder for repeated matches.
|
|
582
|
+
def encode_match_length(length, pos_state, encoder,
|
|
583
|
+
len_encoder = @models.match_len_encoder)
|
|
574
584
|
len = length - MATCH_LEN_MIN
|
|
575
585
|
|
|
576
586
|
if len < LEN_LOW_SYMBOLS
|
|
577
587
|
# Low: 0-7
|
|
578
|
-
encoder.queue_bit(
|
|
588
|
+
encoder.queue_bit(len_encoder.choice, 0)
|
|
579
589
|
encode_bittree(
|
|
580
|
-
|
|
590
|
+
len_encoder.low[pos_state],
|
|
581
591
|
NUM_LEN_LOW_BITS,
|
|
582
592
|
len,
|
|
583
593
|
encoder,
|
|
584
594
|
)
|
|
585
595
|
elsif len < LEN_LOW_SYMBOLS + LEN_MID_SYMBOLS
|
|
586
596
|
# Mid: 8-15
|
|
587
|
-
encoder.queue_bit(
|
|
588
|
-
encoder.queue_bit(
|
|
597
|
+
encoder.queue_bit(len_encoder.choice, 1)
|
|
598
|
+
encoder.queue_bit(len_encoder.choice2, 0)
|
|
589
599
|
encode_bittree(
|
|
590
|
-
|
|
600
|
+
len_encoder.mid[pos_state],
|
|
591
601
|
NUM_LEN_MID_BITS,
|
|
592
602
|
len - LEN_LOW_SYMBOLS,
|
|
593
603
|
encoder,
|
|
594
604
|
)
|
|
595
605
|
else
|
|
596
606
|
# High: 16-271
|
|
597
|
-
encoder.queue_bit(
|
|
598
|
-
encoder.queue_bit(
|
|
607
|
+
encoder.queue_bit(len_encoder.choice, 1)
|
|
608
|
+
encoder.queue_bit(len_encoder.choice2, 1)
|
|
599
609
|
high_len = len - LEN_LOW_SYMBOLS - LEN_MID_SYMBOLS
|
|
600
610
|
encode_bittree(
|
|
601
|
-
|
|
611
|
+
len_encoder.high,
|
|
602
612
|
NUM_LEN_HIGH_BITS,
|
|
603
613
|
high_len,
|
|
604
614
|
encoder,
|
|
@@ -608,7 +618,11 @@ encoder)
|
|
|
608
618
|
|
|
609
619
|
# Encode distance using slot encoding
|
|
610
620
|
def encode_distance(distance, length, encoder)
|
|
611
|
-
|
|
621
|
+
# get_dist_slot expects a 0-based distance (xz fastpos.h:
|
|
622
|
+
# dist 4 -> slot 4, dist 5 -> slot 4). The decoder computes
|
|
623
|
+
# rep0 = base + footer the same 0-based way.
|
|
624
|
+
dist0 = distance - 1
|
|
625
|
+
dist_slot = get_dist_slot(dist0)
|
|
612
626
|
len_state = get_len_to_pos_state(length)
|
|
613
627
|
|
|
614
628
|
# Encode distance slot
|
|
@@ -624,7 +638,7 @@ encoder)
|
|
|
624
638
|
if dist_slot >= START_POS_MODEL_INDEX
|
|
625
639
|
footer_bits = (dist_slot >> 1) - 1
|
|
626
640
|
base = (2 | (dist_slot & 1)) << footer_bits
|
|
627
|
-
dist_reduced =
|
|
641
|
+
dist_reduced = dist0 - base
|
|
628
642
|
|
|
629
643
|
if dist_slot < END_POS_MODEL_INDEX
|
|
630
644
|
# Use probability models
|
|
@@ -1,11 +1,25 @@
|
|
|
1
1
|
# frozen_string_literal: true
|
|
2
2
|
|
|
3
3
|
module Omnizip
|
|
4
|
+
# Bit-level SDK ports: state machines, range coders, match finders,
|
|
5
|
+
# raw LZMA/LZMA2 encoders and decoders.
|
|
6
|
+
#
|
|
7
|
+
# This namespace owns the *primitives*; +Omnizip::Algorithms+ owns
|
|
8
|
+
# the *streaming contract*. The boundary is: +Algorithms+ calls
|
|
9
|
+
# +Implementations+, never the reverse. Files under this tree
|
|
10
|
+
# should never reach across into format-specific concerns (no
|
|
11
|
+
# .7z header parsing, no ZIP central directory, no tar entry
|
|
12
|
+
# metadata).
|
|
13
|
+
#
|
|
14
|
+
# The two top-level sub-namespaces mirror the upstream SDK split:
|
|
15
|
+
# - +SevenZip+ : ports of 7-Zip's LZMA/LZMA2 encoder + decoder
|
|
16
|
+
# - +XZUtils+ : ports of XZ Utils' LZMA/LZMA2 encoder + decoder
|
|
17
|
+
#
|
|
18
|
+
# Despite the LZMA name, the two SDKs diverge in subtle ways (no
|
|
19
|
+
# EOS marker, different chunk padding, different property-byte
|
|
20
|
+
# handling); keeping them in separate trees makes those
|
|
21
|
+
# divergences visible rather than hidden in flag parameters.
|
|
4
22
|
module Implementations
|
|
5
|
-
# Implementation namespace for various algorithm ports
|
|
6
|
-
# This module contains reference implementations ported from external projects
|
|
7
|
-
|
|
8
|
-
# Base classes for implementation inheritance
|
|
9
23
|
autoload :Base, "omnizip/implementations/base"
|
|
10
24
|
end
|
|
11
25
|
end
|
data/lib/omnizip/io/source.rb
CHANGED
|
@@ -5,8 +5,15 @@ module Omnizip
|
|
|
5
5
|
# Model for file entry metadata
|
|
6
6
|
# Wraps CentralDirectoryHeader with a cleaner metadata API
|
|
7
7
|
class EntryMetadata
|
|
8
|
+
include Omnizip::Entry
|
|
9
|
+
|
|
8
10
|
attr_reader :entry
|
|
9
11
|
|
|
12
|
+
def entry_name = entry.name
|
|
13
|
+
def entry_directory? = entry.directory?
|
|
14
|
+
def entry_size = entry.size
|
|
15
|
+
def entry_mtime = entry.time
|
|
16
|
+
|
|
10
17
|
# Initialize metadata for an entry
|
|
11
18
|
# @param entry [Omnizip::Zip::Entry] The entry to manage metadata for
|
|
12
19
|
def initialize(entry)
|
|
@@ -0,0 +1,48 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require "fractor"
|
|
4
|
+
|
|
5
|
+
module Omnizip
|
|
6
|
+
module Parallel
|
|
7
|
+
# Engine owns the Fractor worker-pool dance: build the pool, submit
|
|
8
|
+
# a batch of work, run it to completion, return successful and
|
|
9
|
+
# failed results. Callers (ParallelCompressor, ParallelExtractor,
|
|
10
|
+
# future operations) supply the worker class and the work items;
|
|
11
|
+
# Engine handles the threading.
|
|
12
|
+
#
|
|
13
|
+
# This is the deep module behind the parallel API: the threading
|
|
14
|
+
# shape is identical across every consumer, so it lives here once.
|
|
15
|
+
# Domain-specific concerns (which file to read, which archive to
|
|
16
|
+
# open, what stats to track) stay with the caller.
|
|
17
|
+
class Engine
|
|
18
|
+
# @param worker_class [Class] a +Fractor::Worker+ subclass whose
|
|
19
|
+
# +#process(work)+ turns one work item into a result.
|
|
20
|
+
# @param threads [Integer] worker count.
|
|
21
|
+
def initialize(worker_class:, threads:)
|
|
22
|
+
@worker_class = worker_class
|
|
23
|
+
@threads = threads
|
|
24
|
+
end
|
|
25
|
+
|
|
26
|
+
# Run +work_items+ through the pool and yield each successful
|
|
27
|
+
# +Fractor::WorkResult+ to the caller-supplied block.
|
|
28
|
+
#
|
|
29
|
+
# @param work_items [Array<Fractor::Work>] batch to submit.
|
|
30
|
+
# @yieldparam result [Fractor::WorkResult] a successful result.
|
|
31
|
+
# @return [Array<Fractor::WorkResult>] the failed results, for
|
|
32
|
+
# the caller to decide how to surface.
|
|
33
|
+
def run(work_items)
|
|
34
|
+
pool = Fractor::WorkerPool.new(
|
|
35
|
+
worker_class: @worker_class,
|
|
36
|
+
num_workers: @threads,
|
|
37
|
+
continuous: false,
|
|
38
|
+
)
|
|
39
|
+
pool.start
|
|
40
|
+
pool.submit_batch(work_items)
|
|
41
|
+
pool.run
|
|
42
|
+
|
|
43
|
+
pool.successful_results.each { |r| yield r if block_given? }
|
|
44
|
+
pool.failed_results
|
|
45
|
+
end
|
|
46
|
+
end
|
|
47
|
+
end
|
|
48
|
+
end
|
|
@@ -212,20 +212,11 @@ level: 6)
|
|
|
212
212
|
)
|
|
213
213
|
end
|
|
214
214
|
|
|
215
|
-
#
|
|
216
|
-
|
|
217
|
-
|
|
218
|
-
|
|
219
|
-
|
|
220
|
-
)
|
|
221
|
-
|
|
222
|
-
pool.start
|
|
223
|
-
pool.submit_batch(work_items)
|
|
224
|
-
pool.run
|
|
225
|
-
|
|
226
|
-
# Collect results
|
|
227
|
-
results = pool.successful_results
|
|
228
|
-
errors = pool.failed_results
|
|
215
|
+
# Run the worker pool through the shared engine.
|
|
216
|
+
engine = Engine.new(worker_class: CompressionWorker,
|
|
217
|
+
threads: @options.threads)
|
|
218
|
+
results = []
|
|
219
|
+
errors = engine.run(work_items) { |r| results << r }
|
|
229
220
|
|
|
230
221
|
# Handle errors
|
|
231
222
|
unless errors.empty?
|
|
@@ -238,8 +229,6 @@ level: 6)
|
|
|
238
229
|
# Write archive sequentially (thread-safe)
|
|
239
230
|
write_archive(output, results, compression: compression)
|
|
240
231
|
|
|
241
|
-
pool.shutdown
|
|
242
|
-
|
|
243
232
|
@stats[:end_time] = Time.now
|
|
244
233
|
@stats[:files_processed] = results.size
|
|
245
234
|
|