omnizip 0.3.13 → 0.3.15

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (88) hide show
  1. checksums.yaml +4 -4
  2. data/lib/omnizip/algorithm.rb +36 -1
  3. data/lib/omnizip/algorithms/bzip2.rb +1 -3
  4. data/lib/omnizip/algorithms/deflate.rb +1 -3
  5. data/lib/omnizip/algorithms/deflate64.rb +2 -2
  6. data/lib/omnizip/algorithms/lzma/optimal_encoder.rb +7 -8
  7. data/lib/omnizip/algorithms/lzma.rb +1 -1
  8. data/lib/omnizip/algorithms/lzma2/lzma2_chunk.rb +5 -1
  9. data/lib/omnizip/algorithms/lzma2.rb +4 -12
  10. data/lib/omnizip/algorithms/zstandard/constants.rb +38 -25
  11. data/lib/omnizip/algorithms/zstandard/decoder.rb +101 -136
  12. data/lib/omnizip/algorithms/zstandard/encoder.rb +81 -138
  13. data/lib/omnizip/algorithms/zstandard/frame/block.rb +17 -0
  14. data/lib/omnizip/algorithms/zstandard/frame/header.rb +42 -29
  15. data/lib/omnizip/algorithms/zstandard/fse/bitstream.rb +114 -117
  16. data/lib/omnizip/algorithms/zstandard/fse/encoder.rb +434 -222
  17. data/lib/omnizip/algorithms/zstandard/fse/interleaved.rb +92 -0
  18. data/lib/omnizip/algorithms/zstandard/fse/table.rb +98 -176
  19. data/lib/omnizip/algorithms/zstandard/fse/table_description.rb +203 -0
  20. data/lib/omnizip/algorithms/zstandard/fse.rb +21 -2
  21. data/lib/omnizip/algorithms/zstandard/huffman.rb +195 -188
  22. data/lib/omnizip/algorithms/zstandard/huffman_encoder.rb +232 -255
  23. data/lib/omnizip/algorithms/zstandard/literals.rb +170 -103
  24. data/lib/omnizip/algorithms/zstandard/literals_encoder.rb +47 -199
  25. data/lib/omnizip/algorithms/zstandard/sequences.rb +260 -252
  26. data/lib/omnizip/algorithms/zstandard/xxhash.rb +159 -0
  27. data/lib/omnizip/algorithms/zstandard.rb +20 -24
  28. data/lib/omnizip/algorithms.rb +14 -1
  29. data/lib/omnizip/buffer.rb +1 -2
  30. data/lib/omnizip/commands/archive_list_command.rb +9 -7
  31. data/lib/omnizip/convenience.rb +1 -2
  32. data/lib/omnizip/entry.rb +44 -0
  33. data/lib/omnizip/extraction/selective_extractor.rb +2 -10
  34. data/lib/omnizip/filter_pipeline.rb +1 -1
  35. data/lib/omnizip/filter_registry.rb +18 -16
  36. data/lib/omnizip/filters/bcj2.rb +1 -1
  37. data/lib/omnizip/filters/bcj_arm.rb +1 -1
  38. data/lib/omnizip/filters/bcj_arm64.rb +1 -1
  39. data/lib/omnizip/filters/bcj_ia64.rb +1 -1
  40. data/lib/omnizip/filters/bcj_ppc.rb +1 -1
  41. data/lib/omnizip/filters/bcj_sparc.rb +1 -1
  42. data/lib/omnizip/filters/bcj_x86.rb +1 -1
  43. data/lib/omnizip/filters.rb +0 -2
  44. data/lib/omnizip/formats/bzip2_file.rb +0 -7
  45. data/lib/omnizip/formats/cpio/entry.rb +6 -0
  46. data/lib/omnizip/formats/cpio.rb +0 -6
  47. data/lib/omnizip/formats/gzip.rb +0 -7
  48. data/lib/omnizip/formats/iso/directory_record.rb +7 -0
  49. data/lib/omnizip/formats/iso.rb +0 -5
  50. data/lib/omnizip/formats/lzip.rb +0 -7
  51. data/lib/omnizip/formats/lzma_alone.rb +0 -6
  52. data/lib/omnizip/formats/msi/entry.rb +6 -0
  53. data/lib/omnizip/formats/msi.rb +0 -9
  54. data/lib/omnizip/formats/ole/dirent.rb +6 -0
  55. data/lib/omnizip/formats/ole.rb +0 -9
  56. data/lib/omnizip/formats/rar.rb +0 -6
  57. data/lib/omnizip/formats/rar3/reader.rb +7 -0
  58. data/lib/omnizip/formats/rar5/reader.rb +7 -0
  59. data/lib/omnizip/formats/rpm/entry.rb +6 -0
  60. data/lib/omnizip/formats/seven_zip/models/file_entry.rb +7 -0
  61. data/lib/omnizip/formats/seven_zip.rb +0 -6
  62. data/lib/omnizip/formats/tar/entry.rb +6 -0
  63. data/lib/omnizip/formats/tar.rb +2 -10
  64. data/lib/omnizip/formats/xar/entry.rb +6 -0
  65. data/lib/omnizip/formats/xar.rb +0 -6
  66. data/lib/omnizip/formats/zip.rb +0 -6
  67. data/lib/omnizip/implementations/seven_zip/lzma2/encoder.rb +33 -31
  68. data/lib/omnizip/implementations/xz_utils/lzma2/encoder.rb +177 -163
  69. data/lib/omnizip/implementations.rb +18 -4
  70. data/lib/omnizip/io/source.rb +3 -0
  71. data/lib/omnizip/metadata/entry_metadata.rb +7 -0
  72. data/lib/omnizip/parallel/engine.rb +48 -0
  73. data/lib/omnizip/parallel/parallel_compressor.rb +5 -16
  74. data/lib/omnizip/parallel/parallel_extractor.rb +5 -16
  75. data/lib/omnizip/parallel.rb +1 -0
  76. data/lib/omnizip/profile/profile_registry.rb +1 -2
  77. data/lib/omnizip/registry.rb +0 -1
  78. data/lib/omnizip/rubyzip_compat.rb +0 -1
  79. data/lib/omnizip/temp/temp_file.rb +1 -2
  80. data/lib/omnizip/version.rb +1 -1
  81. data/lib/omnizip/zip/entry.rb +7 -0
  82. data/lib/omnizip/zip/file.rb +4 -8
  83. data/lib/omnizip.rb +1 -1
  84. metadata +7 -6
  85. data/lib/omnizip/filters/filter_base.rb +0 -6
  86. data/lib/omnizip/filters/registration.rb +0 -22
  87. data/lib/omnizip/filters/registry.rb +0 -111
  88. data/lib/omnizip/format_registry.rb +0 -100
@@ -59,6 +59,9 @@ module Omnizip
59
59
  UNCOMPRESSED_MAX = 1 << 21 # 2,097,152 bytes
60
60
  # Maximum COMPRESSED size per chunk: 64KB
61
61
  COMPRESSED_MAX = 1 << 16 # 65,536 bytes
62
+ # Uncompressed chunks store their size in a 16-bit field,
63
+ # so each uncompressed chunk is also limited to 64KB.
64
+ UNCOMPRESSED_CHUNK_MAX = 1 << 16
62
65
 
63
66
  # Initialize the encoder
64
67
  # @param options [Hash] Encoding options
@@ -96,7 +99,6 @@ module Omnizip
96
99
  # CRITICAL: For XZ Utils compatibility, first chunk MUST reset the dictionary
97
100
  # (matches XZ Utils behavior - see lzma2_encoder.c:334-336)
98
101
  # need_dictionary_reset is set to true for the first compressed chunk
99
- @need_properties = false # Properties will be written in first compressed chunk
100
102
  @need_state_reset = false
101
103
  @need_dictionary_reset = true # Always reset dictionary for first chunk (XZ Utils compatibility)
102
104
  end
@@ -111,10 +113,8 @@ module Omnizip
111
113
 
112
114
  # Write property byte if standalone mode (for .lz2 files)
113
115
  # The property byte encodes dictionary size
114
- # Formula: For power-of-2 sizes, d = 2 * (log2_size - 12)
115
116
  if @standalone
116
- prop_byte = encode_dict_size(@dict_size)
117
- output.putc(prop_byte)
117
+ output.putc(encode_dict_size(@dict_size))
118
118
  end
119
119
 
120
120
  input = StringIO.new(input_data)
@@ -125,12 +125,7 @@ module Omnizip
125
125
  chunk_data = input.read(UNCOMPRESSED_MAX)
126
126
  break if chunk_data.nil? || chunk_data.empty?
127
127
 
128
- chunk = encode_chunk(chunk_data)
129
- output.write(chunk.to_bytes)
130
-
131
- @need_properties = false
132
- @need_state_reset = false
133
- @need_dictionary_reset = false
128
+ consume_chunk(chunk_data, output)
134
129
  end
135
130
 
136
131
  # End marker (0x00) is REQUIRED for all LZMA2 streams
@@ -151,60 +146,128 @@ module Omnizip
151
146
 
152
147
  private
153
148
 
154
- def encode_chunk(uncompressed_data)
155
- compressed = try_compress(uncompressed_data)
149
+ # Encode chunk_data (<= UNCOMPRESSED_MAX bytes), emitting one or
150
+ # more LZMA2 chunks.
151
+ #
152
+ # Multiple chunks are required when the compressed output reaches
153
+ # COMPRESSED_MAX (the compressed size is stored in a 16-bit field)
154
+ # or when compression is not beneficial and the data must be
155
+ # stored as uncompressed chunks (each capped at 64KB).
156
+ def consume_chunk(chunk_data, output)
157
+ @match_finder.feed(chunk_data)
158
+ @match_finder.skip(@match_finder.buffer.bytesize)
159
+
160
+ offset = 0
161
+ while offset < chunk_data.bytesize
162
+ slice = chunk_data.byteslice(offset, chunk_data.bytesize - offset)
163
+
164
+ # A state reset must apply BEFORE encoding the chunk that
165
+ # announces it (0xC0/0xE0 control byte), so that the encoder's
166
+ # models match the decoder's from the chunk's first bit.
167
+ reset_lzma_state! if @need_state_reset || @need_dictionary_reset
168
+
169
+ compressed, consumed = try_compress(slice)
170
+
171
+ use_compressed =
172
+ consumed.positive? &&
173
+ compressed.bytesize <= COMPRESSED_MAX &&
174
+ compressed.bytesize < (consumed == slice.bytesize ? slice.bytesize : consumed)
175
+
176
+ if use_compressed
177
+ uncompressed = chunk_data.byteslice(offset, consumed)
178
+ emit_compressed_chunk(output, uncompressed, compressed)
179
+
180
+ offset += consumed
181
+ @need_state_reset = false
182
+ else
183
+ # Compression didn't help (or hit the cap immediately):
184
+ # store the whole slice as uncompressed chunks.
185
+ emit_uncompressed_chunks(output, slice)
186
+
187
+ offset += slice.bytesize
188
+ # After an uncompressed chunk the LZMA state is lost, so the
189
+ # next compressed chunk must reset it (XZ Utils behavior,
190
+ # lzma2_encoder.c).
191
+ @need_state_reset = true
192
+ end
193
+ @need_dictionary_reset = false
194
+ end
195
+
196
+ # Update prev_byte for next chunk
197
+ if chunk_data.bytesize.positive?
198
+ @prev_byte = chunk_data.getbyte(chunk_data.bytesize - 1)
199
+ end
200
+ end
156
201
 
157
- # XZ Utils chunk type selection:
158
- # Use uncompressed chunk if: compressed_size >= uncompressed_size
159
- # Use compressed chunk if: compressed_size < uncompressed_size
160
- # NOTE: Compare only DATA sizes, NOT including headers!
161
- # This matches XZ Utils implementation exactly (lzma2_encoder.c line 205)
202
+ # Reset the LZMA state machine and probability models.
203
+ #
204
+ # MUST be called before encoding any chunk whose control byte
205
+ # announces a state reset (0xC0/0xE0), because the decoder will
206
+ # reinitialize its models on seeing that byte.
207
+ def reset_lzma_state!
208
+ @state = Omnizip::Algorithms::LZMA::LZMAState.new(0)
209
+ @models = Omnizip::Algorithms::LZMA::XzProbabilityModels.new(@lc,
210
+ @lp,
211
+ @pb)
212
+ @prev_byte = 0
213
+ end
214
+
215
+ # Emit a compressed LZMA2 chunk.
216
+ def emit_compressed_chunk(output, uncompressed, compressed)
217
+ # For compressed chunks, properties encode lc/lp/pb:
218
+ # (pb * 5 + lp) * 9 + lc
219
+ chunk_properties = (((@pb * 5) + @lp) * 9) + @lc
220
+
221
+ # A dictionary reset implies a state reset; props accompany
222
+ # every state reset (control 0xE0 or 0xC0).
223
+ state_reset = @need_state_reset || @need_dictionary_reset
224
+
225
+ chunk = Omnizip::Algorithms::LZMA2::LZMA2Chunk.new(
226
+ chunk_type: :compressed,
227
+ uncompressed_data: uncompressed,
228
+ compressed_data: compressed,
229
+ compressed_size: compressed.bytesize,
230
+ properties: chunk_properties,
231
+ need_dict_reset: @need_dictionary_reset,
232
+ need_state_reset: state_reset,
233
+ need_props: state_reset,
234
+ )
235
+ output.write(chunk.to_bytes)
162
236
 
163
- if compressed.bytesize >= uncompressed_data.bytesize
164
- # Use uncompressed chunk (compression didn't help)
237
+ # Update dictionary with the chunk data
238
+ @dictionary.append(uncompressed)
239
+ end
240
+
241
+ # Emit uncompressed LZMA2 chunks for data (splitting at the
242
+ # 64KB per-chunk limit imposed by the 16-bit size field).
243
+ def emit_uncompressed_chunks(output, data)
244
+ offset = 0
245
+ while offset < data.bytesize
246
+ slice = data.byteslice(offset, UNCOMPRESSED_CHUNK_MAX)
165
247
  chunk = Omnizip::Algorithms::LZMA2::LZMA2Chunk.new(
166
248
  chunk_type: :uncompressed,
167
- uncompressed_data: uncompressed_data,
249
+ uncompressed_data: slice,
168
250
  compressed_data: "",
169
251
  need_dict_reset: @need_dictionary_reset,
170
252
  need_state_reset: false,
171
253
  need_props: false,
172
254
  )
173
- # After uncompressed chunk, next chunk needs state reset
174
- # (XZ Utils does this - see lzma2_encoder.c line 211)
175
- @need_state_reset = true
176
- else
177
- # Use compressed chunk (compression helped)
178
- # For compressed chunks, properties encode lc/lp/pb:
179
- # (pb * 5 + lp) * 9 + lc
180
- chunk_properties = (((@pb * 5) + @lp) * 9) + @lc
181
- # CRITICAL: need_props must be TRUE when we're providing properties!
182
- # This tells the chunk to encode properties in the control byte
183
- # CRITICAL: compressed_size includes ALL bytes (LZMA data + flush bytes)
184
- # The flush bytes are part of the range encoder output and must be included
185
- chunk = Omnizip::Algorithms::LZMA2::LZMA2Chunk.new(
186
- chunk_type: :compressed,
187
- uncompressed_data: uncompressed_data,
188
- compressed_data: compressed,
189
- compressed_size: compressed.bytesize, # Full size including flush bytes
190
- properties: chunk_properties,
191
- need_dict_reset: @need_dictionary_reset,
192
- need_state_reset: @need_state_reset,
193
- need_props: true, # Always true for compressed chunks with properties
194
- )
195
- end
196
-
197
- # Update dictionary with the chunk data (done once per chunk)
198
- @dictionary.append(uncompressed_data)
255
+ output.write(chunk.to_bytes)
256
+ @need_dictionary_reset = false
199
257
 
200
- # Update prev_byte for next chunk
201
- if uncompressed_data.bytesize.positive?
202
- @prev_byte = uncompressed_data.getbyte(uncompressed_data.bytesize - 1)
258
+ offset += slice.bytesize
259
+ @dictionary.append(slice)
203
260
  end
204
-
205
- chunk
206
261
  end
207
262
 
263
+ # Compress data starting at the current dictionary position.
264
+ #
265
+ # Stops early (returning fewer bytes than data.bytesize) when the
266
+ # compressed output is about to exceed COMPRESSED_MAX, leaving the
267
+ # range encoder at a clean item boundary.
268
+ #
269
+ # @return [Array<String, Integer>] compressed bytes and number of
270
+ # input bytes consumed
208
271
  def try_compress(data)
209
272
  # Create output buffer to capture compressed data
210
273
  output_buffer = StringIO.new
@@ -213,39 +276,26 @@ module Omnizip
213
276
  # Create range encoder (direct XZ Utils port)
214
277
  encoder = Omnizip::Algorithms::LZMA::XzRangeEncoder.new(output_buffer)
215
278
 
216
- # Feed all data to match finder first
217
- # This ensures all bytes are available for finding matches
218
- @match_finder.feed(data)
219
-
220
- # CRITICAL: Initialize hash table for positions BEFORE encoding starts
221
- # This ensures that matches can be found for repeated data patterns
222
- # Matches XZ Utils lzma_encoder.c: mf_skip() behavior
223
- # We skip to position (start_pos + data.bytesize - MATCH_LEN_MAX),
224
- # but ensure we don't go negative for small inputs
225
- match_len_max = 2 # Minimum match length in LZMA2
226
- end_pos = [
227
- @dictionary.buffer.bytesize + data.bytesize - match_len_max, 0
228
- ].max
229
- @match_finder.skip(end_pos)
230
-
231
- # Position in match finder's buffer for encoding
232
- # Start after the data we just fed
279
+ # The whole chunk has already been fed to the match finder by
280
+ # consume_chunk; positions map from the dictionary position.
233
281
  start_pos = @dictionary.buffer.bytesize
234
282
 
235
- # Store current start position for matched literal encoding
236
- @current_start_pos = start_pos
283
+ # Headroom for pending symbols plus the final flush
284
+ size_cap = COMPRESSED_MAX - 64
237
285
 
238
286
  pos = 0
239
287
  while pos < data.bytesize
240
- # Encode queued symbols if buffer getting full
241
- # Keep headroom for largest operation
242
- # (~30 symbols for match+distance)
243
- if encoder.count > 20
244
- encode_queued_symbols(encoder, output_buffer)
245
- end
288
+ # Drain before every item: a single worst-case match (long
289
+ # length + dist slot >= 14 with up to 30 footer bits) queues
290
+ # ~48 symbols, and RC_SYMBOLS_MAX is 53, so the queue MUST be
291
+ # empty when an item starts or `bit` raises "Symbol buffer
292
+ # overflow" (issue #26).
293
+ encode_queued_symbols(encoder, output_buffer)
294
+
295
+ # Enforce the 64KB compressed-chunk cap at an item boundary
296
+ break if pos.positive? && output_buffer.size >= size_cap
246
297
 
247
298
  # Position in match finder's buffer for encoding
248
- # Start after the data we just fed
249
299
  match_pos = start_pos + pos
250
300
 
251
301
  # Get optimal encoding choice
@@ -265,21 +315,21 @@ module Omnizip
265
315
  # CRITICAL: Use UINT32_MAX to check for literal (not distance.zero?)
266
316
  # because distance=0 means repeated match rep0, not literal!
267
317
  if distance == UINT32_MAX || length == 1
268
- # Encode literal
269
- # puts "[DEBUG] -> LITERAL 0x#{'%02x' % data.getbyte(pos)}" if ENV['DEBUG']
270
- encode_literal(data.getbyte(pos), encoder, pos)
318
+ # Encode literal. The position argument must be GLOBAL
319
+ # (dictionary position), because the decoder keeps counting
320
+ # bytes across chunk boundaries for pos_state.
321
+ encode_literal(data.getbyte(pos), encoder, match_pos)
271
322
  pos += 1
272
323
  elsif distance < REPS
273
324
  # Encode repeated match (distance is 0-3 for rep0-rep3)
274
- # puts "[DEBUG] -> REPEATED MATCH rep#{distance} len=#{length}" if ENV['DEBUG']
275
- encode_repeated_match(distance, length, encoder, pos, match_pos)
325
+ encode_repeated_match(distance, length, encoder, match_pos,
326
+ match_pos)
276
327
  pos += length
277
328
  else
278
329
  # Encode normal match (distance is actual_distance + REPS)
279
330
  actual_distance = distance - REPS
280
- # puts "[DEBUG] -> NORMAL MATCH distance=#{actual_distance} len=#{length}" if ENV['DEBUG']
281
- encode_match(actual_distance, length, encoder, pos, match_pos,
282
- data)
331
+ encode_match(actual_distance, length, encoder, match_pos,
332
+ match_pos, data)
283
333
  pos += length
284
334
  end
285
335
  end
@@ -295,46 +345,12 @@ module Omnizip
295
345
  # This will write additional bytes to output_buffer
296
346
  encode_queued_symbols(encoder, output_buffer)
297
347
 
298
- # Full output includes all bytes (LZMA data + flush bytes)
299
- full_output = output_buffer.string
300
-
301
- puts "[DEBUG] try_compress: full_output.size=#{full_output.bytesize}, encoder.out_total=#{encoder.out_total}" if ENV["DEBUG_FLUSH"]
302
-
303
- # Return all bytes (flush bytes are part of the LZMA data)
304
- full_output
305
- end
306
-
307
- # Encode queued symbols to output
308
- # rubocop:disable Style/CollectionQuerying
309
- def encode_queued_symbols(encoder, output)
310
- return if encoder.count.zero?
311
-
312
- # Encode symbols to buffer
313
- encoder.encode_symbols(temp_buffer, out_pos, 10000)
314
-
315
- # Track size before encoding
316
- size_before = output.size
317
-
318
- # Encode symbols to buffer
319
- encoder.encode_symbols(temp_buffer, out_pos, 10000)
320
-
321
- # Write to output stream
322
- if out_pos.value.positive?
323
- # Use StringCompat.byteslice for Ruby 3.0-3.1 compatibility
324
- # Ruby's [] operator has a bug with null bytes that can return extra bytes
325
- # See: https://bugs.ruby-lang.org/issues/15985
326
- output.write(StringCompat.byteslice(temp_buffer, 0,
327
- out_pos.value))
328
- end
329
-
330
- # Return the number of bytes written
331
- output.size - size_before
348
+ [output_buffer.string, pos]
332
349
  end
333
350
 
334
351
  # Encode queued symbols to output
335
- # rubocop:disable Style/CollectionQuerying
336
352
  def encode_queued_symbols(encoder, output)
337
- return if encoder.count.zero?
353
+ return if encoder.none?
338
354
 
339
355
  # Create temporary buffer for encoding
340
356
  temp_buffer = "\0" * 10000
@@ -360,6 +376,8 @@ module Omnizip
360
376
  end
361
377
 
362
378
  # Encode literal byte
379
+ # pos is the GLOBAL position (dictionary position) because the
380
+ # decoder counts bytes continuously across chunks.
363
381
  def encode_literal(symbol, encoder, pos)
364
382
  pos_state = pos & ((1 << @pb) - 1)
365
383
 
@@ -382,8 +400,7 @@ module Omnizip
382
400
  # Matched literal (compare with match byte at rep0)
383
401
  # XZ Utils: mf->buffer[mf->read_pos - coder->reps[0] - 1 - mf->read_ahead]
384
402
  # We don't use read_ahead, so it's 0
385
- match_pos = @current_start_pos + pos
386
- match_byte_pos = match_pos - @state.reps[0] - 1
403
+ match_byte_pos = pos - @state.reps[0] - 1
387
404
  match_byte = @match_finder.buffer.getbyte(match_byte_pos) if match_byte_pos >= 0 && match_byte_pos < @match_finder.buffer.bytesize
388
405
 
389
406
  # If match_byte is nil (shouldn't happen in normal operation),
@@ -417,8 +434,9 @@ _input_data)
417
434
  encoder.queue_bit(prob_is_rep, 0)
418
435
 
419
436
  # CRITICAL: Update state BEFORE encoding length/distance (XZ Utils order)
420
- # This also updates reps
421
- @state.update_match!(distance)
437
+ # This also updates reps. Reps are stored 0-based (distance - 1),
438
+ # matching XZ Utils and the decoder's rep0 convention.
439
+ @state.update_match!(distance - 1)
422
440
 
423
441
  # Encode length - uses NEW state value
424
442
  encode_match_length(length, pos_state, encoder)
@@ -466,27 +484,15 @@ _input_data)
466
484
 
467
485
  prob_is_rep2 = @models.is_rep2[@state.value]
468
486
  encoder.queue_bit(prob_is_rep2, rep - 2)
469
-
470
- if rep == 3
471
- # Update reps[3] = reps[2] before updating reps[2]
472
- @state.reps[3] = @state.reps[2]
473
- end
474
-
475
- # Update reps[2] = reps[1]
476
- @state.reps[2] = @state.reps[1]
477
487
  end
478
488
 
479
- # Update reps[1] = reps[0]
480
- @state.reps[1] = @state.reps[0]
481
-
482
- # Update reps[0] = distance from reps[rep]
489
+ # XZ Utils rep_match: capture the selected distance BEFORE
490
+ # rotating, then shift reps down (lzma_encoder.c):
491
+ # distance = reps[rep];
492
+ # for (i = rep; i > 0; --i) reps[i] = reps[i-1];
493
+ # reps[0] = distance;
483
494
  distance = @state.reps[rep]
484
-
485
- # Defensive check: distance should never be nil
486
- if distance.nil?
487
- raise "Distance is nil for rep #{rep}, reps=#{@state.reps.inspect}"
488
- end
489
-
495
+ rep.downto(1) { |i| @state.reps[i] = @state.reps[i - 1] }
490
496
  @state.reps[0] = distance
491
497
  end
492
498
 
@@ -494,15 +500,15 @@ _input_data)
494
500
  if length == 1
495
501
  @state.update_short_rep!
496
502
  else
497
- # Encode length
498
- encode_match_length(length, pos_state, encoder)
503
+ # Encode length using the REP length coder (rep matches have
504
+ # their own length models, decoder's @rep_length_coder)
505
+ encode_match_length(length, pos_state, encoder,
506
+ @models.rep_len_encoder)
499
507
  @state.update_long_rep!
500
508
  end
501
509
 
502
- # Update prev_byte (last byte of match)
503
- # For rep match: match_pos - reps[rep] - 1 + length - 1 = match_pos - reps[rep] + length - 2
504
- # But after updating reps above, reps[0] now contains the distance
505
- last_byte_pos = match_pos - @state.reps[0] + length - 1
510
+ # Update prev_byte (last byte of the match region in the stream)
511
+ last_byte_pos = match_pos + length - 1
506
512
  @prev_byte = @match_finder.buffer.getbyte(last_byte_pos) if last_byte_pos >= 0 && last_byte_pos < @match_finder.buffer.bytesize
507
513
  end
508
514
 
@@ -513,12 +519,13 @@ _input_data)
513
519
  # (((((pos) << 8) + (prev_byte)) & (literal_mask)) << (lc)))
514
520
  # where literal_mask = (0x100 << lp) - (0x100 >> lc)
515
521
  #
522
+ # The factor of 3 gives each context a 0x300-sized subcoder and
523
+ # MUST match the decoder's base_offset = 3 * (lit_state << lc).
524
+ #
516
525
  # Returns the flat index into the literal probability array.
517
- # The literal array is now a flat array (matching XZ Utils) with
518
- # size 0x300 << (lc + lp), not a 2D array.
519
526
  def get_literal_state(pos, prev_byte)
520
527
  literal_mask = (0x100 << @lp) - (0x100 >> @lc)
521
- ((((pos << 8) + prev_byte) & literal_mask) << @lc)
528
+ 3 * ((((pos << 8) + prev_byte) & literal_mask) << @lc)
522
529
  end
523
530
 
524
531
  # Get byte from dictionary at distance back
@@ -570,35 +577,38 @@ encoder)
570
577
  end
571
578
 
572
579
  # Encode match length
573
- def encode_match_length(length, pos_state, encoder)
580
+ # len_encoder is the length model set to use: match_len_encoder for
581
+ # normal matches, rep_len_encoder for repeated matches.
582
+ def encode_match_length(length, pos_state, encoder,
583
+ len_encoder = @models.match_len_encoder)
574
584
  len = length - MATCH_LEN_MIN
575
585
 
576
586
  if len < LEN_LOW_SYMBOLS
577
587
  # Low: 0-7
578
- encoder.queue_bit(@models.match_len_encoder.choice, 0)
588
+ encoder.queue_bit(len_encoder.choice, 0)
579
589
  encode_bittree(
580
- @models.match_len_encoder.low[pos_state],
590
+ len_encoder.low[pos_state],
581
591
  NUM_LEN_LOW_BITS,
582
592
  len,
583
593
  encoder,
584
594
  )
585
595
  elsif len < LEN_LOW_SYMBOLS + LEN_MID_SYMBOLS
586
596
  # Mid: 8-15
587
- encoder.queue_bit(@models.match_len_encoder.choice, 1)
588
- encoder.queue_bit(@models.match_len_encoder.choice2, 0)
597
+ encoder.queue_bit(len_encoder.choice, 1)
598
+ encoder.queue_bit(len_encoder.choice2, 0)
589
599
  encode_bittree(
590
- @models.match_len_encoder.mid[pos_state],
600
+ len_encoder.mid[pos_state],
591
601
  NUM_LEN_MID_BITS,
592
602
  len - LEN_LOW_SYMBOLS,
593
603
  encoder,
594
604
  )
595
605
  else
596
606
  # High: 16-271
597
- encoder.queue_bit(@models.match_len_encoder.choice, 1)
598
- encoder.queue_bit(@models.match_len_encoder.choice2, 1)
607
+ encoder.queue_bit(len_encoder.choice, 1)
608
+ encoder.queue_bit(len_encoder.choice2, 1)
599
609
  high_len = len - LEN_LOW_SYMBOLS - LEN_MID_SYMBOLS
600
610
  encode_bittree(
601
- @models.match_len_encoder.high,
611
+ len_encoder.high,
602
612
  NUM_LEN_HIGH_BITS,
603
613
  high_len,
604
614
  encoder,
@@ -608,7 +618,11 @@ encoder)
608
618
 
609
619
  # Encode distance using slot encoding
610
620
  def encode_distance(distance, length, encoder)
611
- dist_slot = get_dist_slot(distance)
621
+ # get_dist_slot expects a 0-based distance (xz fastpos.h:
622
+ # dist 4 -> slot 4, dist 5 -> slot 4). The decoder computes
623
+ # rep0 = base + footer the same 0-based way.
624
+ dist0 = distance - 1
625
+ dist_slot = get_dist_slot(dist0)
612
626
  len_state = get_len_to_pos_state(length)
613
627
 
614
628
  # Encode distance slot
@@ -624,7 +638,7 @@ encoder)
624
638
  if dist_slot >= START_POS_MODEL_INDEX
625
639
  footer_bits = (dist_slot >> 1) - 1
626
640
  base = (2 | (dist_slot & 1)) << footer_bits
627
- dist_reduced = distance - base
641
+ dist_reduced = dist0 - base
628
642
 
629
643
  if dist_slot < END_POS_MODEL_INDEX
630
644
  # Use probability models
@@ -1,11 +1,25 @@
1
1
  # frozen_string_literal: true
2
2
 
3
3
  module Omnizip
4
+ # Bit-level SDK ports: state machines, range coders, match finders,
5
+ # raw LZMA/LZMA2 encoders and decoders.
6
+ #
7
+ # This namespace owns the *primitives*; +Omnizip::Algorithms+ owns
8
+ # the *streaming contract*. The boundary is: +Algorithms+ calls
9
+ # +Implementations+, never the reverse. Files under this tree
10
+ # should never reach across into format-specific concerns (no
11
+ # .7z header parsing, no ZIP central directory, no tar entry
12
+ # metadata).
13
+ #
14
+ # The two top-level sub-namespaces mirror the upstream SDK split:
15
+ # - +SevenZip+ : ports of 7-Zip's LZMA/LZMA2 encoder + decoder
16
+ # - +XZUtils+ : ports of XZ Utils' LZMA/LZMA2 encoder + decoder
17
+ #
18
+ # Despite the LZMA name, the two SDKs diverge in subtle ways (no
19
+ # EOS marker, different chunk padding, different property-byte
20
+ # handling); keeping them in separate trees makes those
21
+ # divergences visible rather than hidden in flag parameters.
4
22
  module Implementations
5
- # Implementation namespace for various algorithm ports
6
- # This module contains reference implementations ported from external projects
7
-
8
- # Base classes for implementation inheritance
9
23
  autoload :Base, "omnizip/implementations/base"
10
24
  end
11
25
  end
@@ -1,5 +1,8 @@
1
1
  # frozen_string_literal: true
2
2
 
3
+ require "stringio"
4
+ require "tempfile"
5
+
3
6
  module Omnizip
4
7
  module IO
5
8
  # Polymorphic adapter for "things we can read bytes from".
@@ -5,8 +5,15 @@ module Omnizip
5
5
  # Model for file entry metadata
6
6
  # Wraps CentralDirectoryHeader with a cleaner metadata API
7
7
  class EntryMetadata
8
+ include Omnizip::Entry
9
+
8
10
  attr_reader :entry
9
11
 
12
+ def entry_name = entry.name
13
+ def entry_directory? = entry.directory?
14
+ def entry_size = entry.size
15
+ def entry_mtime = entry.time
16
+
10
17
  # Initialize metadata for an entry
11
18
  # @param entry [Omnizip::Zip::Entry] The entry to manage metadata for
12
19
  def initialize(entry)
@@ -0,0 +1,48 @@
1
+ # frozen_string_literal: true
2
+
3
+ require "fractor"
4
+
5
+ module Omnizip
6
+ module Parallel
7
+ # Engine owns the Fractor worker-pool dance: build the pool, submit
8
+ # a batch of work, run it to completion, return successful and
9
+ # failed results. Callers (ParallelCompressor, ParallelExtractor,
10
+ # future operations) supply the worker class and the work items;
11
+ # Engine handles the threading.
12
+ #
13
+ # This is the deep module behind the parallel API: the threading
14
+ # shape is identical across every consumer, so it lives here once.
15
+ # Domain-specific concerns (which file to read, which archive to
16
+ # open, what stats to track) stay with the caller.
17
+ class Engine
18
+ # @param worker_class [Class] a +Fractor::Worker+ subclass whose
19
+ # +#process(work)+ turns one work item into a result.
20
+ # @param threads [Integer] worker count.
21
+ def initialize(worker_class:, threads:)
22
+ @worker_class = worker_class
23
+ @threads = threads
24
+ end
25
+
26
+ # Run +work_items+ through the pool and yield each successful
27
+ # +Fractor::WorkResult+ to the caller-supplied block.
28
+ #
29
+ # @param work_items [Array<Fractor::Work>] batch to submit.
30
+ # @yieldparam result [Fractor::WorkResult] a successful result.
31
+ # @return [Array<Fractor::WorkResult>] the failed results, for
32
+ # the caller to decide how to surface.
33
+ def run(work_items)
34
+ pool = Fractor::WorkerPool.new(
35
+ worker_class: @worker_class,
36
+ num_workers: @threads,
37
+ continuous: false,
38
+ )
39
+ pool.start
40
+ pool.submit_batch(work_items)
41
+ pool.run
42
+
43
+ pool.successful_results.each { |r| yield r if block_given? }
44
+ pool.failed_results
45
+ end
46
+ end
47
+ end
48
+ end
@@ -212,20 +212,11 @@ level: 6)
212
212
  )
213
213
  end
214
214
 
215
- # Create worker pool
216
- pool = WorkerPool.new(
217
- worker_class: CompressionWorker,
218
- num_workers: @options.threads,
219
- continuous: false,
220
- )
221
-
222
- pool.start
223
- pool.submit_batch(work_items)
224
- pool.run
225
-
226
- # Collect results
227
- results = pool.successful_results
228
- errors = pool.failed_results
215
+ # Run the worker pool through the shared engine.
216
+ engine = Engine.new(worker_class: CompressionWorker,
217
+ threads: @options.threads)
218
+ results = []
219
+ errors = engine.run(work_items) { |r| results << r }
229
220
 
230
221
  # Handle errors
231
222
  unless errors.empty?
@@ -238,8 +229,6 @@ level: 6)
238
229
  # Write archive sequentially (thread-safe)
239
230
  write_archive(output, results, compression: compression)
240
231
 
241
- pool.shutdown
242
-
243
232
  @stats[:end_time] = Time.now
244
233
  @stats[:files_processed] = results.size
245
234