omnizip 0.3.20 → 0.3.22

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (54) hide show
  1. checksums.yaml +4 -4
  2. data/TODO.refactor/00-overview.md +2 -2
  3. data/TODO.refactor/07-respond-to-replacement.md +225 -79
  4. data/lib/omnizip/algorithm.rb +2 -2
  5. data/lib/omnizip/algorithms/bzip2.rb +1 -0
  6. data/lib/omnizip/algorithms/deflate.rb +1 -0
  7. data/lib/omnizip/algorithms/lzma/xz_utils_decoder.rb +3 -3
  8. data/lib/omnizip/algorithms/lzma.rb +11 -5
  9. data/lib/omnizip/algorithms/lzma2/encoder.rb +2 -0
  10. data/lib/omnizip/algorithms/lzma2/xz_encoder_adapter.rb +2 -0
  11. data/lib/omnizip/algorithms/lzma2.rb +2 -0
  12. data/lib/omnizip/algorithms/ppmd7.rb +2 -0
  13. data/lib/omnizip/algorithms/ppmd_base.rb +2 -0
  14. data/lib/omnizip/algorithms/registration.rb +1 -1
  15. data/lib/omnizip/algorithms/zstandard/decoder.rb +62 -3
  16. data/lib/omnizip/algorithms/zstandard/dictionary.rb +73 -0
  17. data/lib/omnizip/algorithms/zstandard/encoder.rb +110 -30
  18. data/lib/omnizip/algorithms/zstandard/match_finder.rb +21 -5
  19. data/lib/omnizip/algorithms/zstandard.rb +52 -0
  20. data/lib/omnizip/chunked/writer.rb +3 -0
  21. data/lib/omnizip/commands/archive_extract_command.rb +3 -1
  22. data/lib/omnizip/commands/archive_list_command.rb +4 -3
  23. data/lib/omnizip/commands/profile_show_command.rb +2 -0
  24. data/lib/omnizip/convenience.rb +2 -0
  25. data/lib/omnizip/extraction/filter_chain.rb +4 -0
  26. data/lib/omnizip/extraction/selective_extractor.rb +7 -0
  27. data/lib/omnizip/file_type.rb +4 -0
  28. data/lib/omnizip/formats/ole.rb +2 -0
  29. data/lib/omnizip/formats/rar/decompressor.rb +1 -0
  30. data/lib/omnizip/formats/rar/license_validator.rb +2 -0
  31. data/lib/omnizip/formats/rar/reader.rb +1 -2
  32. data/lib/omnizip/formats/rpm.rb +1 -0
  33. data/lib/omnizip/formats/seven_zip/writer.rb +8 -0
  34. data/lib/omnizip/formats/tar/reader.rb +2 -0
  35. data/lib/omnizip/formats/xar/reader.rb +2 -0
  36. data/lib/omnizip/formats/xar/writer.rb +1 -0
  37. data/lib/omnizip/formats/xz.rb +2 -0
  38. data/lib/omnizip/formats/xz_impl/block_decoder.rb +3 -0
  39. data/lib/omnizip/formats/xz_impl/stream_decoder.rb +8 -0
  40. data/lib/omnizip/formats/zip/reader.rb +2 -0
  41. data/lib/omnizip/implementations/seven_zip/lzma/encoder.rb +1 -0
  42. data/lib/omnizip/implementations/xz_utils/lzma2/decoder.rb +4 -9
  43. data/lib/omnizip/implementations/xz_utils/lzma2/encoder.rb +1 -16
  44. data/lib/omnizip/io/buffered_input.rb +1 -0
  45. data/lib/omnizip/io/buffered_output.rb +1 -0
  46. data/lib/omnizip/io/source.rb +4 -0
  47. data/lib/omnizip/io/stream_manager.rb +3 -1
  48. data/lib/omnizip/link_handler.rb +2 -0
  49. data/lib/omnizip/metadata/metadata_validator.rb +2 -0
  50. data/lib/omnizip/models/filter_config.rb +2 -0
  51. data/lib/omnizip/version.rb +1 -1
  52. data/lib/omnizip/zip/file.rb +1 -1
  53. data/lib/omnizip/zip/input_stream.rb +2 -1
  54. metadata +2 -1
@@ -0,0 +1,73 @@
1
+ # frozen_string_literal: true
2
+
3
+ # Copyright (C) 2025 Ribose Inc.
4
+ #
5
+ # Permission is hereby granted, free of charge, to any person obtaining a
6
+ # copy of this software and associated documentation files (the "Software"),
7
+ # to deal in the Software without restriction, including without limitation
8
+ # the rights to use, copy, modify, merge, publish, distribute, sublicense,
9
+ # and/or sell copies of the Software, and to permit persons to whom the
10
+ # Software is furnished to do so, subject to the following conditions:
11
+ #
12
+ # The above copyright notice and this permission notice shall be included in
13
+ # all copies or substantial portions of the Software.
14
+ #
15
+ # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING
20
+ # FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER
21
+ # DEALINGS IN THE SOFTWARE.
22
+
23
+ module Omnizip
24
+ module Algorithms
25
+ class Zstandard
26
+ # Zstandard dictionary (port of the omnizip-rs dict.rs).
27
+ #
28
+ # A dictionary lets the encoder preload a reference-content
29
+ # window so small inputs compress dramatically better: the
30
+ # dictionary content is used as a match-finder prefix, and the
31
+ # frame header carries the dictionary's ID.
32
+ #
33
+ # Wire format (simplified form, as in the Rust reference):
34
+ # magic (4) + dictionary ID (4, LE) + raw content.
35
+ class Dictionary
36
+ DICT_MAGIC = 0xEC30A437
37
+
38
+ attr_reader :id, :content
39
+
40
+ # @param id [Integer] dictionary ID carried in frame headers
41
+ # @param content [String] corpus bytes used as the prefix
42
+ def initialize(id, content)
43
+ @id = id
44
+ @content = content.dup.force_encoding(Encoding::BINARY)
45
+ end
46
+
47
+ def self.from_raw(id, content)
48
+ new(id, content)
49
+ end
50
+
51
+ def self.deserialize(data)
52
+ if data.bytesize < 8
53
+ raise Omnizip::DecompressionError,
54
+ "dictionary too short for magic + id"
55
+ end
56
+ if data.byteslice(0, 4).unpack1("V") != DICT_MAGIC
57
+ raise Omnizip::DecompressionError, "bad dictionary magic"
58
+ end
59
+
60
+ new(data.byteslice(4, 4).unpack1("V"), data.byteslice(8..))
61
+ end
62
+
63
+ def serialize
64
+ [DICT_MAGIC].pack("V") + [@id].pack("V") + @content
65
+ end
66
+
67
+ def ==(other)
68
+ other.is_a?(Dictionary) && id == other.id && content == other.content
69
+ end
70
+ end
71
+ end
72
+ end
73
+ end
@@ -58,42 +58,83 @@ module Omnizip
58
58
  write_frame(data)
59
59
  end
60
60
 
61
+ # Encode a data stream primed with a dictionary prefix (port
62
+ # of the Rust encode_frame_with_dict): the dictionary content
63
+ # is prepended as virtual history, the match finder is seeded
64
+ # with its positions, and only the plaintext is emitted.
65
+ # Frame_Content_Size and the checksum cover the plaintext
66
+ # only; the frame header carries the dictionary's ID and
67
+ # requires the dict-aware decoder.
68
+ #
69
+ # @param data [String]
70
+ # @param dict [Dictionary]
71
+ def encode_stream_with_dict(data, dict)
72
+ data = data.dup.force_encoding(Encoding::BINARY)
73
+ virtual = dict.content + data
74
+ prefix_len = dict.content.bytesize
75
+
76
+ write_u32le(MAGIC_NUMBER)
77
+ write_frame_header(data, dict.id)
78
+ write_blocks(virtual, prefix_len)
79
+ write_u32le(XXHash64.frame_checksum(data)) if @checksum
80
+ end
81
+
61
82
  private
62
83
 
63
- def write_frame(data)
84
+ def write_frame(data, dict = nil)
64
85
  write_u32le(MAGIC_NUMBER)
65
- write_frame_header(data)
86
+ write_frame_header(data, dict && dict.id)
66
87
  write_blocks(data)
67
88
  write_u32le(XXHash64.frame_checksum(data)) if @checksum
68
89
  end
69
90
 
70
- # Single-segment frame header: no window descriptor, no
71
- # dictionary; Frame_Content_Size in 1, 2, 4 or 8 bytes.
72
- def write_frame_header(data)
91
+ # Single-segment frame header: no window descriptor;
92
+ # optional Dictionary_ID, then Frame_Content_Size in 1, 2, 4
93
+ # or 8 bytes.
94
+ def write_frame_header(data, dict_id = nil)
95
+ did_flag, did_bytes = dict_id_encoding(dict_id)
96
+
73
97
  size = data.bytesize
74
98
  if size < 256
75
- @output_stream.putc(0x20) # single segment, FCS 1 byte
99
+ @output_stream.putc(0x20 | did_flag) # single segment, FCS 1 byte
100
+ write_dict_id_bytes(did_bytes)
76
101
  @output_stream.putc(size)
77
102
  elsif size <= 65_791
78
- @output_stream.putc(0x60) # single segment, FCS 2 bytes (+256)
103
+ @output_stream.putc(0x60 | did_flag) # single segment, FCS 2 bytes
104
+ write_dict_id_bytes(did_bytes)
79
105
  write_u16le(size - 256)
80
106
  elsif size <= 0xFFFFFFFF
81
- @output_stream.putc(0xA0) # single segment, FCS 4 bytes
107
+ @output_stream.putc(0xA0 | did_flag) # single segment, FCS 4 bytes
108
+ write_dict_id_bytes(did_bytes)
82
109
  write_u32le(size)
83
110
  else
84
- @output_stream.putc(0xE0) # single segment, FCS 8 bytes
111
+ @output_stream.putc(0xE0 | did_flag) # single segment, FCS 8 bytes
112
+ write_dict_id_bytes(did_bytes)
85
113
  write_u64le(size)
86
114
  end
87
115
  end
88
116
 
117
+ def dict_id_encoding(dict_id)
118
+ return [0, "".b] if dict_id.nil? || dict_id.zero?
119
+ return [1, [dict_id].pack("C")] if dict_id <= 0xFF
120
+ return [2, [dict_id].pack("v")] if dict_id <= 0xFFFF
121
+
122
+ [3, [dict_id].pack("V")]
123
+ end
124
+
125
+ def write_dict_id_bytes(bytes)
126
+ @output_stream.write(bytes)
127
+ end
128
+
89
129
  # rubocop:disable Metrics/MethodLength
90
130
  # rubocop:disable-next Metrics/AbcSize
91
- def write_blocks(data)
92
- return write_empty_last_block if data.empty?
131
+ def write_blocks(data, start = 0)
132
+ return write_empty_last_block if data.bytesize <= start
93
133
 
94
134
  params = MatchFinder.params_for_level(@level, data.bytesize)
95
135
  ms = MatchFinder::MatchState.new(params[:hash_log])
96
136
  ms.enable_chain(params[:chain]) if params[:chain].positive?
137
+ ms.seed_prefix(data, start) if start.positive?
97
138
 
98
139
  # LDM (long-distance matching) at high levels on multi-block
99
140
  # inputs: a sparse table over the whole frame finds matches
@@ -115,32 +156,55 @@ module Omnizip
115
156
  # untouched.
116
157
  reps = [1, 4, 8]
117
158
 
118
- offset = 0
159
+ offset = start
119
160
  while offset < data.bytesize
120
161
  block_end = [offset + BLOCK_MAX_SIZE, data.bytesize].min
121
162
  is_last = block_end == data.bytesize
122
163
  chunk = data.byteslice(offset, block_end - offset)
123
164
 
124
- if chunk.bytesize >= 4 && rle_chunk?(chunk)
125
- write_block_header(is_last ? 1 : 0, BLOCK_TYPE_RLE,
126
- chunk.bytesize)
127
- @output_stream.putc(chunk.getbyte(0))
128
- else
129
- seq_store = MatchFinder::SeqStore.new(reps.dup)
130
- MatchFinder.compress_range(data, offset, block_end, seq_store,
131
- ms, params[:min_match],
132
- params[:lazy], ldm, max_distance)
133
- content, new_reps = try_compressed(seq_store, reps)
134
- if content.nil?
135
- write_block_header(is_last ? 1 : 0, BLOCK_TYPE_RAW,
136
- chunk.bytesize)
137
- @output_stream.write(chunk)
165
+ # Adaptive splitting (port of the Rust reference): a
166
+ # heterogeneous chunk — first and second halves' byte
167
+ # distributions diverge — is split into 16 KiB sub-blocks
168
+ # so each Huffman/FSE table fits its region; homogeneous
169
+ # chunks keep the single block and its minimal header
170
+ # overhead.
171
+ step = if chunk.bytesize >= 32 * 1024 &&
172
+ halves_diverge(chunk)
173
+ 16 * 1024
174
+ else
175
+ chunk.bytesize
176
+ end
177
+
178
+ sub = offset
179
+ while sub < block_end
180
+ sub_end = [sub + step, block_end].min
181
+ sub_last = is_last && sub_end == block_end
182
+ sub_chunk = data.byteslice(sub, sub_end - sub)
183
+
184
+ if sub_chunk.bytesize >= 4 && rle_chunk?(sub_chunk)
185
+ write_block_header(sub_last ? 1 : 0, BLOCK_TYPE_RLE,
186
+ sub_chunk.bytesize)
187
+ @output_stream.putc(sub_chunk.getbyte(0))
138
188
  else
139
- write_block_header(is_last ? 1 : 0, BLOCK_TYPE_COMPRESSED,
140
- content.bytesize)
141
- @output_stream.write(content)
142
- reps = new_reps
189
+ seq_store = MatchFinder::SeqStore.new(reps.dup)
190
+ MatchFinder.compress_range(data, sub, sub_end, seq_store,
191
+ ms, params[:min_match],
192
+ params[:lazy], ldm, max_distance)
193
+ content, new_reps = try_compressed(seq_store, reps)
194
+ if content.nil?
195
+ write_block_header(sub_last ? 1 : 0, BLOCK_TYPE_RAW,
196
+ sub_chunk.bytesize)
197
+ @output_stream.write(sub_chunk)
198
+ else
199
+ write_block_header(sub_last ? 1 : 0,
200
+ BLOCK_TYPE_COMPRESSED,
201
+ content.bytesize)
202
+ @output_stream.write(content)
203
+ reps = new_reps
204
+ end
143
205
  end
206
+
207
+ sub = sub_end
144
208
  end
145
209
 
146
210
  offset = block_end
@@ -152,6 +216,22 @@ module Omnizip
152
216
  write_block_header(1, BLOCK_TYPE_RAW, 0)
153
217
  end
154
218
 
219
+ # Whether the byte distributions of a chunk's halves diverge
220
+ # (total-variation sum above 0.5 = "different content
221
+ # regimes", per the Rust reference).
222
+ def halves_diverge(chunk)
223
+ mid = chunk.bytesize / 2
224
+ ha = Array.new(256, 0)
225
+ hb = Array.new(256, 0)
226
+ chunk.byteslice(0, mid).each_byte { |b| ha[b] += 1 }
227
+ chunk.byteslice(mid, chunk.bytesize - mid).each_byte { |b| hb[b] += 1 }
228
+ la = mid.to_f
229
+ lb = (chunk.bytesize - mid).to_f
230
+ tvd = 0.0
231
+ 256.times { |i| tvd += ((ha[i] / la) - (hb[i] / lb)).abs }
232
+ tvd > 0.5
233
+ end
234
+
155
235
  # Build a Compressed_Block content: literals section plus the
156
236
  # sequences section. Returns [content, wire_reps], or [nil,
157
237
  # nil] when the result is not smaller than the raw chunk.
@@ -88,6 +88,18 @@ module Omnizip
88
88
  def disable_chain
89
89
  @max_chain = 0
90
90
  end
91
+
92
+ # Seed the hash table with a prefix's positions (dictionary
93
+ # content prepended to the plaintext): the most recent
94
+ # position per hash wins, matching insert order.
95
+ def seed_prefix(src, prefix_len)
96
+ return if prefix_len < MIN_MATCH
97
+
98
+ limit = prefix_len - MIN_MATCH + 1
99
+ (0...limit).each do |pos|
100
+ @hash_table[MatchFinder.hash4(src, pos, @hash_log)] = pos
101
+ end
102
+ end
91
103
  end
92
104
 
93
105
  module_function
@@ -178,15 +190,18 @@ module Omnizip
178
190
  limit = block_end - mm
179
191
 
180
192
  while ip < limit
181
- match = if ldm
193
+ match = if ldm || lazy.positive?
194
+ # Lazy levels share the LDM loop's rep0 fast-path
195
+ # (with backward extension): without it, levels
196
+ # 6+ measured worse than the greedy default level
197
+ # because every rep-offset match paid full offset
198
+ # coding.
182
199
  find_match_ldm(src, ip, ms, mm, limit, anchor,
183
200
  seq_store.rep_offsets[0], ldm,
184
201
  max_distance)
185
- elsif lazy.zero?
202
+ else
186
203
  find_greedy_match(src, ip, ms, mm, limit, anchor,
187
204
  seq_store.rep_offsets[0])
188
- else
189
- find_best_match(src, ip, ms, mm, limit)
190
205
  end
191
206
  if match
192
207
  dist, len = match
@@ -247,7 +262,8 @@ module Omnizip
247
262
  ldm, max_distance)
248
263
  return nil unless dist
249
264
 
250
- [dist, len, ip]
265
+ back = backward_extension(src, ip, anchor, ip - dist)
266
+ [dist, len + back, ip - back]
251
267
  end
252
268
 
253
269
  # Greedy-path match search (port of the Rust
@@ -1,5 +1,7 @@
1
1
  # frozen_string_literal: true
2
2
 
3
+ require "stringio"
4
+
3
5
  # Copyright (C) 2025 Ribose Inc.
4
6
  #
5
7
  # Permission is hereby granted, free of charge, to any person obtaining a
@@ -58,6 +60,7 @@ module Omnizip
58
60
  "omnizip/algorithms/zstandard/sequences_encoder"
59
61
  autoload :MatchFinder, "omnizip/algorithms/zstandard/match_finder"
60
62
  autoload :LdmHashTable, "omnizip/algorithms/zstandard/ldm"
63
+ autoload :Dictionary, "omnizip/algorithms/zstandard/dictionary"
61
64
  autoload :XXHash64, "omnizip/algorithms/zstandard/xxhash"
62
65
 
63
66
  # Frame and FSE modules
@@ -101,6 +104,54 @@ module Omnizip
101
104
  output_stream.write(decompressed)
102
105
  end
103
106
 
107
+ # Compress data primed with a dictionary prefix: the dictionary
108
+ # content acts as shared history the match finder can reference,
109
+ # dramatically improving ratios on small inputs. The resulting
110
+ # frame carries the dictionary's ID and requires
111
+ # decompress_with_dict to decode.
112
+ #
113
+ # @param input_stream [IO] plaintext
114
+ # @param output_stream [IO] compressed frame
115
+ # @param dict [Dictionary]
116
+ # @param options [Hash] :level, :checksum
117
+ # @return [void]
118
+ def compress_with_dict(input_stream, output_stream, dict, options = {})
119
+ encoder = Encoder.new(output_stream, build_encoder_options(options))
120
+ encoder.encode_stream_with_dict(input_stream.read, dict)
121
+ end
122
+
123
+ # Decompress a frame produced by compress_with_dict. The frame's
124
+ # dictionary ID is verified against `dict` before decoding.
125
+ #
126
+ # @param input_stream [IO] compressed frame
127
+ # @param output_stream [IO] plaintext
128
+ # @param dict [Dictionary]
129
+ # @return [void]
130
+ def decompress_with_dict(input_stream, output_stream, dict)
131
+ output_stream.set_encoding(Encoding::BINARY)
132
+ decoder = Decoder.new(input_stream)
133
+ output_stream.write(decoder.decode_stream_with_dict(dict))
134
+ end
135
+
136
+ # Class-level convenience mirrors of the dictionary API.
137
+ #
138
+ # @return [String] compressed frame / decompressed plaintext
139
+ def self.compress_with_dict(data, dict, **options)
140
+ output = StringIO.new
141
+ output.set_encoding(Encoding::BINARY)
142
+ new(options).compress_with_dict(
143
+ StringIO.new(data.to_s.b), output, dict, options
144
+ )
145
+ output.string
146
+ end
147
+
148
+ def self.decompress_with_dict(data, dict)
149
+ output = StringIO.new
150
+ output.set_encoding(Encoding::BINARY)
151
+ new.decompress_with_dict(StringIO.new(data.to_s.b), output, dict)
152
+ output.string
153
+ end
154
+
104
155
  private
105
156
 
106
157
  # Build encoder options from compression options
@@ -116,6 +167,7 @@ module Omnizip
116
167
  level = case options
117
168
  when nil then nil
118
169
  when Hash then options[:level]
170
+ # allowed: options is a public parameter; any object with #level is honored
119
171
  else options.level if options.respond_to?(:level)
120
172
  end
121
173
 
@@ -1,5 +1,8 @@
1
1
  # frozen_string_literal: true
2
2
 
3
+ require "fileutils"
4
+ require "stringio"
5
+
3
6
  module Omnizip
4
7
  module Chunked
5
8
  # Write large files incrementally in chunks
@@ -16,6 +16,8 @@
16
16
  # See the COPYING file for the complete text of the license.
17
17
  #
18
18
 
19
+ require "fileutils"
20
+
19
21
  module Omnizip
20
22
  module Commands
21
23
  # Command to extract .7z archives to a directory.
@@ -222,7 +224,7 @@ module Omnizip
222
224
 
223
225
  extracted.size
224
226
  ensure
225
- archive&.close if archive.respond_to?(:close)
227
+ archive.close if archive.is_a?(Omnizip::Zip::File)
226
228
  end
227
229
 
228
230
  def extract_gzip(archive_file, output_dir, verbose)
@@ -69,7 +69,7 @@ module Omnizip
69
69
  entries = if patterns || excludes
70
70
  filter_entries(archive, patterns, excludes)
71
71
  else
72
- archive.respond_to?(:entries) ? archive.entries : archive.to_a
72
+ archive.entries
73
73
  end
74
74
 
75
75
  if count_only
@@ -91,6 +91,8 @@ module Omnizip
91
91
  rescue StandardError => e
92
92
  raise Omnizip::CompressionError,
93
93
  "Failed to list archive: #{e.message}"
94
+ ensure
95
+ archive.close if archive.is_a?(Omnizip::Zip::File)
94
96
  end
95
97
 
96
98
  def filter_entries(archive, patterns, excludes)
@@ -102,8 +104,7 @@ module Omnizip
102
104
  # Add exclude patterns
103
105
  excludes&.each { |pattern| filter.exclude_pattern(pattern) }
104
106
 
105
- entries = archive.respond_to?(:entries) ? archive.entries : archive.to_a
106
- filter.filter(entries)
107
+ filter.filter(archive.entries)
107
108
  end
108
109
 
109
110
  def display_simple_listing_filtered(entries)
@@ -33,6 +33,8 @@ module Omnizip
33
33
  puts "Solid: #{profile.solid}"
34
34
  puts "Description: #{profile.description}"
35
35
 
36
+ # ProfileRegistry accepts any CompressionProfile subclass.
37
+ # allowed: a registered subclass may expose base_profile
36
38
  return unless profile.respond_to?(:base_profile) && profile.base_profile
37
39
 
38
40
  puts "Based on: #{profile.base_profile.name}"
@@ -123,6 +123,7 @@ module Omnizip
123
123
  end
124
124
 
125
125
  handler = Omnizip::ArchiveHandler.for(format)
126
+ # allowed: handler is a registered duck; missing add_entry raises below
126
127
  if handler.respond_to?(:add_entry)
127
128
  handler.add_entry(archive_path, entry_name, source_path)
128
129
  else
@@ -142,6 +143,7 @@ module Omnizip
142
143
  def remove_from_archive(archive_path, entry_name, format: DEFAULT_FORMAT)
143
144
  require_archive!(archive_path)
144
145
  handler = Omnizip::ArchiveHandler.for(format)
146
+ # allowed: handler is a registered duck; missing remove_entry raises below
145
147
  unless handler.respond_to?(:remove_entry)
146
148
  raise Omnizip::UnsupportedFormatError,
147
149
  "Format #{format.inspect} does not support removing entries"
@@ -159,10 +159,14 @@ module Omnizip
159
159
  entry
160
160
  else
161
161
  # Try common filename methods
162
+ # Narrowing this changes which files a filter matches.
163
+ # allowed: entries may be foreign objects
162
164
  if entry.respond_to?(:name)
163
165
  entry.name
166
+ # allowed: entries may be foreign objects
164
167
  elsif entry.respond_to?(:path)
165
168
  entry.path
169
+ # allowed: entries may be foreign objects
166
170
  elsif entry.respond_to?(:filename)
167
171
  entry.filename
168
172
  else
@@ -1,5 +1,7 @@
1
1
  # frozen_string_literal: true
2
2
 
3
+ require "fileutils"
4
+
3
5
  module Omnizip
4
6
  module Extraction
5
7
  # Coordinates selective extraction from archives
@@ -120,8 +122,10 @@ module Omnizip
120
122
  #
121
123
  # @return [Array] All entries
122
124
  def list_all
125
+ # allowed: public API takes any archive object, including foreign ones
123
126
  if @archive.respond_to?(:entries)
124
127
  @archive.entries
128
+ # allowed: public API takes any Enumerable archive
125
129
  elsif @archive.respond_to?(:each)
126
130
  @archive.to_a
127
131
  else
@@ -150,10 +154,13 @@ module Omnizip
150
154
  # @param entry [Object] Entry to read
151
155
  # @return [String] Entry content
152
156
  def read_entry_content(entry)
157
+ # allowed: entries come from a caller-supplied archive
153
158
  if entry.respond_to?(:read)
154
159
  entry.read
160
+ # allowed: rubyzip-style entries expose get_input_stream, not read
155
161
  elsif entry.respond_to?(:get_input_stream)
156
162
  entry.get_input_stream.read
163
+ # allowed: some archives read entry content from the archive
157
164
  elsif @archive.respond_to?(:read)
158
165
  @archive.read(entry)
159
166
  else
@@ -1,6 +1,7 @@
1
1
  # frozen_string_literal: true
2
2
 
3
3
  require "marcel"
4
+ require "stringio"
4
5
 
5
6
  module Omnizip
6
7
  # File type detection module using Marcel for MIME type detection
@@ -95,16 +96,19 @@ module Omnizip
95
96
  return nil unless io
96
97
 
97
98
  # Save current position
99
+ # allowed: caller-supplied IO; only saved when the stream tracks one
98
100
  original_pos = io.pos if io.respond_to?(:pos)
99
101
 
100
102
  mime_type = Marcel::MimeType.for(io, name: filename)
101
103
 
102
104
  # Restore position
105
+ # allowed: caller-supplied IO capability probe
103
106
  io.seek(original_pos) if original_pos && io.respond_to?(:seek)
104
107
 
105
108
  mime_type
106
109
  rescue StandardError
107
110
  # Attempt to restore position even on error
111
+ # allowed: caller-supplied IO capability probe
108
112
  io.seek(original_pos) if original_pos && io.respond_to?(:seek)
109
113
  nil
110
114
  end
@@ -1,5 +1,7 @@
1
1
  # frozen_string_literal: true
2
2
 
3
+ require "fileutils"
4
+
3
5
  module Omnizip
4
6
  module Formats
5
7
  # OLE compound document format support
@@ -1,6 +1,7 @@
1
1
  # frozen_string_literal: true
2
2
 
3
3
  require "English"
4
+ require "fileutils"
4
5
  require "tmpdir"
5
6
  module Omnizip
6
7
  module Formats
@@ -1,5 +1,7 @@
1
1
  # frozen_string_literal: true
2
2
 
3
+ require "fileutils"
4
+
3
5
  module Omnizip
4
6
  module Formats
5
7
  module Rar
@@ -1,6 +1,7 @@
1
1
  # frozen_string_literal: true
2
2
 
3
3
  require "fileutils"
4
+ require "stringio"
4
5
 
5
6
  module Omnizip
6
7
  module Formats
@@ -287,8 +288,6 @@ module Omnizip
287
288
  # @param entry [Models::RarEntry] File entry
288
289
  # @return [StringIO] Compressed data stream
289
290
  def read_compressed_data(entry)
290
- require "stringio"
291
-
292
291
  # Find the entry's data offset in the archive
293
292
  File.open(@file_path, "rb") do |io|
294
293
  # Skip signature and headers
@@ -1,5 +1,6 @@
1
1
  # frozen_string_literal: true
2
2
 
3
+ require "fileutils"
3
4
  require "stringio"
4
5
  require "tempfile"
5
6
 
@@ -490,7 +490,15 @@ num_files)
490
490
 
491
491
  def compress_with_lzma2(data)
492
492
  # Use 7-Zip SDK LZMA2 encoder for 7-Zip format
493
+ #
494
+ # The encoder dictionary is sized to the data but never
495
+ # exceeds the size announced in the coder properties
496
+ # (options default 8MB): the decoder allocates its window
497
+ # from the announced size, so a larger encoder dictionary
498
+ # could emit matches reading outside the decoder window.
499
+ announced = @options[:dict_size] || (8 * 1024 * 1024)
493
500
  dict_size = [4096, data.bytesize].max
501
+ dict_size = announced if announced < dict_size
494
502
 
495
503
  encoder = Omnizip::Implementations::SevenZip::LZMA2::Encoder.new(
496
504
  dict_size: dict_size,
@@ -1,5 +1,7 @@
1
1
  # frozen_string_literal: true
2
2
 
3
+ require "fileutils"
4
+
3
5
  module Omnizip
4
6
  module Formats
5
7
  module Tar
@@ -2,6 +2,8 @@
2
2
 
3
3
  require "zlib"
4
4
  require "digest"
5
+ require "fileutils"
6
+ require "stringio"
5
7
 
6
8
  module Omnizip
7
9
  module Formats
@@ -3,6 +3,7 @@
3
3
  require "zlib"
4
4
  require "digest"
5
5
  require "fileutils"
6
+ require "stringio"
6
7
 
7
8
  module Omnizip
8
9
  module Formats
@@ -1,5 +1,7 @@
1
1
  # frozen_string_literal: true
2
2
 
3
+ require "stringio"
4
+
3
5
  module Omnizip
4
6
  module Formats
5
7
  # XZ compression format