omnizip 0.3.25 → 0.3.26

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -41,11 +41,8 @@ module Omnizip
41
41
  class BZip2 < Algorithm
42
42
  # Nested classes - autoloaded
43
43
  autoload :Bwt, "omnizip/algorithms/bzip2/bwt"
44
- autoload :Mtf, "omnizip/algorithms/bzip2/mtf"
45
44
  autoload :Rle, "omnizip/algorithms/bzip2/rle"
46
- autoload :Huffman, "omnizip/algorithms/bzip2/huffman"
47
- autoload :Encoder, "omnizip/algorithms/bzip2/encoder"
48
- autoload :Decoder, "omnizip/algorithms/bzip2/decoder"
45
+ autoload :Bz2, "omnizip/algorithms/bzip2/bz2"
49
46
 
50
47
  # Cross-namespace dependencies - autoloaded
51
48
  autoload :Crc32, "omnizip/checksums/crc32"
@@ -69,10 +66,8 @@ module Omnizip
69
66
  # @param options [Models::CompressionOptions] Compression options
70
67
  # @return [void]
71
68
  def compress(input_stream, output_stream, options = nil)
72
- input_data = input_stream.read
73
- encoder = Encoder.new(output_stream,
74
- build_encoder_options(options))
75
- encoder.encode_stream(input_data)
69
+ level = level_from(options)
70
+ output_stream.write(Bz2.compress(input_stream.read, level))
76
71
  end
77
72
 
78
73
  # Decompress BZip2-compressed data
@@ -83,43 +78,20 @@ module Omnizip
83
78
  # @return [void]
84
79
  def decompress(input_stream, output_stream, _options = nil)
85
80
  output_stream.set_encoding(Encoding::BINARY)
86
- decoder = Decoder.new(input_stream)
87
- decompressed = decoder.decode_stream
88
- output_stream.write(decompressed)
81
+ output_stream.write(Bz2.decompress(input_stream.read))
89
82
  end
90
83
 
91
84
  private
92
85
 
93
- # Build encoder options from compression options
94
- #
95
- # @param options [Models::CompressionOptions, nil] Compression opts
96
- # @return [Hash] Encoder options
97
- def build_encoder_options(options)
98
- return {} if options.nil?
99
-
100
- opts = {}
86
+ # Compression level (1-9) from options; bzip2's levels map
87
+ # directly to 100 KB..900 KB block sizes.
88
+ def level_from(options)
89
+ return 9 if options.nil?
90
+ return (options[:level] || 9).clamp(1, 9) if options.is_a?(Hash)
101
91
 
102
92
  # allowed: options is a public parameter; any object with #level is honored
103
- if options.respond_to?(:level)
104
- level = options.level || 9
105
- opts[:block_size] = block_size_for_level(level)
106
- end
107
-
108
- opts
109
- end
110
-
111
- # Get block size based on compression level
112
- #
113
- # BZip2 traditionally uses levels 1-9 corresponding to
114
- # 100KB-900KB block sizes
115
- #
116
- # @param level [Integer] Compression level (1-9)
117
- # @return [Integer] Block size in bytes
118
- def block_size_for_level(level)
119
- # Clamp level to valid range
120
- level = [[level, 1].max, 9].min
121
- # Each level = 100KB
122
- level * 100_000
93
+ level = options.respond_to?(:level) ? options.level : nil
94
+ (level || 9).clamp(1, 9)
123
95
  end
124
96
  end
125
97
  end
@@ -1,5 +1,5 @@
1
1
  # frozen_string_literal: true
2
2
 
3
3
  module Omnizip
4
- VERSION = "0.3.25"
4
+ VERSION = "0.3.26"
5
5
  end
metadata CHANGED
@@ -1,7 +1,7 @@
1
1
  --- !ruby/object:Gem::Specification
2
2
  name: omnizip
3
3
  version: !ruby/object:Gem::Version
4
- version: 0.3.25
4
+ version: 0.3.26
5
5
  platform: ruby
6
6
  authors:
7
7
  - Ribose Inc.
@@ -226,10 +226,7 @@ files:
226
226
  - lib/omnizip/algorithms/.keep
227
227
  - lib/omnizip/algorithms/bzip2.rb
228
228
  - lib/omnizip/algorithms/bzip2/bwt.rb
229
- - lib/omnizip/algorithms/bzip2/decoder.rb
230
- - lib/omnizip/algorithms/bzip2/encoder.rb
231
- - lib/omnizip/algorithms/bzip2/huffman.rb
232
- - lib/omnizip/algorithms/bzip2/mtf.rb
229
+ - lib/omnizip/algorithms/bzip2/bz2.rb
233
230
  - lib/omnizip/algorithms/bzip2/rle.rb
234
231
  - lib/omnizip/algorithms/deflate.rb
235
232
  - lib/omnizip/algorithms/deflate/constants.rb
@@ -1,187 +0,0 @@
1
- # frozen_string_literal: true
2
-
3
- # Copyright (C) 2025 Ribose Inc.
4
- #
5
- # Permission is hereby granted, free of charge, to any person obtaining a
6
- # copy of this software and associated documentation files (the "Software"),
7
- # to deal in the Software without restriction, including without limitation
8
- # the rights to use, copy, modify, merge, publish, distribute, sublicense,
9
- # and/or sell copies of the Software, and to permit persons to whom the
10
- # Software is furnished to do so, subject to the following conditions:
11
- #
12
- # The above copyright notice and this permission notice shall be included in
13
- # all copies or substantial portions of the Software.
14
- #
15
- # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
- # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
- # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
- # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
- # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING
20
- # FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER
21
- # DEALINGS IN THE SOFTWARE.
22
-
23
- module Omnizip
24
- module Algorithms
25
- class BZip2 < Algorithm
26
- # BZip2 Decoder
27
- #
28
- # Reverses the full BZip2 compression pipeline:
29
- # 1. Read block headers
30
- # 2. Decode Huffman coding
31
- # 3. Reverse Run-Length Encoding (RLE)
32
- # 4. Reverse Move-to-Front Transform (MTF)
33
- # 5. Reverse Burrows-Wheeler Transform (BWT)
34
- # 6. Verify CRC32 checksum
35
- #
36
- # Processes each block independently and concatenates results.
37
- class Decoder
38
- attr_reader :input
39
-
40
- # Initialize decoder
41
- #
42
- # @param input [IO] Input stream
43
- # @param options [Hash] Decoding options
44
- def initialize(input, _options = {})
45
- @input = input
46
- @bwt = Bwt.new
47
- @mtf = Mtf.new
48
- @rle = Rle.new
49
- @huffman = Huffman.new
50
- end
51
-
52
- # Decode stream using BZip2 algorithm
53
- #
54
- # @return [String] Decoded data
55
- def decode_stream
56
- result = []
57
-
58
- # Read and decode all blocks
59
- loop do
60
- block_data = decode_block
61
- break unless block_data
62
-
63
- result << block_data
64
- end
65
-
66
- result.join.b
67
- end
68
-
69
- private
70
-
71
- # Decode single block
72
- #
73
- # @return [String, nil] Decoded block or nil if no more blocks
74
- def decode_block
75
- # Read block header
76
- crc_bytes = @input.read(4)
77
- return nil unless crc_bytes && crc_bytes.length == 4
78
-
79
- expected_crc = crc_bytes.unpack1("N")
80
- primary_index = @input.read(4).unpack1("N")
81
- @input.read(4).unpack1("N")
82
- rle_length = @input.read(4).unpack1("N")
83
-
84
- # Read Huffman code table
85
- codes = read_huffman_codes
86
-
87
- # Read encoded data
88
- encoded_length = @input.read(4).unpack1("N")
89
- encoded_data = @input.read(encoded_length)
90
-
91
- # Rebuild Huffman tree from codes
92
- tree = rebuild_huffman_tree(codes)
93
-
94
- # Decode Huffman
95
- rle_data = @huffman.decode(encoded_data, tree, rle_length)
96
-
97
- # Reverse RLE
98
- mtf_data = @rle.decode(rle_data)
99
-
100
- # Reverse MTF
101
- bwt_data = @mtf.decode(mtf_data)
102
-
103
- # Reverse BWT
104
- original_data = @bwt.decode(bwt_data, primary_index)
105
-
106
- # Verify CRC
107
- actual_crc = Checksums::Crc32.calculate(original_data)
108
- if actual_crc != expected_crc
109
- raise "CRC mismatch: expected #{expected_crc}, " \
110
- "got #{actual_crc}"
111
- end
112
-
113
- original_data
114
- end
115
-
116
- # Read Huffman code table from stream
117
- #
118
- # @return [Hash<Integer, String>] Symbol => binary code
119
- def read_huffman_codes
120
- codes = {}
121
- code_count = @input.read(2).unpack1("n")
122
-
123
- code_count.times do
124
- symbol = @input.read(1).unpack1("C")
125
- code_length = @input.read(1).unpack1("C")
126
- codes[symbol] = code_length
127
- end
128
-
129
- codes
130
- end
131
-
132
- # Rebuild Huffman tree from code lengths
133
- #
134
- # Creates canonical Huffman codes and builds tree
135
- #
136
- # @param code_lengths [Hash<Integer, Integer>] Symbol => length
137
- # @return [Huffman::Node] Root of Huffman tree
138
- def rebuild_huffman_tree(code_lengths)
139
- # Sort symbols by code length, then by symbol value
140
- sorted_symbols = code_lengths.sort_by { |sym, len| [len, sym] }
141
-
142
- # Generate canonical codes
143
- codes = {}
144
- code_value = 0
145
- prev_length = 0
146
-
147
- sorted_symbols.each do |symbol, length|
148
- # Shift code value for new length
149
- code_value <<= (length - prev_length)
150
- codes[symbol] = format("%0#{length}b", code_value)
151
- code_value += 1
152
- prev_length = length
153
- end
154
-
155
- # Build tree from codes
156
- build_tree_from_codes(codes)
157
- end
158
-
159
- # Build Huffman tree from code strings
160
- #
161
- # @param codes [Hash<Integer, String>] Symbol => binary code
162
- # @return [Huffman::Node] Root node
163
- def build_tree_from_codes(codes)
164
- root = Huffman::Node.new(nil, 0)
165
-
166
- codes.each do |symbol, code|
167
- current = root
168
-
169
- code.each_char do |bit|
170
- if bit == "0"
171
- current.left ||= Huffman::Node.new(nil, 0)
172
- current = current.left
173
- else
174
- current.right ||= Huffman::Node.new(nil, 0)
175
- current = current.right
176
- end
177
- end
178
-
179
- current.symbol = symbol
180
- end
181
-
182
- root
183
- end
184
- end
185
- end
186
- end
187
- end
@@ -1,231 +0,0 @@
1
- # frozen_string_literal: true
2
-
3
- # Copyright (C) 2025 Ribose Inc.
4
- #
5
- # Permission is hereby granted, free of charge, to any person obtaining a
6
- # copy of this software and associated documentation files (the "Software"),
7
- # to deal in the Software without restriction, including without limitation
8
- # the rights to use, copy, modify, merge, publish, distribute, sublicense,
9
- # and/or sell copies of the Software, and to permit persons to whom the
10
- # Software is furnished to do so, subject to the following conditions:
11
- #
12
- # The above copyright notice and this permission notice shall be included in
13
- # all copies or substantial portions of the Software.
14
- #
15
- # THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
- # IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
- # FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
- # AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
- # LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING
20
- # FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER
21
- # DEALINGS IN THE SOFTWARE.
22
-
23
- module Omnizip
24
- module Algorithms
25
- class BZip2 < Algorithm
26
- # BZip2 Encoder
27
- #
28
- # Orchestrates the full BZip2 compression pipeline:
29
- # 1. Block splitting (configurable block size)
30
- # 2. Burrows-Wheeler Transform (BWT)
31
- # 3. Move-to-Front Transform (MTF)
32
- # 4. Run-Length Encoding (RLE)
33
- # 5. Huffman Coding
34
- # 6. CRC32 checksum calculation
35
- #
36
- # Each block is compressed independently for better parallelization
37
- # potential and error recovery.
38
- class Encoder
39
- attr_reader :output, :block_size
40
-
41
- # Block size constants (in bytes)
42
- MIN_BLOCK_SIZE = 100_000 # 100KB
43
- MAX_BLOCK_SIZE = 900_000 # 900KB
44
- DEFAULT_BLOCK_SIZE = 900_000 # 900KB (level 9)
45
-
46
- # Initialize encoder
47
- #
48
- # @param output [IO] Output stream
49
- # @param options [Hash] Encoding options
50
- # @option options [Integer] :block_size Block size in bytes
51
- def initialize(output, options = {})
52
- @output = output
53
- @block_size = validate_block_size(
54
- options[:block_size] || DEFAULT_BLOCK_SIZE,
55
- )
56
- @bwt = Bwt.new
57
- @mtf = Mtf.new
58
- @rle = Rle.new
59
- @huffman = Huffman.new
60
- end
61
-
62
- # Encode stream using BZip2 algorithm
63
- #
64
- # @param input [String] Input data to encode
65
- # @return [void]
66
- def encode_stream(input)
67
- return if input.empty?
68
-
69
- # Split into blocks and encode each
70
- blocks = split_into_blocks(input)
71
- blocks.each { |block| encode_block(block) }
72
- end
73
-
74
- private
75
-
76
- # Validate and clamp block size to valid range
77
- #
78
- # @param size [Integer] Requested block size
79
- # @return [Integer] Validated block size
80
- def validate_block_size(size)
81
- size.clamp(MIN_BLOCK_SIZE, MAX_BLOCK_SIZE)
82
- end
83
-
84
- # Split input into blocks
85
- #
86
- # @param input [String] Input data
87
- # @return [Array<String>] Array of blocks
88
- def split_into_blocks(input)
89
- blocks = []
90
- offset = 0
91
-
92
- while offset < input.length
93
- block = input[offset, @block_size]
94
- blocks << block if block && !block.empty?
95
- offset += @block_size
96
- end
97
-
98
- blocks
99
- end
100
-
101
- # Encode single block through full pipeline
102
- #
103
- # @param block [String] Block data
104
- # @return [void]
105
- def encode_block(block)
106
- # Calculate CRC of original data
107
- crc = Checksums::Crc32.calculate(block)
108
-
109
- # Apply BWT
110
- bwt_data, primary_index = @bwt.encode(block)
111
-
112
- # Apply MTF
113
- mtf_data = @mtf.encode(bwt_data)
114
-
115
- # Apply RLE
116
- rle_data = @rle.encode(mtf_data)
117
-
118
- # Build frequency table for Huffman
119
- frequencies = build_frequency_table(rle_data)
120
-
121
- # Build Huffman tree and generate codes
122
- tree = @huffman.build_tree(frequencies)
123
- codes = generate_canonical_codes(tree)
124
-
125
- # Encode data with Huffman
126
- encoded_data = @huffman.encode(rle_data, codes)
127
-
128
- # Write block to output
129
- write_block(crc, primary_index, block.length, codes,
130
- encoded_data, rle_data.length)
131
- end
132
-
133
- # Build frequency table from data
134
- #
135
- # @param data [String] Input data
136
- # @return [Hash<Integer, Integer>] Byte => frequency
137
- def build_frequency_table(data)
138
- freq = Hash.new(0)
139
- data.each_byte { |byte| freq[byte] += 1 }
140
- freq
141
- end
142
-
143
- # Generate canonical Huffman codes from tree
144
- #
145
- # @param tree [Huffman::Node] Huffman tree root
146
- # @return [Hash<Integer, String>] Symbol => canonical code
147
- def generate_canonical_codes(tree)
148
- # Get standard codes first
149
- standard_codes = @huffman.generate_codes(tree)
150
-
151
- # Convert to code lengths
152
- code_lengths = {}
153
- standard_codes.each do |symbol, code|
154
- code_lengths[symbol] = code.length
155
- end
156
-
157
- # Generate canonical codes from lengths
158
- # Sort by (length, symbol) to ensure deterministic ordering
159
- sorted_symbols = code_lengths.sort_by { |sym, len| [len, sym] }
160
-
161
- canonical_codes = {}
162
- code_value = 0
163
- prev_length = 0
164
-
165
- sorted_symbols.each do |symbol, length|
166
- # Shift code value for new length
167
- code_value <<= (length - prev_length)
168
- canonical_codes[symbol] = format("%0#{length}b", code_value)
169
- code_value += 1
170
- prev_length = length
171
- end
172
-
173
- canonical_codes
174
- end
175
-
176
- # Write encoded block to output
177
- #
178
- # @param crc [Integer] CRC32 of original block
179
- # @param primary_index [Integer] BWT primary index
180
- # @param original_length [Integer] Original block length
181
- # @param codes [Hash] Huffman codes
182
- # @param encoded_data [String] Huffman-encoded data
183
- # @param rle_length [Integer] Length after RLE
184
- # @return [void]
185
- def write_block(crc, primary_index, original_length, codes,
186
- encoded_data, rle_length)
187
- write_block_header(crc, primary_index, original_length, rle_length)
188
- write_huffman_codes(codes)
189
- write_encoded_data(encoded_data)
190
- end
191
-
192
- # Write block header
193
- #
194
- # @param crc [Integer] CRC32 checksum
195
- # @param primary_index [Integer] BWT primary index
196
- # @param original_length [Integer] Original block length
197
- # @param rle_length [Integer] RLE length
198
- # @return [void]
199
- def write_block_header(crc, primary_index, original_length,
200
- rle_length)
201
- @output.write([crc].pack("N"))
202
- @output.write([primary_index].pack("N"))
203
- @output.write([original_length].pack("N"))
204
- @output.write([rle_length].pack("N"))
205
- end
206
-
207
- # Write Huffman codes to output
208
- #
209
- # @param codes [Hash] Huffman codes
210
- # @return [void]
211
- def write_huffman_codes(codes)
212
- @output.write([codes.size].pack("n"))
213
-
214
- codes.each do |symbol, code|
215
- @output.write([symbol].pack("C"))
216
- @output.write([code.length].pack("C"))
217
- end
218
- end
219
-
220
- # Write encoded data to output
221
- #
222
- # @param encoded_data [String] Huffman-encoded data
223
- # @return [void]
224
- def write_encoded_data(encoded_data)
225
- @output.write([encoded_data.length].pack("N"))
226
- @output.write(encoded_data)
227
- end
228
- end
229
- end
230
- end
231
- end