omnizip 0.3.22 → 0.3.24
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
checksums.yaml
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
---
|
|
2
2
|
SHA256:
|
|
3
|
-
metadata.gz:
|
|
4
|
-
data.tar.gz:
|
|
3
|
+
metadata.gz: a0119d9a2a28f5bc71cc0aa63f142b373196b8ca3f4abc0db93a71e0a9e45051
|
|
4
|
+
data.tar.gz: b3f6b837610a9005322e25dc5fd9d79a34529b9cd29d2c4d634efb3d64ad50fa
|
|
5
5
|
SHA512:
|
|
6
|
-
metadata.gz:
|
|
7
|
-
data.tar.gz:
|
|
6
|
+
metadata.gz: dfa90d09bbe46eed96bec5565d1751e2bed9d47ee0e9a52b587c3cbad87aaf650e1083a4e8e9d3a22b7089ff0cf218386a2f55f637691f0de363d7646c114b3a
|
|
7
|
+
data.tar.gz: 5fecb94fb627f48411717bd7dcc4c9243f86b4030b4a24a951d8f0816a123e6ec7288cd4c6920e70a89fa0297e0d082389370341d4792d87985c6198ec7c2bbe
|
data/CHANGELOG.md
CHANGED
|
@@ -7,6 +7,57 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
|
|
|
7
7
|
|
|
8
8
|
## [Unreleased]
|
|
9
9
|
|
|
10
|
+
## [0.3.24] - 2026-08-26
|
|
11
|
+
|
|
12
|
+
### Changed
|
|
13
|
+
- LZMA2 encoder hot paths avoid per-call allocations: the
|
|
14
|
+
range-encoder symbol drain reuses its 10 KB scratch buffer
|
|
15
|
+
(profiles showed String#* at ~half of encoder CPU), and the optimal
|
|
16
|
+
parser compares 8-byte windows and selects best matches without
|
|
17
|
+
intermediate strings/arrays. ~180k fewer object allocations and
|
|
18
|
+
~94 MB less churn per 138 KB compressed, with byte-identical
|
|
19
|
+
output; covers both the xz and 7-Zip paths.
|
|
20
|
+
|
|
21
|
+
## [0.3.23] - 2026-08-26
|
|
22
|
+
|
|
23
|
+
### Changed
|
|
24
|
+
- Zstd encoder hot paths use allocation-free integer reads instead of
|
|
25
|
+
`String#byteslice` + `unpack1` comparisons (profiling showed GC at
|
|
26
|
+
52% of encoder time): default-level compression is ~3.4x faster and
|
|
27
|
+
level 22 ~1.9x on the benchmark corpus, with byte-identical output.
|
|
28
|
+
|
|
29
|
+
## [0.3.22] - 2026-08-25
|
|
30
|
+
|
|
31
|
+
### Added
|
|
32
|
+
- Zstandard dictionary compression (`Dictionary.from_raw` /
|
|
33
|
+
`serialize` / `deserialize`, `compress_with_dict`,
|
|
34
|
+
`decompress_with_dict`): the dictionary content primes the match
|
|
35
|
+
finder as shared history and the frame header carries the
|
|
36
|
+
Dictionary_ID, which the decoder verifies (Phase-1 scope, as in the
|
|
37
|
+
Rust reference; entropy-table preloading is future work).
|
|
38
|
+
|
|
39
|
+
## [0.3.21] - 2026-08-25
|
|
40
|
+
|
|
41
|
+
### Fixed
|
|
42
|
+
- 7-Zip encoder dictionary is now capped at the size announced in the
|
|
43
|
+
coder properties (prevents matches reading outside the decoder
|
|
44
|
+
window for inputs above the announced size).
|
|
45
|
+
|
|
46
|
+
### Changed
|
|
47
|
+
- Zstd lazy levels (6+) gained the rep0 fast-path and backward
|
|
48
|
+
extension: levels 6-12 improve 0.180 -> 0.154 and levels 19-22
|
|
49
|
+
0.146 -> 0.126 on the benchmark corpus; the level scale is now
|
|
50
|
+
monotonic.
|
|
51
|
+
- Zstd adaptive block splitting: heterogeneous chunks of 32 KiB or
|
|
52
|
+
more split into 16 KiB sub-blocks so entropy tables fit each
|
|
53
|
+
content regime.
|
|
54
|
+
|
|
55
|
+
## [0.3.20] - 2026-08-25
|
|
56
|
+
|
|
57
|
+
### Removed
|
|
58
|
+
- Dead `SevenZipLZMA2`/`XZLZMA2` algorithm wrappers (never
|
|
59
|
+
autoloaded; one referenced a nonexistent decoder constant).
|
|
60
|
+
|
|
10
61
|
## [0.3.19] - 2026-08-25
|
|
11
62
|
|
|
12
63
|
### Fixed
|
|
@@ -67,11 +67,12 @@ module Omnizip
|
|
|
67
67
|
].min
|
|
68
68
|
|
|
69
69
|
len = 2
|
|
70
|
-
# Compare 8 bytes at a time using 64-bit integers
|
|
70
|
+
# Compare 8 bytes at a time using 64-bit integers read
|
|
71
|
+
# without intermediate strings (byteslice + unpack1
|
|
72
|
+
# allocates two Strings per step and dominated GC time).
|
|
71
73
|
while len + 8 <= max_match_len
|
|
72
|
-
|
|
73
|
-
|
|
74
|
-
v2 = buf.byteslice(buf_back_index + len, 8).unpack1("Q<")
|
|
74
|
+
v1 = read8(buf, buf_pos + len)
|
|
75
|
+
v2 = read8(buf, buf_back_index + len)
|
|
75
76
|
break if v1 != v2
|
|
76
77
|
|
|
77
78
|
len += 8
|
|
@@ -104,7 +105,15 @@ module Omnizip
|
|
|
104
105
|
# A match may reference anything before `position`: bytes already
|
|
105
106
|
# committed to the dictionary *and* bytes decoded earlier within
|
|
106
107
|
# the chunk currently being encoded.
|
|
107
|
-
best_normal =
|
|
108
|
+
best_normal = nil
|
|
109
|
+
matches.each do |m|
|
|
110
|
+
next if best_normal &&
|
|
111
|
+
(m.length < best_normal.length ||
|
|
112
|
+
(m.length == best_normal.length &&
|
|
113
|
+
m.distance <= best_normal.distance))
|
|
114
|
+
|
|
115
|
+
best_normal = m
|
|
116
|
+
end
|
|
108
117
|
|
|
109
118
|
if best_normal && best_normal.length >= nice_len &&
|
|
110
119
|
best_normal.distance <= position
|
|
@@ -126,6 +135,18 @@ module Omnizip
|
|
|
126
135
|
end
|
|
127
136
|
end
|
|
128
137
|
|
|
138
|
+
# Allocation-free little-endian 8-byte read.
|
|
139
|
+
def read8(buf, pos)
|
|
140
|
+
buf.getbyte(pos) |
|
|
141
|
+
(buf.getbyte(pos + 1) << 8) |
|
|
142
|
+
(buf.getbyte(pos + 2) << 16) |
|
|
143
|
+
(buf.getbyte(pos + 3) << 24) |
|
|
144
|
+
(buf.getbyte(pos + 4) << 32) |
|
|
145
|
+
(buf.getbyte(pos + 5) << 40) |
|
|
146
|
+
(buf.getbyte(pos + 6) << 48) |
|
|
147
|
+
(buf.getbyte(pos + 7) << 56)
|
|
148
|
+
end
|
|
149
|
+
|
|
129
150
|
# Constants
|
|
130
151
|
REPS = 4
|
|
131
152
|
MATCH_LEN_MAX = 273 # From lzma.h
|
|
@@ -131,11 +131,29 @@ module Omnizip
|
|
|
131
131
|
min_match: MIN_MATCH_ECONOMICAL }
|
|
132
132
|
end
|
|
133
133
|
|
|
134
|
-
|
|
135
|
-
|
|
134
|
+
# Allocation-free little-endian reads; the hot loops call
|
|
135
|
+
# these millions of times, and String#byteslice + unpack1
|
|
136
|
+
# showed up as half the encoder's time via GC pressure.
|
|
137
|
+
def read4(src, pos)
|
|
138
|
+
src.getbyte(pos) |
|
|
136
139
|
(src.getbyte(pos + 1) << 8) |
|
|
137
140
|
(src.getbyte(pos + 2) << 16) |
|
|
138
141
|
(src.getbyte(pos + 3) << 24)
|
|
142
|
+
end
|
|
143
|
+
|
|
144
|
+
def read8(src, pos)
|
|
145
|
+
src.getbyte(pos) |
|
|
146
|
+
(src.getbyte(pos + 1) << 8) |
|
|
147
|
+
(src.getbyte(pos + 2) << 16) |
|
|
148
|
+
(src.getbyte(pos + 3) << 24) |
|
|
149
|
+
(src.getbyte(pos + 4) << 32) |
|
|
150
|
+
(src.getbyte(pos + 5) << 40) |
|
|
151
|
+
(src.getbyte(pos + 6) << 48) |
|
|
152
|
+
(src.getbyte(pos + 7) << 56)
|
|
153
|
+
end
|
|
154
|
+
|
|
155
|
+
def hash4(src, pos, h_bits)
|
|
156
|
+
v = read4(src, pos)
|
|
139
157
|
# 32-bit wrapping multiply (C/uint32_t), then the high bits.
|
|
140
158
|
((v * PRIME4_BYTES) & 0xFFFFFFFF) >> (32 - h_bits)
|
|
141
159
|
end
|
|
@@ -148,8 +166,8 @@ module Omnizip
|
|
|
148
166
|
b_size = b.bytesize
|
|
149
167
|
while len + 8 <= limit && a_pos + len + 8 <= a_size &&
|
|
150
168
|
b_pos + len + 8 <= b_size
|
|
151
|
-
wa = a
|
|
152
|
-
wb = b
|
|
169
|
+
wa = read8(a, a_pos + len)
|
|
170
|
+
wb = read8(b, b_pos + len)
|
|
153
171
|
if wa == wb
|
|
154
172
|
len += 8
|
|
155
173
|
else
|
|
@@ -289,8 +307,7 @@ module Omnizip
|
|
|
289
307
|
# rubocop:disable-next Metrics/AbcSize
|
|
290
308
|
def rep0_match(src, ip, rep0, min_match, limit, anchor)
|
|
291
309
|
return nil if rep0 <= 0 || ip <= rep0
|
|
292
|
-
return nil if src
|
|
293
|
-
src.byteslice(ip - rep0, MIN_MATCH)
|
|
310
|
+
return nil if read4(src, ip) != read4(src, ip - rep0)
|
|
294
311
|
|
|
295
312
|
m_len = MIN_MATCH + count_match(
|
|
296
313
|
src, ip + MIN_MATCH, src, ip + MIN_MATCH - rep0,
|
|
@@ -346,7 +363,7 @@ module Omnizip
|
|
|
346
363
|
break if dist >= max_distance
|
|
347
364
|
break if candidate + MIN_MATCH > size
|
|
348
365
|
|
|
349
|
-
if src
|
|
366
|
+
if read4(src, ip) == read4(src, candidate)
|
|
350
367
|
m_len = MIN_MATCH + count_match(src, ip + MIN_MATCH,
|
|
351
368
|
src, candidate + MIN_MATCH,
|
|
352
369
|
[max_extend - MIN_MATCH, 0].max)
|
|
@@ -397,7 +414,7 @@ module Omnizip
|
|
|
397
414
|
if candidate.positive? && candidate < ip
|
|
398
415
|
dist = ip - candidate
|
|
399
416
|
if dist < max_distance && candidate + MIN_MATCH <= size &&
|
|
400
|
-
src
|
|
417
|
+
read4(src, ip) == read4(src, candidate)
|
|
401
418
|
best_len = MIN_MATCH + count_match(
|
|
402
419
|
src, ip + MIN_MATCH, src, candidate + MIN_MATCH,
|
|
403
420
|
[limit + MIN_MATCH_ECONOMICAL - ip - MIN_MATCH, 0].max
|
|
@@ -339,8 +339,11 @@ module Omnizip
|
|
|
339
339
|
def encode_queued_symbols(encoder, output)
|
|
340
340
|
return if encoder.none?
|
|
341
341
|
|
|
342
|
-
#
|
|
343
|
-
|
|
342
|
+
# Reused scratch buffer: allocating a fresh 10 KB String
|
|
343
|
+
# per drain showed up as half the encoder's runtime via
|
|
344
|
+
# GC (String#* in profiles).
|
|
345
|
+
@symbol_buffer ||= "\0".b * 10_000
|
|
346
|
+
temp_buffer = @symbol_buffer
|
|
344
347
|
out_pos = Omnizip::Algorithms::LZMA::IntRef.new(0)
|
|
345
348
|
|
|
346
349
|
# Track size before encoding
|
data/lib/omnizip/version.rb
CHANGED
metadata
CHANGED
|
@@ -1,14 +1,14 @@
|
|
|
1
1
|
--- !ruby/object:Gem::Specification
|
|
2
2
|
name: omnizip
|
|
3
3
|
version: !ruby/object:Gem::Version
|
|
4
|
-
version: 0.3.
|
|
4
|
+
version: 0.3.24
|
|
5
5
|
platform: ruby
|
|
6
6
|
authors:
|
|
7
7
|
- Ribose Inc.
|
|
8
8
|
autorequire:
|
|
9
9
|
bindir: exe
|
|
10
10
|
cert_chain: []
|
|
11
|
-
date: 2026-08-
|
|
11
|
+
date: 2026-08-26 00:00:00.000000000 Z
|
|
12
12
|
dependencies:
|
|
13
13
|
- !ruby/object:Gem::Dependency
|
|
14
14
|
name: base64
|