zip_kit 6.3.3 → 6.3.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +14 -0
- data/bench/write_buffer_bench.rb +183 -0
- data/lib/zip_kit/streamer/heuristic.rb +5 -6
- data/lib/zip_kit/streamer.rb +12 -2
- data/lib/zip_kit/version.rb +1 -1
- data/lib/zip_kit/write_buffer.rb +49 -13
- data/lib/zip_kit/zip_writer.rb +2 -0
- data/rbi/zip_kit.rbi +29 -5
- data/rbi/zip_kit.rbs +27 -4
- data/zip_kit.gemspec +8 -4
- metadata +68 -39
checksums.yaml
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
---
|
|
2
2
|
SHA256:
|
|
3
|
-
metadata.gz:
|
|
4
|
-
data.tar.gz:
|
|
3
|
+
metadata.gz: 83cdf922faca6692b384bc30ba8290f0f15a4567ebce61fccb792a5b5d7993fc
|
|
4
|
+
data.tar.gz: ba3a6fee022ba663d77306850246ac8b33ba5d1e10d81f636ace18cc882dc99f
|
|
5
5
|
SHA512:
|
|
6
|
-
metadata.gz:
|
|
7
|
-
data.tar.gz:
|
|
6
|
+
metadata.gz: ad24aba7c7359f01a11e94ddd1e469fdac2a77c59ece3a113cacebaf900e1cec9a9eee462b385c05d9f1789f9f5c6b48da5098b060e2c5b2c99cc5e312c4d435
|
|
7
|
+
data.tar.gz: 96eb1200ee5bdf95e2c9f6b8f91d856607f7981f0eabf7c5cee75dbb3efa200c7642f0886ffdad1965da123281c9cfe8d545614616702cadf4dc207da83a293d
|
data/CHANGELOG.md
CHANGED
|
@@ -1,3 +1,17 @@
|
|
|
1
|
+
## Unreleased
|
|
2
|
+
|
|
3
|
+
## 6.3.5
|
|
4
|
+
|
|
5
|
+
* Make `WriteBuffer#<<` much faster for lots of tiny writes, like the XML fragments written by libraries such as caxlsx. Strings are no longer copied with `String#b` before being appended: on Ruby 3.4+ `String#append_as_bytes` is used, on older Rubies the string is converted only if Ruby refuses to append it. Mixing binary strings and non-ASCII strings still works, and the writable still always receives binary strings.
|
|
6
|
+
* `WriteBuffer` now flushes before a write would make the buffer exceed its size (instead of after), so that it never outputs chunks larger than the buffer size, and writes which are larger than the buffer size are still passed through without getting copied.
|
|
7
|
+
* `WriteBuffer` no longer preallocates (and immediately discards) a zero-filled String of twice its buffer size, which makes writing ZIPs with lots of small files faster.
|
|
8
|
+
* Require `stringio` in `ZipWriter`, which uses it. Previously writing a ZIP could fail with a `NameError` if nothing else in the process had loaded `stringio`.
|
|
9
|
+
* Add `bench/write_buffer_bench.rb`
|
|
10
|
+
|
|
11
|
+
## 6.3.4
|
|
12
|
+
|
|
13
|
+
* Fix a bug whereby `rollback!` would cause an exception without any entries having been written yet (rollback on first entry).
|
|
14
|
+
|
|
1
15
|
## 6.3.3
|
|
2
16
|
|
|
3
17
|
* Make sure `Writable#<<` converts the strings it is given into binary if they are not already in binary. This fixes an issue where `Heuristic` would suddenly start forwarding strings as-is to downstream callees. There is a lot of spots where the string-to-write gets forwarded and converting in every single one will be quite wasteful, but it can be handy to do in a few key places.
|
|
@@ -0,0 +1,183 @@
|
|
|
1
|
+
# Benchmarks ZipKit::WriteBuffer with lots of tiny writes (like the XML fragments
|
|
2
|
+
# a library such as caxlsx produces) and with large writes, comparing it to the
|
|
3
|
+
# previous implementations of WriteBuffer.
|
|
4
|
+
#
|
|
5
|
+
# bundle exec ruby bench/write_buffer_bench.rb
|
|
6
|
+
require "bundler"
|
|
7
|
+
Bundler.setup
|
|
8
|
+
|
|
9
|
+
require "benchmark"
|
|
10
|
+
require "benchmark/ips"
|
|
11
|
+
require_relative "../lib/zip_kit"
|
|
12
|
+
|
|
13
|
+
# Initialization and flushing as in zip_kit 6.3.x. These are not subclasses of ZipKit::WriteBuffer
|
|
14
|
+
# on purpose: with Ruby 3.4, instances of a subclass can have a different object shape, which
|
|
15
|
+
# can make the instance variable caches of methods they share with ZipKit::WriteBuffer miss.
|
|
16
|
+
module WriteBuffer63x
|
|
17
|
+
def initialize(writable, buffer_size)
|
|
18
|
+
@buf = ("\0".b * (buffer_size * 2)).clear
|
|
19
|
+
@buffer_size = buffer_size
|
|
20
|
+
@writable = writable
|
|
21
|
+
end
|
|
22
|
+
|
|
23
|
+
def flush
|
|
24
|
+
unless @buf.empty?
|
|
25
|
+
@writable << @buf
|
|
26
|
+
@buf.clear
|
|
27
|
+
end
|
|
28
|
+
self
|
|
29
|
+
end
|
|
30
|
+
end
|
|
31
|
+
|
|
32
|
+
# WriteBuffer#<< as of zip_kit 6.3.2
|
|
33
|
+
class WriteBuffer632
|
|
34
|
+
include WriteBuffer63x
|
|
35
|
+
|
|
36
|
+
def <<(data)
|
|
37
|
+
if data.bytesize >= @buffer_size
|
|
38
|
+
flush unless @buf.empty?
|
|
39
|
+
@writable << data
|
|
40
|
+
else
|
|
41
|
+
@buf << data
|
|
42
|
+
flush if @buf.bytesize >= @buffer_size
|
|
43
|
+
end
|
|
44
|
+
self
|
|
45
|
+
end
|
|
46
|
+
end
|
|
47
|
+
|
|
48
|
+
# WriteBuffer#<< as of zip_kit 6.3.3/6.3.4, which copies every string with String#b
|
|
49
|
+
class WriteBuffer634
|
|
50
|
+
include WriteBuffer63x
|
|
51
|
+
|
|
52
|
+
def <<(string)
|
|
53
|
+
if string.bytesize >= @buffer_size
|
|
54
|
+
flush
|
|
55
|
+
@writable << string.b
|
|
56
|
+
else
|
|
57
|
+
@buf << string.b
|
|
58
|
+
flush if @buf.bytesize >= @buffer_size
|
|
59
|
+
end
|
|
60
|
+
self
|
|
61
|
+
end
|
|
62
|
+
end
|
|
63
|
+
|
|
64
|
+
# The simplest possible buffer, which does not deal with encodings or large writes
|
|
65
|
+
class NaiveBuffer
|
|
66
|
+
def initialize(io, buffer_size)
|
|
67
|
+
@io = io
|
|
68
|
+
@buffer_size = buffer_size
|
|
69
|
+
@buf = "".b
|
|
70
|
+
end
|
|
71
|
+
|
|
72
|
+
def <<(fragment)
|
|
73
|
+
@buf << fragment
|
|
74
|
+
flush if @buf.bytesize >= @buffer_size
|
|
75
|
+
self
|
|
76
|
+
end
|
|
77
|
+
|
|
78
|
+
def flush
|
|
79
|
+
return if @buf.empty?
|
|
80
|
+
@io << @buf
|
|
81
|
+
@buf.clear
|
|
82
|
+
end
|
|
83
|
+
end
|
|
84
|
+
|
|
85
|
+
BUFFER_SIZE = 64 * 1024
|
|
86
|
+
IMPLEMENTATIONS = {
|
|
87
|
+
"WriteBuffer" => ZipKit::WriteBuffer,
|
|
88
|
+
"WriteBuffer (6.3.2)" => WriteBuffer632,
|
|
89
|
+
"WriteBuffer (6.3.4, String#b)" => WriteBuffer634,
|
|
90
|
+
"Naive String buffer" => NaiveBuffer
|
|
91
|
+
}
|
|
92
|
+
|
|
93
|
+
# Fragments like the ones produced by caxlsx when writing out a worksheet: mostly frozen
|
|
94
|
+
# literals and short dynamic strings (UTF-8 or US-ASCII), mostly ASCII-only.
|
|
95
|
+
fragments = []
|
|
96
|
+
20_000.times do |row|
|
|
97
|
+
fragments << "<row r=\"" << (row + 1).to_s << "\">"
|
|
98
|
+
6.times do |col|
|
|
99
|
+
fragments << "<c r=\"" << "#{("A".ord + col).chr}#{row + 1}" << "\" s=\"" << "1" << "\""
|
|
100
|
+
fragments << ((col == 3) ? " t=\"inlineStr\"><is><t>Grüße, #{row}</t></is>" : "><v>#{row * col}</v>")
|
|
101
|
+
fragments << "</c>"
|
|
102
|
+
end
|
|
103
|
+
fragments << "</row>"
|
|
104
|
+
end
|
|
105
|
+
n_bytes = fragments.sum(&:bytesize)
|
|
106
|
+
|
|
107
|
+
puts RUBY_DESCRIPTION
|
|
108
|
+
puts "#{fragments.length} tiny writes (#{n_bytes} bytes, #{(n_bytes.to_f / fragments.length).round(1)} bytes per write on average)"
|
|
109
|
+
|
|
110
|
+
Benchmark.ips do |x|
|
|
111
|
+
x.config(time: 5, warmup: 2)
|
|
112
|
+
IMPLEMENTATIONS.each do |name, buffer_class|
|
|
113
|
+
x.report("#{name}, tiny writes into CRC32") do
|
|
114
|
+
buf = buffer_class.new(ZipKit::StreamCRC32.new, BUFFER_SIZE)
|
|
115
|
+
fragments.each { |fragment| buf << fragment }
|
|
116
|
+
buf.flush
|
|
117
|
+
end
|
|
118
|
+
end
|
|
119
|
+
x.compare!
|
|
120
|
+
end
|
|
121
|
+
|
|
122
|
+
Benchmark.ips do |x|
|
|
123
|
+
x.config(time: 5, warmup: 2)
|
|
124
|
+
IMPLEMENTATIONS.each do |name, buffer_class|
|
|
125
|
+
x.report("#{name}, tiny writes into write_deflated_file") do
|
|
126
|
+
ZipKit::Streamer.open(ZipKit::NullWriter) do |zip|
|
|
127
|
+
zip.write_deflated_file("sheet.xml") do |sink|
|
|
128
|
+
buf = buffer_class.new(sink, BUFFER_SIZE)
|
|
129
|
+
fragments.each { |fragment| buf << fragment }
|
|
130
|
+
buf.flush
|
|
131
|
+
end
|
|
132
|
+
end
|
|
133
|
+
end
|
|
134
|
+
end
|
|
135
|
+
x.report("No buffer, tiny writes into write_deflated_file") do
|
|
136
|
+
ZipKit::Streamer.open(ZipKit::NullWriter) do |zip|
|
|
137
|
+
zip.write_deflated_file("sheet.xml") do |sink|
|
|
138
|
+
fragments.each { |fragment| sink << fragment }
|
|
139
|
+
end
|
|
140
|
+
end
|
|
141
|
+
end
|
|
142
|
+
x.compare!
|
|
143
|
+
end
|
|
144
|
+
|
|
145
|
+
large_chunk = Random.new(42).bytes(1024 * 1024)
|
|
146
|
+
Benchmark.ips do |x|
|
|
147
|
+
x.config(time: 5, warmup: 2)
|
|
148
|
+
IMPLEMENTATIONS.each do |name, buffer_class|
|
|
149
|
+
x.report("#{name}, 64 writes of 1MB into CRC32") do
|
|
150
|
+
buf = buffer_class.new(ZipKit::StreamCRC32.new, BUFFER_SIZE)
|
|
151
|
+
64.times { buf << large_chunk }
|
|
152
|
+
buf.flush
|
|
153
|
+
end
|
|
154
|
+
end
|
|
155
|
+
x.compare!
|
|
156
|
+
end
|
|
157
|
+
|
|
158
|
+
__END__
|
|
159
|
+
|
|
160
|
+
Apple M1 Pro, macOS 15.7
|
|
161
|
+
|
|
162
|
+
ruby 3.4.1 (2024-12-25 revision 48d4efcb85) +PRISM [arm64-darwin24], without YJIT
|
|
163
|
+
920000 tiny writes (5189483 bytes, 5.6 bytes per write on average)
|
|
164
|
+
|
|
165
|
+
Comparison:
|
|
166
|
+
Naive String buffer, tiny writes into CRC32: 12.1 i/s
|
|
167
|
+
WriteBuffer, tiny writes into CRC32: 12.0 i/s - 1.01x slower
|
|
168
|
+
WriteBuffer (6.3.2), tiny writes into CRC32: 10.3 i/s - 1.17x slower
|
|
169
|
+
WriteBuffer (6.3.4, String#b), tiny writes into CRC32: 8.0 i/s - 1.51x slower
|
|
170
|
+
|
|
171
|
+
Comparison:
|
|
172
|
+
Naive String buffer, tiny writes into write_deflated_file: 7.6 i/s
|
|
173
|
+
WriteBuffer, tiny writes into write_deflated_file: 7.4 i/s - 1.02x slower
|
|
174
|
+
WriteBuffer (6.3.2), tiny writes into write_deflated_file: 6.8 i/s - 1.12x slower
|
|
175
|
+
WriteBuffer (6.3.4, String#b), tiny writes into write_deflated_file: 5.7 i/s - 1.33x slower
|
|
176
|
+
No buffer, tiny writes into write_deflated_file: 1.8 i/s - 4.18x slower
|
|
177
|
+
|
|
178
|
+
Comparison:
|
|
179
|
+
WriteBuffer (6.3.2), 64 writes of 1MB into CRC32: 464.4 i/s
|
|
180
|
+
WriteBuffer (6.3.4, String#b), 64 writes of 1MB into CRC32: 463.0 i/s - same-ish: difference falls within error
|
|
181
|
+
WriteBuffer, 64 writes of 1MB into CRC32: 462.9 i/s - same-ish: difference falls within error
|
|
182
|
+
Naive String buffer, 64 writes of 1MB into CRC32: 121.5 i/s - 3.82x slower
|
|
183
|
+
|
|
@@ -23,7 +23,7 @@ class ZipKit::Streamer::Heuristic < ZipKit::Streamer::Writable
|
|
|
23
23
|
@filename = filename
|
|
24
24
|
@write_file_options = write_file_options
|
|
25
25
|
|
|
26
|
-
@buf =
|
|
26
|
+
@buf = +"".b # Just use a mutable String
|
|
27
27
|
@deflater = ::Zlib::Deflate.new(Zlib::DEFAULT_COMPRESSION, -::Zlib::MAX_WBITS)
|
|
28
28
|
@bytes_deflated = 0
|
|
29
29
|
|
|
@@ -37,7 +37,7 @@ class ZipKit::Streamer::Heuristic < ZipKit::Streamer::Writable
|
|
|
37
37
|
else
|
|
38
38
|
@buf << bytes
|
|
39
39
|
@deflater.deflate(bytes) { |chunk| @bytes_deflated += chunk.bytesize }
|
|
40
|
-
decide if @buf.
|
|
40
|
+
decide if @buf.bytesize > BYTES_WRITTEN_THRESHOLD
|
|
41
41
|
end
|
|
42
42
|
self
|
|
43
43
|
end
|
|
@@ -61,7 +61,7 @@ class ZipKit::Streamer::Heuristic < ZipKit::Streamer::Writable
|
|
|
61
61
|
|
|
62
62
|
# If the deflated version is smaller than the stored one
|
|
63
63
|
# - use deflate, otherwise stored
|
|
64
|
-
ratio = @bytes_deflated / @buf.
|
|
64
|
+
ratio = @bytes_deflated / @buf.bytesize.to_f
|
|
65
65
|
@winner = if ratio <= MINIMUM_VIABLE_COMPRESSION
|
|
66
66
|
@streamer.write_deflated_file(@filename, **@write_file_options)
|
|
67
67
|
else
|
|
@@ -69,9 +69,8 @@ class ZipKit::Streamer::Heuristic < ZipKit::Streamer::Writable
|
|
|
69
69
|
end
|
|
70
70
|
|
|
71
71
|
# Copy the buffered uncompressed data into the newly initialized writable
|
|
72
|
-
@buf
|
|
73
|
-
|
|
74
|
-
@buf.truncate(0)
|
|
72
|
+
@winner << @buf
|
|
73
|
+
@buf.clear
|
|
75
74
|
ensure
|
|
76
75
|
@deflater.close
|
|
77
76
|
end
|
data/lib/zip_kit/streamer.rb
CHANGED
|
@@ -273,6 +273,11 @@ class ZipKit::Streamer
|
|
|
273
273
|
# output (using `IO.copy_stream` is a good approach).
|
|
274
274
|
# @return [ZipKit::Streamer::Writable] without a block - the Writable sink which has to be closed manually
|
|
275
275
|
def write_file(filename, modification_time: Time.now.utc, unix_permissions: nil, &blk)
|
|
276
|
+
# Reset rollback state when starting a new entry attempt, so that if this entry
|
|
277
|
+
# fails before writing a header, rollback! won't use stale values from a previous entry
|
|
278
|
+
@offset_before_last_local_file_header = nil
|
|
279
|
+
@remove_last_file_at_rollback = false
|
|
280
|
+
|
|
276
281
|
writable = ZipKit::Streamer::Heuristic.new(self, filename, modification_time: modification_time, unix_permissions: unix_permissions)
|
|
277
282
|
yield_or_return_writable(writable, &blk)
|
|
278
283
|
end
|
|
@@ -510,8 +515,13 @@ class ZipKit::Streamer
|
|
|
510
515
|
end
|
|
511
516
|
|
|
512
517
|
# Create filler for the truncated or unusable local file entry that did get written into the output
|
|
513
|
-
|
|
514
|
-
@
|
|
518
|
+
# Only create a filler if a local file header was actually written (indicated by
|
|
519
|
+
# @offset_before_last_local_file_header being set). If it's nil, no header was written,
|
|
520
|
+
# so there's nothing to create a filler for.
|
|
521
|
+
if @offset_before_last_local_file_header
|
|
522
|
+
filler_size_bytes = @out.tell - @offset_before_last_local_file_header
|
|
523
|
+
@files << Filler.new(filler_size_bytes)
|
|
524
|
+
end
|
|
515
525
|
|
|
516
526
|
@out.tell
|
|
517
527
|
end
|
data/lib/zip_kit/version.rb
CHANGED
data/lib/zip_kit/write_buffer.rb
CHANGED
|
@@ -12,10 +12,23 @@
|
|
|
12
12
|
# lots of very small writes, and some degree of speedup (about 20%) can be achieved
|
|
13
13
|
# with a buffer of a few KB.
|
|
14
14
|
#
|
|
15
|
-
#
|
|
16
|
-
#
|
|
17
|
-
#
|
|
18
|
-
#
|
|
15
|
+
# The WriteBuffer is also useful in front of a `write_file` / `write_deflated_file` writable
|
|
16
|
+
# if you are going to be appending lots of tiny strings (like XML fragments) to it. Every write
|
|
17
|
+
# into a writable goes through Zlib separately, so coalescing those writes into bigger chunks
|
|
18
|
+
# is much faster.
|
|
19
|
+
#
|
|
20
|
+
# All strings appended to the WriteBuffer are appended as bytes, and the buffer String
|
|
21
|
+
# given to the writable is always in binary encoding (`Encoding::BINARY`). You can therefore mix
|
|
22
|
+
# binary strings and strings in other encodings (for instance UTF-8 with non-ASCII characters)
|
|
23
|
+
# without getting an `Encoding::CompatibilityError`. No intermediate copies of the strings
|
|
24
|
+
# you append (like `String#b` would create) are made.
|
|
25
|
+
#
|
|
26
|
+
# Note that there is no guarantee that the write buffer is going to flush at exactly
|
|
27
|
+
# the given `buffer_size`. The buffer gets flushed when the next write would make it exceed
|
|
28
|
+
# `buffer_size`, so the chunks it outputs are usually a bit smaller than that (strings with
|
|
29
|
+
# multibyte characters can make it go slightly over). For writes of `buffer_size` or larger
|
|
30
|
+
# it will first `flush` and then write through the oversized chunk, without buffering it.
|
|
31
|
+
# This helps conserve memory. Also note that the buffer will *not* duplicate strings for you
|
|
19
32
|
# and *will* yield the same buffer String over and over, so if you are storing it in an
|
|
20
33
|
# Array you might need to duplicate it.
|
|
21
34
|
#
|
|
@@ -25,16 +38,19 @@
|
|
|
25
38
|
# to `<<`. Therefore, if you need to retain the output of the WriteBuffer in, say, an Array,
|
|
26
39
|
# you might need to `.dup` the `String` it gives you.
|
|
27
40
|
class ZipKit::WriteBuffer
|
|
41
|
+
# String#append_as_bytes (Ruby 3.4+) appends the bytes of the string without any encoding
|
|
42
|
+
# negotiation, so the buffer always stays binary. Without it we use String#<<, see `append_bytes`.
|
|
43
|
+
APPEND_AS_BYTES = String.instance_methods.include?(:append_as_bytes)
|
|
44
|
+
|
|
28
45
|
# Creates a new WriteBuffer bypassing into a given writable object
|
|
29
46
|
#
|
|
30
47
|
# @param writable[#<<] An object that responds to `#<<` with a String as argument
|
|
31
48
|
# @param buffer_size[Integer] How many bytes to buffer
|
|
32
49
|
def initialize(writable, buffer_size)
|
|
33
|
-
#
|
|
34
|
-
#
|
|
35
|
-
#
|
|
36
|
-
|
|
37
|
-
@buf = ("\0".b * (buffer_size * 2)).clear
|
|
50
|
+
# No capacity gets preallocated. String#clear releases the memory held by the String,
|
|
51
|
+
# so after the first flush the buffer would have to grow again anyway - and many
|
|
52
|
+
# WriteBuffers (like the ones used for the CRC32 of small ZIP entries) never fill up.
|
|
53
|
+
@buf = "".b
|
|
38
54
|
@buffer_size = buffer_size
|
|
39
55
|
@writable = writable
|
|
40
56
|
end
|
|
@@ -45,11 +61,15 @@ class ZipKit::WriteBuffer
|
|
|
45
61
|
# @param string[String] data to be written
|
|
46
62
|
# @return self
|
|
47
63
|
def <<(string)
|
|
48
|
-
if string.bytesize
|
|
49
|
-
|
|
64
|
+
if @buf.bytesize + string.bytesize < @buffer_size
|
|
65
|
+
APPEND_AS_BYTES ? @buf.append_as_bytes(string) : append_bytes(string)
|
|
66
|
+
elsif string.bytesize >= @buffer_size
|
|
67
|
+
flush
|
|
68
|
+
# String#b does not copy the bytes of a large String, the new String shares them
|
|
50
69
|
@writable << string.b
|
|
51
70
|
else
|
|
52
|
-
@buf
|
|
71
|
+
flush if @buf.bytesize + string.bytesize > @buffer_size
|
|
72
|
+
append_bytes(string)
|
|
53
73
|
flush if @buf.bytesize >= @buffer_size
|
|
54
74
|
end
|
|
55
75
|
self
|
|
@@ -60,7 +80,8 @@ class ZipKit::WriteBuffer
|
|
|
60
80
|
# @return self
|
|
61
81
|
def flush
|
|
62
82
|
unless @buf.empty?
|
|
63
|
-
|
|
83
|
+
# force_encoding does not copy the String, it only changes its encoding
|
|
84
|
+
@writable << @buf.force_encoding(Encoding::BINARY)
|
|
64
85
|
@buf.clear
|
|
65
86
|
end
|
|
66
87
|
self
|
|
@@ -68,4 +89,19 @@ class ZipKit::WriteBuffer
|
|
|
68
89
|
|
|
69
90
|
# `flush!` was renamed to `flush` but we preserve this method for backwards compatibility
|
|
70
91
|
alias_method :flush!, :flush
|
|
92
|
+
|
|
93
|
+
private
|
|
94
|
+
|
|
95
|
+
# Appends the bytes of the string without copying it. Without String#append_as_bytes (Ruby < 3.4)
|
|
96
|
+
# String#<< is used, which may change the encoding of the buffer or raise if the encodings are
|
|
97
|
+
# incompatible - in that case we append the bytes of the string instead. The buffer is forced
|
|
98
|
+
# back into binary before it is handed to the writable, see `flush`.
|
|
99
|
+
def append_bytes(string)
|
|
100
|
+
return @buf.append_as_bytes(string) if APPEND_AS_BYTES
|
|
101
|
+
|
|
102
|
+
@buf << string
|
|
103
|
+
rescue Encoding::CompatibilityError
|
|
104
|
+
@buf.force_encoding(Encoding::BINARY)
|
|
105
|
+
@buf << string.b
|
|
106
|
+
end
|
|
71
107
|
end
|
data/lib/zip_kit/zip_writer.rb
CHANGED
|
@@ -1,5 +1,7 @@
|
|
|
1
1
|
# frozen_string_literal: true
|
|
2
2
|
|
|
3
|
+
require "stringio"
|
|
4
|
+
|
|
3
5
|
# A low-level ZIP file data writer. You can use it to write out various headers and central directory elements
|
|
4
6
|
# separately. The class handles the actual encoding of the data according to the ZIP format APPNOTE document.
|
|
5
7
|
#
|
data/rbi/zip_kit.rbi
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
# typed: strong
|
|
2
2
|
module ZipKit
|
|
3
|
-
VERSION = T.let("6.3.
|
|
3
|
+
VERSION = T.let("6.3.5", T.untyped)
|
|
4
4
|
|
|
5
5
|
class Railtie < Rails::Railtie
|
|
6
6
|
end
|
|
@@ -1737,10 +1737,23 @@ end, T.untyped)
|
|
|
1737
1737
|
# lots of very small writes, and some degree of speedup (about 20%) can be achieved
|
|
1738
1738
|
# with a buffer of a few KB.
|
|
1739
1739
|
#
|
|
1740
|
-
#
|
|
1741
|
-
#
|
|
1742
|
-
#
|
|
1743
|
-
#
|
|
1740
|
+
# The WriteBuffer is also useful in front of a `write_file` / `write_deflated_file` writable
|
|
1741
|
+
# if you are going to be appending lots of tiny strings (like XML fragments) to it. Every write
|
|
1742
|
+
# into a writable goes through Zlib separately, so coalescing those writes into bigger chunks
|
|
1743
|
+
# is much faster.
|
|
1744
|
+
#
|
|
1745
|
+
# All strings appended to the WriteBuffer are appended as bytes, and the buffer String
|
|
1746
|
+
# given to the writable is always in binary encoding (`Encoding::BINARY`). You can therefore mix
|
|
1747
|
+
# binary strings and strings in other encodings (for instance UTF-8 with non-ASCII characters)
|
|
1748
|
+
# without getting an `Encoding::CompatibilityError`. No intermediate copies of the strings
|
|
1749
|
+
# you append (like `String#b` would create) are made.
|
|
1750
|
+
#
|
|
1751
|
+
# Note that there is no guarantee that the write buffer is going to flush at exactly
|
|
1752
|
+
# the given `buffer_size`. The buffer gets flushed when the next write would make it exceed
|
|
1753
|
+
# `buffer_size`, so the chunks it outputs are usually a bit smaller than that (strings with
|
|
1754
|
+
# multibyte characters can make it go slightly over). For writes of `buffer_size` or larger
|
|
1755
|
+
# it will first `flush` and then write through the oversized chunk, without buffering it.
|
|
1756
|
+
# This helps conserve memory. Also note that the buffer will *not* duplicate strings for you
|
|
1744
1757
|
# and *will* yield the same buffer String over and over, so if you are storing it in an
|
|
1745
1758
|
# Array you might need to duplicate it.
|
|
1746
1759
|
#
|
|
@@ -1750,6 +1763,8 @@ end, T.untyped)
|
|
|
1750
1763
|
# to `<<`. Therefore, if you need to retain the output of the WriteBuffer in, say, an Array,
|
|
1751
1764
|
# you might need to `.dup` the `String` it gives you.
|
|
1752
1765
|
class WriteBuffer
|
|
1766
|
+
APPEND_AS_BYTES = T.let(String.instance_methods.include?(:append_as_bytes), T.untyped)
|
|
1767
|
+
|
|
1753
1768
|
# sord duck - #<< looks like a duck type, replacing with untyped
|
|
1754
1769
|
# Creates a new WriteBuffer bypassing into a given writable object
|
|
1755
1770
|
#
|
|
@@ -1773,6 +1788,15 @@ end, T.untyped)
|
|
|
1773
1788
|
# _@return_ — self
|
|
1774
1789
|
sig { returns(T.untyped) }
|
|
1775
1790
|
def flush; end
|
|
1791
|
+
|
|
1792
|
+
# sord omit - no YARD type given for "string", using untyped
|
|
1793
|
+
# sord omit - no YARD return type given, using untyped
|
|
1794
|
+
# Appends the bytes of the string without copying it. Without String#append_as_bytes (Ruby < 3.4)
|
|
1795
|
+
# String#<< is used, which may change the encoding of the buffer or raise if the encodings are
|
|
1796
|
+
# incompatible - in that case we append the bytes of the string instead. The buffer is forced
|
|
1797
|
+
# back into binary before it is handed to the writable, see `flush`.
|
|
1798
|
+
sig { params(string: T.untyped).returns(T.untyped) }
|
|
1799
|
+
def append_bytes(string); end
|
|
1776
1800
|
end
|
|
1777
1801
|
|
|
1778
1802
|
# A lot of objects in ZipKit accept bytes that may be sent
|
data/rbi/zip_kit.rbs
CHANGED
|
@@ -1510,10 +1510,23 @@ module ZipKit
|
|
|
1510
1510
|
# lots of very small writes, and some degree of speedup (about 20%) can be achieved
|
|
1511
1511
|
# with a buffer of a few KB.
|
|
1512
1512
|
#
|
|
1513
|
-
#
|
|
1514
|
-
#
|
|
1515
|
-
#
|
|
1516
|
-
#
|
|
1513
|
+
# The WriteBuffer is also useful in front of a `write_file` / `write_deflated_file` writable
|
|
1514
|
+
# if you are going to be appending lots of tiny strings (like XML fragments) to it. Every write
|
|
1515
|
+
# into a writable goes through Zlib separately, so coalescing those writes into bigger chunks
|
|
1516
|
+
# is much faster.
|
|
1517
|
+
#
|
|
1518
|
+
# All strings appended to the WriteBuffer are appended as bytes, and the buffer String
|
|
1519
|
+
# given to the writable is always in binary encoding (`Encoding::BINARY`). You can therefore mix
|
|
1520
|
+
# binary strings and strings in other encodings (for instance UTF-8 with non-ASCII characters)
|
|
1521
|
+
# without getting an `Encoding::CompatibilityError`. No intermediate copies of the strings
|
|
1522
|
+
# you append (like `String#b` would create) are made.
|
|
1523
|
+
#
|
|
1524
|
+
# Note that there is no guarantee that the write buffer is going to flush at exactly
|
|
1525
|
+
# the given `buffer_size`. The buffer gets flushed when the next write would make it exceed
|
|
1526
|
+
# `buffer_size`, so the chunks it outputs are usually a bit smaller than that (strings with
|
|
1527
|
+
# multibyte characters can make it go slightly over). For writes of `buffer_size` or larger
|
|
1528
|
+
# it will first `flush` and then write through the oversized chunk, without buffering it.
|
|
1529
|
+
# This helps conserve memory. Also note that the buffer will *not* duplicate strings for you
|
|
1517
1530
|
# and *will* yield the same buffer String over and over, so if you are storing it in an
|
|
1518
1531
|
# Array you might need to duplicate it.
|
|
1519
1532
|
#
|
|
@@ -1523,6 +1536,8 @@ module ZipKit
|
|
|
1523
1536
|
# to `<<`. Therefore, if you need to retain the output of the WriteBuffer in, say, an Array,
|
|
1524
1537
|
# you might need to `.dup` the `String` it gives you.
|
|
1525
1538
|
class WriteBuffer
|
|
1539
|
+
APPEND_AS_BYTES: untyped
|
|
1540
|
+
|
|
1526
1541
|
# sord duck - #<< looks like a duck type, replacing with untyped
|
|
1527
1542
|
# Creates a new WriteBuffer bypassing into a given writable object
|
|
1528
1543
|
#
|
|
@@ -1543,6 +1558,14 @@ module ZipKit
|
|
|
1543
1558
|
#
|
|
1544
1559
|
# _@return_ — self
|
|
1545
1560
|
def flush: () -> untyped
|
|
1561
|
+
|
|
1562
|
+
# sord omit - no YARD type given for "string", using untyped
|
|
1563
|
+
# sord omit - no YARD return type given, using untyped
|
|
1564
|
+
# Appends the bytes of the string without copying it. Without String#append_as_bytes (Ruby < 3.4)
|
|
1565
|
+
# String#<< is used, which may change the encoding of the buffer or raise if the encodings are
|
|
1566
|
+
# incompatible - in that case we append the bytes of the string instead. The buffer is forced
|
|
1567
|
+
# back into binary before it is handed to the writable, see `flush`.
|
|
1568
|
+
def append_bytes: (untyped string) -> untyped
|
|
1546
1569
|
end
|
|
1547
1570
|
|
|
1548
1571
|
# A lot of objects in ZipKit accept bytes that may be sent
|
data/zip_kit.gemspec
CHANGED
|
@@ -39,18 +39,22 @@ Gem::Specification.new do |spec|
|
|
|
39
39
|
spec.add_development_dependency "rspec", "~> 3"
|
|
40
40
|
spec.add_development_dependency "rspec-mocks", "~> 3.10", ">= 3.10.2" # ruby 3 compatibility
|
|
41
41
|
spec.add_development_dependency "complexity_assert"
|
|
42
|
-
spec.add_development_dependency "coderay"
|
|
43
42
|
spec.add_development_dependency "benchmark-ips"
|
|
44
43
|
spec.add_development_dependency "allocation_stats", "~> 0.1.5"
|
|
45
44
|
spec.add_development_dependency "yard", "~> 0.9"
|
|
46
45
|
spec.add_development_dependency "standard"
|
|
47
46
|
spec.add_development_dependency "magic_frozen_string_literal"
|
|
48
47
|
spec.add_development_dependency "puma"
|
|
49
|
-
spec.add_development_dependency "mutex_m" # Some deps use it but it is no longer in stdlib since 3.4
|
|
50
|
-
spec.add_development_dependency "bigdecimal" # Some deps use it but it is no longer in stdlib since 3.4
|
|
51
48
|
spec.add_development_dependency "rails", "~> 5" # For testing RailsStreaming against an actual Rails controller
|
|
52
49
|
spec.add_development_dependency "actionpack", "~> 5" # For testing RailsStreaming against an actual Rails controller
|
|
50
|
+
spec.add_development_dependency "sinatra" # We test streaming a ZIP out of an actual Sinatra app too
|
|
53
51
|
spec.add_development_dependency "nokogiri", "~> 1", ">= 1.13" # Rails 5 does by mistake use an older Nokogiri otherwise
|
|
54
|
-
spec.add_development_dependency "sinatra"
|
|
55
52
|
spec.add_development_dependency "sord"
|
|
53
|
+
|
|
54
|
+
# Some deps use stdlib gems which are no longer available in stdlib
|
|
55
|
+
spec.add_development_dependency "mutex_m"
|
|
56
|
+
spec.add_development_dependency "abbrev"
|
|
57
|
+
spec.add_development_dependency "ostruct"
|
|
58
|
+
spec.add_development_dependency "bigdecimal"
|
|
59
|
+
spec.add_development_dependency "benchmark"
|
|
56
60
|
end
|
metadata
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
--- !ruby/object:Gem::Specification
|
|
2
2
|
name: zip_kit
|
|
3
3
|
version: !ruby/object:Gem::Version
|
|
4
|
-
version: 6.3.
|
|
4
|
+
version: 6.3.5
|
|
5
5
|
platform: ruby
|
|
6
6
|
authors:
|
|
7
7
|
- Julik Tarkhanov
|
|
@@ -11,7 +11,7 @@ authors:
|
|
|
11
11
|
- Felix Bünemann
|
|
12
12
|
bindir: exe
|
|
13
13
|
cert_chain: []
|
|
14
|
-
date:
|
|
14
|
+
date: 2026-09-30 00:00:00.000000000 Z
|
|
15
15
|
dependencies:
|
|
16
16
|
- !ruby/object:Gem::Dependency
|
|
17
17
|
name: bundler
|
|
@@ -117,20 +117,6 @@ dependencies:
|
|
|
117
117
|
- - ">="
|
|
118
118
|
- !ruby/object:Gem::Version
|
|
119
119
|
version: '0'
|
|
120
|
-
- !ruby/object:Gem::Dependency
|
|
121
|
-
name: coderay
|
|
122
|
-
requirement: !ruby/object:Gem::Requirement
|
|
123
|
-
requirements:
|
|
124
|
-
- - ">="
|
|
125
|
-
- !ruby/object:Gem::Version
|
|
126
|
-
version: '0'
|
|
127
|
-
type: :development
|
|
128
|
-
prerelease: false
|
|
129
|
-
version_requirements: !ruby/object:Gem::Requirement
|
|
130
|
-
requirements:
|
|
131
|
-
- - ">="
|
|
132
|
-
- !ruby/object:Gem::Version
|
|
133
|
-
version: '0'
|
|
134
120
|
- !ruby/object:Gem::Dependency
|
|
135
121
|
name: benchmark-ips
|
|
136
122
|
requirement: !ruby/object:Gem::Requirement
|
|
@@ -216,21 +202,35 @@ dependencies:
|
|
|
216
202
|
- !ruby/object:Gem::Version
|
|
217
203
|
version: '0'
|
|
218
204
|
- !ruby/object:Gem::Dependency
|
|
219
|
-
name:
|
|
205
|
+
name: rails
|
|
220
206
|
requirement: !ruby/object:Gem::Requirement
|
|
221
207
|
requirements:
|
|
222
|
-
- - "
|
|
208
|
+
- - "~>"
|
|
223
209
|
- !ruby/object:Gem::Version
|
|
224
|
-
version: '
|
|
210
|
+
version: '5'
|
|
225
211
|
type: :development
|
|
226
212
|
prerelease: false
|
|
227
213
|
version_requirements: !ruby/object:Gem::Requirement
|
|
228
214
|
requirements:
|
|
229
|
-
- - "
|
|
215
|
+
- - "~>"
|
|
230
216
|
- !ruby/object:Gem::Version
|
|
231
|
-
version: '
|
|
217
|
+
version: '5'
|
|
232
218
|
- !ruby/object:Gem::Dependency
|
|
233
|
-
name:
|
|
219
|
+
name: actionpack
|
|
220
|
+
requirement: !ruby/object:Gem::Requirement
|
|
221
|
+
requirements:
|
|
222
|
+
- - "~>"
|
|
223
|
+
- !ruby/object:Gem::Version
|
|
224
|
+
version: '5'
|
|
225
|
+
type: :development
|
|
226
|
+
prerelease: false
|
|
227
|
+
version_requirements: !ruby/object:Gem::Requirement
|
|
228
|
+
requirements:
|
|
229
|
+
- - "~>"
|
|
230
|
+
- !ruby/object:Gem::Version
|
|
231
|
+
version: '5'
|
|
232
|
+
- !ruby/object:Gem::Dependency
|
|
233
|
+
name: sinatra
|
|
234
234
|
requirement: !ruby/object:Gem::Requirement
|
|
235
235
|
requirements:
|
|
236
236
|
- - ">="
|
|
@@ -244,55 +244,83 @@ dependencies:
|
|
|
244
244
|
- !ruby/object:Gem::Version
|
|
245
245
|
version: '0'
|
|
246
246
|
- !ruby/object:Gem::Dependency
|
|
247
|
-
name:
|
|
247
|
+
name: nokogiri
|
|
248
248
|
requirement: !ruby/object:Gem::Requirement
|
|
249
249
|
requirements:
|
|
250
250
|
- - "~>"
|
|
251
251
|
- !ruby/object:Gem::Version
|
|
252
|
-
version: '
|
|
252
|
+
version: '1'
|
|
253
|
+
- - ">="
|
|
254
|
+
- !ruby/object:Gem::Version
|
|
255
|
+
version: '1.13'
|
|
253
256
|
type: :development
|
|
254
257
|
prerelease: false
|
|
255
258
|
version_requirements: !ruby/object:Gem::Requirement
|
|
256
259
|
requirements:
|
|
257
260
|
- - "~>"
|
|
258
261
|
- !ruby/object:Gem::Version
|
|
259
|
-
version: '
|
|
262
|
+
version: '1'
|
|
263
|
+
- - ">="
|
|
264
|
+
- !ruby/object:Gem::Version
|
|
265
|
+
version: '1.13'
|
|
260
266
|
- !ruby/object:Gem::Dependency
|
|
261
|
-
name:
|
|
267
|
+
name: sord
|
|
262
268
|
requirement: !ruby/object:Gem::Requirement
|
|
263
269
|
requirements:
|
|
264
|
-
- - "
|
|
270
|
+
- - ">="
|
|
265
271
|
- !ruby/object:Gem::Version
|
|
266
|
-
version: '
|
|
272
|
+
version: '0'
|
|
267
273
|
type: :development
|
|
268
274
|
prerelease: false
|
|
269
275
|
version_requirements: !ruby/object:Gem::Requirement
|
|
270
276
|
requirements:
|
|
271
|
-
- - "
|
|
277
|
+
- - ">="
|
|
272
278
|
- !ruby/object:Gem::Version
|
|
273
|
-
version: '
|
|
279
|
+
version: '0'
|
|
274
280
|
- !ruby/object:Gem::Dependency
|
|
275
|
-
name:
|
|
281
|
+
name: mutex_m
|
|
276
282
|
requirement: !ruby/object:Gem::Requirement
|
|
277
283
|
requirements:
|
|
278
|
-
- - "
|
|
284
|
+
- - ">="
|
|
279
285
|
- !ruby/object:Gem::Version
|
|
280
|
-
version: '
|
|
286
|
+
version: '0'
|
|
287
|
+
type: :development
|
|
288
|
+
prerelease: false
|
|
289
|
+
version_requirements: !ruby/object:Gem::Requirement
|
|
290
|
+
requirements:
|
|
281
291
|
- - ">="
|
|
282
292
|
- !ruby/object:Gem::Version
|
|
283
|
-
version: '
|
|
293
|
+
version: '0'
|
|
294
|
+
- !ruby/object:Gem::Dependency
|
|
295
|
+
name: abbrev
|
|
296
|
+
requirement: !ruby/object:Gem::Requirement
|
|
297
|
+
requirements:
|
|
298
|
+
- - ">="
|
|
299
|
+
- !ruby/object:Gem::Version
|
|
300
|
+
version: '0'
|
|
284
301
|
type: :development
|
|
285
302
|
prerelease: false
|
|
286
303
|
version_requirements: !ruby/object:Gem::Requirement
|
|
287
304
|
requirements:
|
|
288
|
-
- - "
|
|
305
|
+
- - ">="
|
|
289
306
|
- !ruby/object:Gem::Version
|
|
290
|
-
version: '
|
|
307
|
+
version: '0'
|
|
308
|
+
- !ruby/object:Gem::Dependency
|
|
309
|
+
name: ostruct
|
|
310
|
+
requirement: !ruby/object:Gem::Requirement
|
|
311
|
+
requirements:
|
|
291
312
|
- - ">="
|
|
292
313
|
- !ruby/object:Gem::Version
|
|
293
|
-
version: '
|
|
314
|
+
version: '0'
|
|
315
|
+
type: :development
|
|
316
|
+
prerelease: false
|
|
317
|
+
version_requirements: !ruby/object:Gem::Requirement
|
|
318
|
+
requirements:
|
|
319
|
+
- - ">="
|
|
320
|
+
- !ruby/object:Gem::Version
|
|
321
|
+
version: '0'
|
|
294
322
|
- !ruby/object:Gem::Dependency
|
|
295
|
-
name:
|
|
323
|
+
name: bigdecimal
|
|
296
324
|
requirement: !ruby/object:Gem::Requirement
|
|
297
325
|
requirements:
|
|
298
326
|
- - ">="
|
|
@@ -306,7 +334,7 @@ dependencies:
|
|
|
306
334
|
- !ruby/object:Gem::Version
|
|
307
335
|
version: '0'
|
|
308
336
|
- !ruby/object:Gem::Dependency
|
|
309
|
-
name:
|
|
337
|
+
name: benchmark
|
|
310
338
|
requirement: !ruby/object:Gem::Requirement
|
|
311
339
|
requirements:
|
|
312
340
|
- - ">="
|
|
@@ -339,6 +367,7 @@ files:
|
|
|
339
367
|
- README.md
|
|
340
368
|
- RUBYZIP_DIFFERENCES.md
|
|
341
369
|
- Rakefile
|
|
370
|
+
- bench/write_buffer_bench.rb
|
|
342
371
|
- examples/archive_size_estimate.rb
|
|
343
372
|
- examples/config.ru
|
|
344
373
|
- examples/deferred_write.rb
|