zip_kit 6.3.3 → 6.3.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
checksums.yaml CHANGED
@@ -1,7 +1,7 @@
1
1
  ---
2
2
  SHA256:
3
- metadata.gz: e037e0584d0dbee2e4ab4d934567d04564a26e4f30348f538bda6f7697f8d700
4
- data.tar.gz: e069300aae55e54c3d88ed1a6c1e23803bdffbc92a7764de3f58a38a892f2b82
3
+ metadata.gz: 83cdf922faca6692b384bc30ba8290f0f15a4567ebce61fccb792a5b5d7993fc
4
+ data.tar.gz: ba3a6fee022ba663d77306850246ac8b33ba5d1e10d81f636ace18cc882dc99f
5
5
  SHA512:
6
- metadata.gz: d35e87fc08da2212b9238e4f39df44091454ff77ef41e9810785aaa6855b46c017b285672b1fd0c0bb36df3255a24219331185ce1fc5f7dcf49584c46cce92f0
7
- data.tar.gz: b50c429db7b8bbfe7b62d6b14be6cd0827cf1937f6065078b28a6d19b5e016e90e3ac2af0a3567b645ce14f44e4767ac42a5f5c20bb0e0d65ba207b310c02fc0
6
+ metadata.gz: ad24aba7c7359f01a11e94ddd1e469fdac2a77c59ece3a113cacebaf900e1cec9a9eee462b385c05d9f1789f9f5c6b48da5098b060e2c5b2c99cc5e312c4d435
7
+ data.tar.gz: 96eb1200ee5bdf95e2c9f6b8f91d856607f7981f0eabf7c5cee75dbb3efa200c7642f0886ffdad1965da123281c9cfe8d545614616702cadf4dc207da83a293d
data/CHANGELOG.md CHANGED
@@ -1,3 +1,17 @@
1
+ ## Unreleased
2
+
3
+ ## 6.3.5
4
+
5
+ * Make `WriteBuffer#<<` much faster for lots of tiny writes, like the XML fragments written by libraries such as caxlsx. Strings are no longer copied with `String#b` before being appended: on Ruby 3.4+ `String#append_as_bytes` is used, on older Rubies the string is converted only if Ruby refuses to append it. Mixing binary strings and non-ASCII strings still works, and the writable still always receives binary strings.
6
+ * `WriteBuffer` now flushes before a write would make the buffer exceed its size (instead of after), so that it never outputs chunks larger than the buffer size, and writes which are larger than the buffer size are still passed through without getting copied.
7
+ * `WriteBuffer` no longer preallocates (and immediately discards) a zero-filled String of twice its buffer size, which makes writing ZIPs with lots of small files faster.
8
+ * Require `stringio` in `ZipWriter`, which uses it. Previously writing a ZIP could fail with a `NameError` if nothing else in the process had loaded `stringio`.
9
+ * Add `bench/write_buffer_bench.rb`
10
+
11
+ ## 6.3.4
12
+
13
+ * Fix a bug whereby `rollback!` would cause an exception without any entries having been written yet (rollback on first entry).
14
+
1
15
  ## 6.3.3
2
16
 
3
17
  * Make sure `Writable#<<` converts the strings it is given into binary if they are not already in binary. This fixes an issue where `Heuristic` would suddenly start forwarding strings as-is to downstream callees. There is a lot of spots where the string-to-write gets forwarded and converting in every single one will be quite wasteful, but it can be handy to do in a few key places.
@@ -0,0 +1,183 @@
1
+ # Benchmarks ZipKit::WriteBuffer with lots of tiny writes (like the XML fragments
2
+ # a library such as caxlsx produces) and with large writes, comparing it to the
3
+ # previous implementations of WriteBuffer.
4
+ #
5
+ # bundle exec ruby bench/write_buffer_bench.rb
6
+ require "bundler"
7
+ Bundler.setup
8
+
9
+ require "benchmark"
10
+ require "benchmark/ips"
11
+ require_relative "../lib/zip_kit"
12
+
13
+ # Initialization and flushing as in zip_kit 6.3.x. These are not subclasses of ZipKit::WriteBuffer
14
+ # on purpose: with Ruby 3.4, instances of a subclass can have a different object shape, which
15
+ # can make the instance variable caches of methods they share with ZipKit::WriteBuffer miss.
16
+ module WriteBuffer63x
17
+ def initialize(writable, buffer_size)
18
+ @buf = ("\0".b * (buffer_size * 2)).clear
19
+ @buffer_size = buffer_size
20
+ @writable = writable
21
+ end
22
+
23
+ def flush
24
+ unless @buf.empty?
25
+ @writable << @buf
26
+ @buf.clear
27
+ end
28
+ self
29
+ end
30
+ end
31
+
32
+ # WriteBuffer#<< as of zip_kit 6.3.2
33
+ class WriteBuffer632
34
+ include WriteBuffer63x
35
+
36
+ def <<(data)
37
+ if data.bytesize >= @buffer_size
38
+ flush unless @buf.empty?
39
+ @writable << data
40
+ else
41
+ @buf << data
42
+ flush if @buf.bytesize >= @buffer_size
43
+ end
44
+ self
45
+ end
46
+ end
47
+
48
+ # WriteBuffer#<< as of zip_kit 6.3.3/6.3.4, which copies every string with String#b
49
+ class WriteBuffer634
50
+ include WriteBuffer63x
51
+
52
+ def <<(string)
53
+ if string.bytesize >= @buffer_size
54
+ flush
55
+ @writable << string.b
56
+ else
57
+ @buf << string.b
58
+ flush if @buf.bytesize >= @buffer_size
59
+ end
60
+ self
61
+ end
62
+ end
63
+
64
+ # The simplest possible buffer, which does not deal with encodings or large writes
65
+ class NaiveBuffer
66
+ def initialize(io, buffer_size)
67
+ @io = io
68
+ @buffer_size = buffer_size
69
+ @buf = "".b
70
+ end
71
+
72
+ def <<(fragment)
73
+ @buf << fragment
74
+ flush if @buf.bytesize >= @buffer_size
75
+ self
76
+ end
77
+
78
+ def flush
79
+ return if @buf.empty?
80
+ @io << @buf
81
+ @buf.clear
82
+ end
83
+ end
84
+
85
+ BUFFER_SIZE = 64 * 1024
86
+ IMPLEMENTATIONS = {
87
+ "WriteBuffer" => ZipKit::WriteBuffer,
88
+ "WriteBuffer (6.3.2)" => WriteBuffer632,
89
+ "WriteBuffer (6.3.4, String#b)" => WriteBuffer634,
90
+ "Naive String buffer" => NaiveBuffer
91
+ }
92
+
93
+ # Fragments like the ones produced by caxlsx when writing out a worksheet: mostly frozen
94
+ # literals and short dynamic strings (UTF-8 or US-ASCII), mostly ASCII-only.
95
+ fragments = []
96
+ 20_000.times do |row|
97
+ fragments << "<row r=\"" << (row + 1).to_s << "\">"
98
+ 6.times do |col|
99
+ fragments << "<c r=\"" << "#{("A".ord + col).chr}#{row + 1}" << "\" s=\"" << "1" << "\""
100
+ fragments << ((col == 3) ? " t=\"inlineStr\"><is><t>Grüße, #{row}</t></is>" : "><v>#{row * col}</v>")
101
+ fragments << "</c>"
102
+ end
103
+ fragments << "</row>"
104
+ end
105
+ n_bytes = fragments.sum(&:bytesize)
106
+
107
+ puts RUBY_DESCRIPTION
108
+ puts "#{fragments.length} tiny writes (#{n_bytes} bytes, #{(n_bytes.to_f / fragments.length).round(1)} bytes per write on average)"
109
+
110
+ Benchmark.ips do |x|
111
+ x.config(time: 5, warmup: 2)
112
+ IMPLEMENTATIONS.each do |name, buffer_class|
113
+ x.report("#{name}, tiny writes into CRC32") do
114
+ buf = buffer_class.new(ZipKit::StreamCRC32.new, BUFFER_SIZE)
115
+ fragments.each { |fragment| buf << fragment }
116
+ buf.flush
117
+ end
118
+ end
119
+ x.compare!
120
+ end
121
+
122
+ Benchmark.ips do |x|
123
+ x.config(time: 5, warmup: 2)
124
+ IMPLEMENTATIONS.each do |name, buffer_class|
125
+ x.report("#{name}, tiny writes into write_deflated_file") do
126
+ ZipKit::Streamer.open(ZipKit::NullWriter) do |zip|
127
+ zip.write_deflated_file("sheet.xml") do |sink|
128
+ buf = buffer_class.new(sink, BUFFER_SIZE)
129
+ fragments.each { |fragment| buf << fragment }
130
+ buf.flush
131
+ end
132
+ end
133
+ end
134
+ end
135
+ x.report("No buffer, tiny writes into write_deflated_file") do
136
+ ZipKit::Streamer.open(ZipKit::NullWriter) do |zip|
137
+ zip.write_deflated_file("sheet.xml") do |sink|
138
+ fragments.each { |fragment| sink << fragment }
139
+ end
140
+ end
141
+ end
142
+ x.compare!
143
+ end
144
+
145
+ large_chunk = Random.new(42).bytes(1024 * 1024)
146
+ Benchmark.ips do |x|
147
+ x.config(time: 5, warmup: 2)
148
+ IMPLEMENTATIONS.each do |name, buffer_class|
149
+ x.report("#{name}, 64 writes of 1MB into CRC32") do
150
+ buf = buffer_class.new(ZipKit::StreamCRC32.new, BUFFER_SIZE)
151
+ 64.times { buf << large_chunk }
152
+ buf.flush
153
+ end
154
+ end
155
+ x.compare!
156
+ end
157
+
158
+ __END__
159
+
160
+ Apple M1 Pro, macOS 15.7
161
+
162
+ ruby 3.4.1 (2024-12-25 revision 48d4efcb85) +PRISM [arm64-darwin24], without YJIT
163
+ 920000 tiny writes (5189483 bytes, 5.6 bytes per write on average)
164
+
165
+ Comparison:
166
+ Naive String buffer, tiny writes into CRC32: 12.1 i/s
167
+ WriteBuffer, tiny writes into CRC32: 12.0 i/s - 1.01x slower
168
+ WriteBuffer (6.3.2), tiny writes into CRC32: 10.3 i/s - 1.17x slower
169
+ WriteBuffer (6.3.4, String#b), tiny writes into CRC32: 8.0 i/s - 1.51x slower
170
+
171
+ Comparison:
172
+ Naive String buffer, tiny writes into write_deflated_file: 7.6 i/s
173
+ WriteBuffer, tiny writes into write_deflated_file: 7.4 i/s - 1.02x slower
174
+ WriteBuffer (6.3.2), tiny writes into write_deflated_file: 6.8 i/s - 1.12x slower
175
+ WriteBuffer (6.3.4, String#b), tiny writes into write_deflated_file: 5.7 i/s - 1.33x slower
176
+ No buffer, tiny writes into write_deflated_file: 1.8 i/s - 4.18x slower
177
+
178
+ Comparison:
179
+ WriteBuffer (6.3.2), 64 writes of 1MB into CRC32: 464.4 i/s
180
+ WriteBuffer (6.3.4, String#b), 64 writes of 1MB into CRC32: 463.0 i/s - same-ish: difference falls within error
181
+ WriteBuffer, 64 writes of 1MB into CRC32: 462.9 i/s - same-ish: difference falls within error
182
+ Naive String buffer, 64 writes of 1MB into CRC32: 121.5 i/s - 3.82x slower
183
+
@@ -23,7 +23,7 @@ class ZipKit::Streamer::Heuristic < ZipKit::Streamer::Writable
23
23
  @filename = filename
24
24
  @write_file_options = write_file_options
25
25
 
26
- @buf = StringIO.new.binmode
26
+ @buf = +"".b # Just use a mutable String
27
27
  @deflater = ::Zlib::Deflate.new(Zlib::DEFAULT_COMPRESSION, -::Zlib::MAX_WBITS)
28
28
  @bytes_deflated = 0
29
29
 
@@ -37,7 +37,7 @@ class ZipKit::Streamer::Heuristic < ZipKit::Streamer::Writable
37
37
  else
38
38
  @buf << bytes
39
39
  @deflater.deflate(bytes) { |chunk| @bytes_deflated += chunk.bytesize }
40
- decide if @buf.size > BYTES_WRITTEN_THRESHOLD
40
+ decide if @buf.bytesize > BYTES_WRITTEN_THRESHOLD
41
41
  end
42
42
  self
43
43
  end
@@ -61,7 +61,7 @@ class ZipKit::Streamer::Heuristic < ZipKit::Streamer::Writable
61
61
 
62
62
  # If the deflated version is smaller than the stored one
63
63
  # - use deflate, otherwise stored
64
- ratio = @bytes_deflated / @buf.size.to_f
64
+ ratio = @bytes_deflated / @buf.bytesize.to_f
65
65
  @winner = if ratio <= MINIMUM_VIABLE_COMPRESSION
66
66
  @streamer.write_deflated_file(@filename, **@write_file_options)
67
67
  else
@@ -69,9 +69,8 @@ class ZipKit::Streamer::Heuristic < ZipKit::Streamer::Writable
69
69
  end
70
70
 
71
71
  # Copy the buffered uncompressed data into the newly initialized writable
72
- @buf.rewind
73
- IO.copy_stream(@buf, @winner)
74
- @buf.truncate(0)
72
+ @winner << @buf
73
+ @buf.clear
75
74
  ensure
76
75
  @deflater.close
77
76
  end
@@ -273,6 +273,11 @@ class ZipKit::Streamer
273
273
  # output (using `IO.copy_stream` is a good approach).
274
274
  # @return [ZipKit::Streamer::Writable] without a block - the Writable sink which has to be closed manually
275
275
  def write_file(filename, modification_time: Time.now.utc, unix_permissions: nil, &blk)
276
+ # Reset rollback state when starting a new entry attempt, so that if this entry
277
+ # fails before writing a header, rollback! won't use stale values from a previous entry
278
+ @offset_before_last_local_file_header = nil
279
+ @remove_last_file_at_rollback = false
280
+
276
281
  writable = ZipKit::Streamer::Heuristic.new(self, filename, modification_time: modification_time, unix_permissions: unix_permissions)
277
282
  yield_or_return_writable(writable, &blk)
278
283
  end
@@ -510,8 +515,13 @@ class ZipKit::Streamer
510
515
  end
511
516
 
512
517
  # Create filler for the truncated or unusable local file entry that did get written into the output
513
- filler_size_bytes = @out.tell - @offset_before_last_local_file_header
514
- @files << Filler.new(filler_size_bytes)
518
+ # Only create a filler if a local file header was actually written (indicated by
519
+ # @offset_before_last_local_file_header being set). If it's nil, no header was written,
520
+ # so there's nothing to create a filler for.
521
+ if @offset_before_last_local_file_header
522
+ filler_size_bytes = @out.tell - @offset_before_last_local_file_header
523
+ @files << Filler.new(filler_size_bytes)
524
+ end
515
525
 
516
526
  @out.tell
517
527
  end
@@ -1,5 +1,5 @@
1
1
  # frozen_string_literal: true
2
2
 
3
3
  module ZipKit
4
- VERSION = "6.3.3"
4
+ VERSION = "6.3.5"
5
5
  end
@@ -12,10 +12,23 @@
12
12
  # lots of very small writes, and some degree of speedup (about 20%) can be achieved
13
13
  # with a buffer of a few KB.
14
14
  #
15
- # Note that there is no guarantee that the write buffer is going to flush at or above
16
- # the given `buffer_size`, because for writes which exceed the buffer size it will
17
- # first `flush` and then write through the oversized chunk, without buffering it. This
18
- # helps conserve memory. Also note that the buffer will *not* duplicate strings for you
15
+ # The WriteBuffer is also useful in front of a `write_file` / `write_deflated_file` writable
16
+ # if you are going to be appending lots of tiny strings (like XML fragments) to it. Every write
17
+ # into a writable goes through Zlib separately, so coalescing those writes into bigger chunks
18
+ # is much faster.
19
+ #
20
+ # All strings appended to the WriteBuffer are appended as bytes, and the buffer String
21
+ # given to the writable is always in binary encoding (`Encoding::BINARY`). You can therefore mix
22
+ # binary strings and strings in other encodings (for instance UTF-8 with non-ASCII characters)
23
+ # without getting an `Encoding::CompatibilityError`. No intermediate copies of the strings
24
+ # you append (like `String#b` would create) are made.
25
+ #
26
+ # Note that there is no guarantee that the write buffer is going to flush at exactly
27
+ # the given `buffer_size`. The buffer gets flushed when the next write would make it exceed
28
+ # `buffer_size`, so the chunks it outputs are usually a bit smaller than that (strings with
29
+ # multibyte characters can make it go slightly over). For writes of `buffer_size` or larger
30
+ # it will first `flush` and then write through the oversized chunk, without buffering it.
31
+ # This helps conserve memory. Also note that the buffer will *not* duplicate strings for you
19
32
  # and *will* yield the same buffer String over and over, so if you are storing it in an
20
33
  # Array you might need to duplicate it.
21
34
  #
@@ -25,16 +38,19 @@
25
38
  # to `<<`. Therefore, if you need to retain the output of the WriteBuffer in, say, an Array,
26
39
  # you might need to `.dup` the `String` it gives you.
27
40
  class ZipKit::WriteBuffer
41
+ # String#append_as_bytes (Ruby 3.4+) appends the bytes of the string without any encoding
42
+ # negotiation, so the buffer always stays binary. Without it we use String#<<, see `append_bytes`.
43
+ APPEND_AS_BYTES = String.instance_methods.include?(:append_as_bytes)
44
+
28
45
  # Creates a new WriteBuffer bypassing into a given writable object
29
46
  #
30
47
  # @param writable[#<<] An object that responds to `#<<` with a String as argument
31
48
  # @param buffer_size[Integer] How many bytes to buffer
32
49
  def initialize(writable, buffer_size)
33
- # Allocating the buffer using a zero-padded String as a variation
34
- # on using capacity:, which JRuby apparently does not like very much. The
35
- # desire here is that the buffer doesn't have to be resized during the lifetime
36
- # of the object.
37
- @buf = ("\0".b * (buffer_size * 2)).clear
50
+ # No capacity gets preallocated. String#clear releases the memory held by the String,
51
+ # so after the first flush the buffer would have to grow again anyway - and many
52
+ # WriteBuffers (like the ones used for the CRC32 of small ZIP entries) never fill up.
53
+ @buf = "".b
38
54
  @buffer_size = buffer_size
39
55
  @writable = writable
40
56
  end
@@ -45,11 +61,15 @@ class ZipKit::WriteBuffer
45
61
  # @param string[String] data to be written
46
62
  # @return self
47
63
  def <<(string)
48
- if string.bytesize >= @buffer_size
49
- flush # <- this is were we can output less than @buffer_size
64
+ if @buf.bytesize + string.bytesize < @buffer_size
65
+ APPEND_AS_BYTES ? @buf.append_as_bytes(string) : append_bytes(string)
66
+ elsif string.bytesize >= @buffer_size
67
+ flush
68
+ # String#b does not copy the bytes of a large String, the new String shares them
50
69
  @writable << string.b
51
70
  else
52
- @buf << string.b
71
+ flush if @buf.bytesize + string.bytesize > @buffer_size
72
+ append_bytes(string)
53
73
  flush if @buf.bytesize >= @buffer_size
54
74
  end
55
75
  self
@@ -60,7 +80,8 @@ class ZipKit::WriteBuffer
60
80
  # @return self
61
81
  def flush
62
82
  unless @buf.empty?
63
- @writable << @buf
83
+ # force_encoding does not copy the String, it only changes its encoding
84
+ @writable << @buf.force_encoding(Encoding::BINARY)
64
85
  @buf.clear
65
86
  end
66
87
  self
@@ -68,4 +89,19 @@ class ZipKit::WriteBuffer
68
89
 
69
90
  # `flush!` was renamed to `flush` but we preserve this method for backwards compatibility
70
91
  alias_method :flush!, :flush
92
+
93
+ private
94
+
95
+ # Appends the bytes of the string without copying it. Without String#append_as_bytes (Ruby < 3.4)
96
+ # String#<< is used, which may change the encoding of the buffer or raise if the encodings are
97
+ # incompatible - in that case we append the bytes of the string instead. The buffer is forced
98
+ # back into binary before it is handed to the writable, see `flush`.
99
+ def append_bytes(string)
100
+ return @buf.append_as_bytes(string) if APPEND_AS_BYTES
101
+
102
+ @buf << string
103
+ rescue Encoding::CompatibilityError
104
+ @buf.force_encoding(Encoding::BINARY)
105
+ @buf << string.b
106
+ end
71
107
  end
@@ -1,5 +1,7 @@
1
1
  # frozen_string_literal: true
2
2
 
3
+ require "stringio"
4
+
3
5
  # A low-level ZIP file data writer. You can use it to write out various headers and central directory elements
4
6
  # separately. The class handles the actual encoding of the data according to the ZIP format APPNOTE document.
5
7
  #
data/rbi/zip_kit.rbi CHANGED
@@ -1,6 +1,6 @@
1
1
  # typed: strong
2
2
  module ZipKit
3
- VERSION = T.let("6.3.3", T.untyped)
3
+ VERSION = T.let("6.3.5", T.untyped)
4
4
 
5
5
  class Railtie < Rails::Railtie
6
6
  end
@@ -1737,10 +1737,23 @@ end, T.untyped)
1737
1737
  # lots of very small writes, and some degree of speedup (about 20%) can be achieved
1738
1738
  # with a buffer of a few KB.
1739
1739
  #
1740
- # Note that there is no guarantee that the write buffer is going to flush at or above
1741
- # the given `buffer_size`, because for writes which exceed the buffer size it will
1742
- # first `flush` and then write through the oversized chunk, without buffering it. This
1743
- # helps conserve memory. Also note that the buffer will *not* duplicate strings for you
1740
+ # The WriteBuffer is also useful in front of a `write_file` / `write_deflated_file` writable
1741
+ # if you are going to be appending lots of tiny strings (like XML fragments) to it. Every write
1742
+ # into a writable goes through Zlib separately, so coalescing those writes into bigger chunks
1743
+ # is much faster.
1744
+ #
1745
+ # All strings appended to the WriteBuffer are appended as bytes, and the buffer String
1746
+ # given to the writable is always in binary encoding (`Encoding::BINARY`). You can therefore mix
1747
+ # binary strings and strings in other encodings (for instance UTF-8 with non-ASCII characters)
1748
+ # without getting an `Encoding::CompatibilityError`. No intermediate copies of the strings
1749
+ # you append (like `String#b` would create) are made.
1750
+ #
1751
+ # Note that there is no guarantee that the write buffer is going to flush at exactly
1752
+ # the given `buffer_size`. The buffer gets flushed when the next write would make it exceed
1753
+ # `buffer_size`, so the chunks it outputs are usually a bit smaller than that (strings with
1754
+ # multibyte characters can make it go slightly over). For writes of `buffer_size` or larger
1755
+ # it will first `flush` and then write through the oversized chunk, without buffering it.
1756
+ # This helps conserve memory. Also note that the buffer will *not* duplicate strings for you
1744
1757
  # and *will* yield the same buffer String over and over, so if you are storing it in an
1745
1758
  # Array you might need to duplicate it.
1746
1759
  #
@@ -1750,6 +1763,8 @@ end, T.untyped)
1750
1763
  # to `<<`. Therefore, if you need to retain the output of the WriteBuffer in, say, an Array,
1751
1764
  # you might need to `.dup` the `String` it gives you.
1752
1765
  class WriteBuffer
1766
+ APPEND_AS_BYTES = T.let(String.instance_methods.include?(:append_as_bytes), T.untyped)
1767
+
1753
1768
  # sord duck - #<< looks like a duck type, replacing with untyped
1754
1769
  # Creates a new WriteBuffer bypassing into a given writable object
1755
1770
  #
@@ -1773,6 +1788,15 @@ end, T.untyped)
1773
1788
  # _@return_ — self
1774
1789
  sig { returns(T.untyped) }
1775
1790
  def flush; end
1791
+
1792
+ # sord omit - no YARD type given for "string", using untyped
1793
+ # sord omit - no YARD return type given, using untyped
1794
+ # Appends the bytes of the string without copying it. Without String#append_as_bytes (Ruby < 3.4)
1795
+ # String#<< is used, which may change the encoding of the buffer or raise if the encodings are
1796
+ # incompatible - in that case we append the bytes of the string instead. The buffer is forced
1797
+ # back into binary before it is handed to the writable, see `flush`.
1798
+ sig { params(string: T.untyped).returns(T.untyped) }
1799
+ def append_bytes(string); end
1776
1800
  end
1777
1801
 
1778
1802
  # A lot of objects in ZipKit accept bytes that may be sent
data/rbi/zip_kit.rbs CHANGED
@@ -1510,10 +1510,23 @@ module ZipKit
1510
1510
  # lots of very small writes, and some degree of speedup (about 20%) can be achieved
1511
1511
  # with a buffer of a few KB.
1512
1512
  #
1513
- # Note that there is no guarantee that the write buffer is going to flush at or above
1514
- # the given `buffer_size`, because for writes which exceed the buffer size it will
1515
- # first `flush` and then write through the oversized chunk, without buffering it. This
1516
- # helps conserve memory. Also note that the buffer will *not* duplicate strings for you
1513
+ # The WriteBuffer is also useful in front of a `write_file` / `write_deflated_file` writable
1514
+ # if you are going to be appending lots of tiny strings (like XML fragments) to it. Every write
1515
+ # into a writable goes through Zlib separately, so coalescing those writes into bigger chunks
1516
+ # is much faster.
1517
+ #
1518
+ # All strings appended to the WriteBuffer are appended as bytes, and the buffer String
1519
+ # given to the writable is always in binary encoding (`Encoding::BINARY`). You can therefore mix
1520
+ # binary strings and strings in other encodings (for instance UTF-8 with non-ASCII characters)
1521
+ # without getting an `Encoding::CompatibilityError`. No intermediate copies of the strings
1522
+ # you append (like `String#b` would create) are made.
1523
+ #
1524
+ # Note that there is no guarantee that the write buffer is going to flush at exactly
1525
+ # the given `buffer_size`. The buffer gets flushed when the next write would make it exceed
1526
+ # `buffer_size`, so the chunks it outputs are usually a bit smaller than that (strings with
1527
+ # multibyte characters can make it go slightly over). For writes of `buffer_size` or larger
1528
+ # it will first `flush` and then write through the oversized chunk, without buffering it.
1529
+ # This helps conserve memory. Also note that the buffer will *not* duplicate strings for you
1517
1530
  # and *will* yield the same buffer String over and over, so if you are storing it in an
1518
1531
  # Array you might need to duplicate it.
1519
1532
  #
@@ -1523,6 +1536,8 @@ module ZipKit
1523
1536
  # to `<<`. Therefore, if you need to retain the output of the WriteBuffer in, say, an Array,
1524
1537
  # you might need to `.dup` the `String` it gives you.
1525
1538
  class WriteBuffer
1539
+ APPEND_AS_BYTES: untyped
1540
+
1526
1541
  # sord duck - #<< looks like a duck type, replacing with untyped
1527
1542
  # Creates a new WriteBuffer bypassing into a given writable object
1528
1543
  #
@@ -1543,6 +1558,14 @@ module ZipKit
1543
1558
  #
1544
1559
  # _@return_ — self
1545
1560
  def flush: () -> untyped
1561
+
1562
+ # sord omit - no YARD type given for "string", using untyped
1563
+ # sord omit - no YARD return type given, using untyped
1564
+ # Appends the bytes of the string without copying it. Without String#append_as_bytes (Ruby < 3.4)
1565
+ # String#<< is used, which may change the encoding of the buffer or raise if the encodings are
1566
+ # incompatible - in that case we append the bytes of the string instead. The buffer is forced
1567
+ # back into binary before it is handed to the writable, see `flush`.
1568
+ def append_bytes: (untyped string) -> untyped
1546
1569
  end
1547
1570
 
1548
1571
  # A lot of objects in ZipKit accept bytes that may be sent
data/zip_kit.gemspec CHANGED
@@ -39,18 +39,22 @@ Gem::Specification.new do |spec|
39
39
  spec.add_development_dependency "rspec", "~> 3"
40
40
  spec.add_development_dependency "rspec-mocks", "~> 3.10", ">= 3.10.2" # ruby 3 compatibility
41
41
  spec.add_development_dependency "complexity_assert"
42
- spec.add_development_dependency "coderay"
43
42
  spec.add_development_dependency "benchmark-ips"
44
43
  spec.add_development_dependency "allocation_stats", "~> 0.1.5"
45
44
  spec.add_development_dependency "yard", "~> 0.9"
46
45
  spec.add_development_dependency "standard"
47
46
  spec.add_development_dependency "magic_frozen_string_literal"
48
47
  spec.add_development_dependency "puma"
49
- spec.add_development_dependency "mutex_m" # Some deps use it but it is no longer in stdlib since 3.4
50
- spec.add_development_dependency "bigdecimal" # Some deps use it but it is no longer in stdlib since 3.4
51
48
  spec.add_development_dependency "rails", "~> 5" # For testing RailsStreaming against an actual Rails controller
52
49
  spec.add_development_dependency "actionpack", "~> 5" # For testing RailsStreaming against an actual Rails controller
50
+ spec.add_development_dependency "sinatra" # We test streaming a ZIP out of an actual Sinatra app too
53
51
  spec.add_development_dependency "nokogiri", "~> 1", ">= 1.13" # Rails 5 does by mistake use an older Nokogiri otherwise
54
- spec.add_development_dependency "sinatra"
55
52
  spec.add_development_dependency "sord"
53
+
54
+ # Some deps use stdlib gems which are no longer available in stdlib
55
+ spec.add_development_dependency "mutex_m"
56
+ spec.add_development_dependency "abbrev"
57
+ spec.add_development_dependency "ostruct"
58
+ spec.add_development_dependency "bigdecimal"
59
+ spec.add_development_dependency "benchmark"
56
60
  end
metadata CHANGED
@@ -1,7 +1,7 @@
1
1
  --- !ruby/object:Gem::Specification
2
2
  name: zip_kit
3
3
  version: !ruby/object:Gem::Version
4
- version: 6.3.3
4
+ version: 6.3.5
5
5
  platform: ruby
6
6
  authors:
7
7
  - Julik Tarkhanov
@@ -11,7 +11,7 @@ authors:
11
11
  - Felix Bünemann
12
12
  bindir: exe
13
13
  cert_chain: []
14
- date: 2025-04-18 00:00:00.000000000 Z
14
+ date: 2026-09-30 00:00:00.000000000 Z
15
15
  dependencies:
16
16
  - !ruby/object:Gem::Dependency
17
17
  name: bundler
@@ -117,20 +117,6 @@ dependencies:
117
117
  - - ">="
118
118
  - !ruby/object:Gem::Version
119
119
  version: '0'
120
- - !ruby/object:Gem::Dependency
121
- name: coderay
122
- requirement: !ruby/object:Gem::Requirement
123
- requirements:
124
- - - ">="
125
- - !ruby/object:Gem::Version
126
- version: '0'
127
- type: :development
128
- prerelease: false
129
- version_requirements: !ruby/object:Gem::Requirement
130
- requirements:
131
- - - ">="
132
- - !ruby/object:Gem::Version
133
- version: '0'
134
120
  - !ruby/object:Gem::Dependency
135
121
  name: benchmark-ips
136
122
  requirement: !ruby/object:Gem::Requirement
@@ -216,21 +202,35 @@ dependencies:
216
202
  - !ruby/object:Gem::Version
217
203
  version: '0'
218
204
  - !ruby/object:Gem::Dependency
219
- name: mutex_m
205
+ name: rails
220
206
  requirement: !ruby/object:Gem::Requirement
221
207
  requirements:
222
- - - ">="
208
+ - - "~>"
223
209
  - !ruby/object:Gem::Version
224
- version: '0'
210
+ version: '5'
225
211
  type: :development
226
212
  prerelease: false
227
213
  version_requirements: !ruby/object:Gem::Requirement
228
214
  requirements:
229
- - - ">="
215
+ - - "~>"
230
216
  - !ruby/object:Gem::Version
231
- version: '0'
217
+ version: '5'
232
218
  - !ruby/object:Gem::Dependency
233
- name: bigdecimal
219
+ name: actionpack
220
+ requirement: !ruby/object:Gem::Requirement
221
+ requirements:
222
+ - - "~>"
223
+ - !ruby/object:Gem::Version
224
+ version: '5'
225
+ type: :development
226
+ prerelease: false
227
+ version_requirements: !ruby/object:Gem::Requirement
228
+ requirements:
229
+ - - "~>"
230
+ - !ruby/object:Gem::Version
231
+ version: '5'
232
+ - !ruby/object:Gem::Dependency
233
+ name: sinatra
234
234
  requirement: !ruby/object:Gem::Requirement
235
235
  requirements:
236
236
  - - ">="
@@ -244,55 +244,83 @@ dependencies:
244
244
  - !ruby/object:Gem::Version
245
245
  version: '0'
246
246
  - !ruby/object:Gem::Dependency
247
- name: rails
247
+ name: nokogiri
248
248
  requirement: !ruby/object:Gem::Requirement
249
249
  requirements:
250
250
  - - "~>"
251
251
  - !ruby/object:Gem::Version
252
- version: '5'
252
+ version: '1'
253
+ - - ">="
254
+ - !ruby/object:Gem::Version
255
+ version: '1.13'
253
256
  type: :development
254
257
  prerelease: false
255
258
  version_requirements: !ruby/object:Gem::Requirement
256
259
  requirements:
257
260
  - - "~>"
258
261
  - !ruby/object:Gem::Version
259
- version: '5'
262
+ version: '1'
263
+ - - ">="
264
+ - !ruby/object:Gem::Version
265
+ version: '1.13'
260
266
  - !ruby/object:Gem::Dependency
261
- name: actionpack
267
+ name: sord
262
268
  requirement: !ruby/object:Gem::Requirement
263
269
  requirements:
264
- - - "~>"
270
+ - - ">="
265
271
  - !ruby/object:Gem::Version
266
- version: '5'
272
+ version: '0'
267
273
  type: :development
268
274
  prerelease: false
269
275
  version_requirements: !ruby/object:Gem::Requirement
270
276
  requirements:
271
- - - "~>"
277
+ - - ">="
272
278
  - !ruby/object:Gem::Version
273
- version: '5'
279
+ version: '0'
274
280
  - !ruby/object:Gem::Dependency
275
- name: nokogiri
281
+ name: mutex_m
276
282
  requirement: !ruby/object:Gem::Requirement
277
283
  requirements:
278
- - - "~>"
284
+ - - ">="
279
285
  - !ruby/object:Gem::Version
280
- version: '1'
286
+ version: '0'
287
+ type: :development
288
+ prerelease: false
289
+ version_requirements: !ruby/object:Gem::Requirement
290
+ requirements:
281
291
  - - ">="
282
292
  - !ruby/object:Gem::Version
283
- version: '1.13'
293
+ version: '0'
294
+ - !ruby/object:Gem::Dependency
295
+ name: abbrev
296
+ requirement: !ruby/object:Gem::Requirement
297
+ requirements:
298
+ - - ">="
299
+ - !ruby/object:Gem::Version
300
+ version: '0'
284
301
  type: :development
285
302
  prerelease: false
286
303
  version_requirements: !ruby/object:Gem::Requirement
287
304
  requirements:
288
- - - "~>"
305
+ - - ">="
289
306
  - !ruby/object:Gem::Version
290
- version: '1'
307
+ version: '0'
308
+ - !ruby/object:Gem::Dependency
309
+ name: ostruct
310
+ requirement: !ruby/object:Gem::Requirement
311
+ requirements:
291
312
  - - ">="
292
313
  - !ruby/object:Gem::Version
293
- version: '1.13'
314
+ version: '0'
315
+ type: :development
316
+ prerelease: false
317
+ version_requirements: !ruby/object:Gem::Requirement
318
+ requirements:
319
+ - - ">="
320
+ - !ruby/object:Gem::Version
321
+ version: '0'
294
322
  - !ruby/object:Gem::Dependency
295
- name: sinatra
323
+ name: bigdecimal
296
324
  requirement: !ruby/object:Gem::Requirement
297
325
  requirements:
298
326
  - - ">="
@@ -306,7 +334,7 @@ dependencies:
306
334
  - !ruby/object:Gem::Version
307
335
  version: '0'
308
336
  - !ruby/object:Gem::Dependency
309
- name: sord
337
+ name: benchmark
310
338
  requirement: !ruby/object:Gem::Requirement
311
339
  requirements:
312
340
  - - ">="
@@ -339,6 +367,7 @@ files:
339
367
  - README.md
340
368
  - RUBYZIP_DIFFERENCES.md
341
369
  - Rakefile
370
+ - bench/write_buffer_bench.rb
342
371
  - examples/archive_size_estimate.rb
343
372
  - examples/config.ru
344
373
  - examples/deferred_write.rb