zip_kit 6.3.4 → 6.3.6
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/.github/workflows/ci.yml +14 -0
- data/CHANGELOG.md +15 -0
- data/Rakefile +8 -0
- data/bench/write_buffer_bench.rb +183 -0
- data/lib/zip_kit/block_deflate.rb +1 -1
- data/lib/zip_kit/block_write.rb +12 -5
- data/lib/zip_kit/rack_chunked_body.rb +1 -1
- data/lib/zip_kit/streamer.rb +5 -6
- data/lib/zip_kit/version.rb +1 -1
- data/lib/zip_kit/write_buffer.rb +49 -13
- data/lib/zip_kit/zip_writer.rb +4 -2
- data/rbi/zip_kit.rbi +47 -9
- data/rbi/zip_kit.rbs +39 -5
- metadata +3 -2
checksums.yaml
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
---
|
|
2
2
|
SHA256:
|
|
3
|
-
metadata.gz:
|
|
4
|
-
data.tar.gz:
|
|
3
|
+
metadata.gz: 687884a060911fe25101e41be05bd32f4b50068c2ac592acec9937e5b0912f47
|
|
4
|
+
data.tar.gz: c209fb4839008ca84b956839db24ff65d8ec3ae390c4666aabe34743258409ec
|
|
5
5
|
SHA512:
|
|
6
|
-
metadata.gz:
|
|
7
|
-
data.tar.gz:
|
|
6
|
+
metadata.gz: bf6bf944be2ec57fd53950c046f5ed28565c5d97f73edc91fbaf1094319d0c2ad47b0a6c6ae1194e69f7fdab3e85beb479b41755e499497fe7f4ab2423529858
|
|
7
|
+
data.tar.gz: 1d54ad55c06b14297057bc594ff45f74a551b453e9cc5d846595abf238289ed0b7b9175177d089db87bceb9601c9322773a44d4601d3a6b32fb30f331f7bbcff
|
data/.github/workflows/ci.yml
CHANGED
|
@@ -37,6 +37,20 @@ jobs:
|
|
|
37
37
|
RUBYOPT: "--enable=frozen-string-literal --debug=frozen-string-literal"
|
|
38
38
|
run: "bundle exec rspec --backtrace --fail-fast"
|
|
39
39
|
|
|
40
|
+
test_ractors:
|
|
41
|
+
name: "Ractor tests (Ruby 4.0)"
|
|
42
|
+
runs-on: ubuntu-22.04
|
|
43
|
+
steps:
|
|
44
|
+
- name: Checkout
|
|
45
|
+
uses: actions/checkout@v4
|
|
46
|
+
- name: Setup Ruby
|
|
47
|
+
uses: ruby/setup-ruby@v1
|
|
48
|
+
with:
|
|
49
|
+
ruby-version: '4.0'
|
|
50
|
+
bundler-cache: true
|
|
51
|
+
- name: "Tests"
|
|
52
|
+
run: "bundle exec rspec --backtrace --fail-fast spec/zip_kit/ractor_spec.rb"
|
|
53
|
+
|
|
40
54
|
lint_baseline_ruby: # We need to use syntax appropriate for the minimum supported Ruby version
|
|
41
55
|
name: Lint (Ruby 2.6 syntax)
|
|
42
56
|
runs-on: ubuntu-22.04
|
data/CHANGELOG.md
CHANGED
|
@@ -1,3 +1,18 @@
|
|
|
1
|
+
## Unreleased
|
|
2
|
+
|
|
3
|
+
## 6.3.6
|
|
4
|
+
|
|
5
|
+
* Fix the end of central directory record counting entries discarded by `rollback!`, which made archives with a rolled back entry unreadable for readers which check the entry count (including `ZipKit::FileReader`).
|
|
6
|
+
* Make ZipKit usable from within Ractors: freeze the computed String constants in `ZipWriter`, `BlockDeflate` and `RackChunkedBody`, and stop defining the raising `BlockWrite` methods with `define_method`. Ractor tests run on Ruby 4.0 in CI.
|
|
7
|
+
|
|
8
|
+
## 6.3.5
|
|
9
|
+
|
|
10
|
+
* Make `WriteBuffer#<<` much faster for lots of tiny writes, like the XML fragments written by libraries such as caxlsx. Strings are no longer copied with `String#b` before being appended: on Ruby 3.4+ `String#append_as_bytes` is used, on older Rubies the string is converted only if Ruby refuses to append it. Mixing binary strings and non-ASCII strings still works, and the writable still always receives binary strings.
|
|
11
|
+
* `WriteBuffer` now flushes before a write would make the buffer exceed its size (instead of after), so that it never outputs chunks larger than the buffer size, and writes which are larger than the buffer size are still passed through without getting copied.
|
|
12
|
+
* `WriteBuffer` no longer preallocates (and immediately discards) a zero-filled String of twice its buffer size, which makes writing ZIPs with lots of small files faster.
|
|
13
|
+
* Require `stringio` in `ZipWriter`, which uses it. Previously writing a ZIP could fail with a `NameError` if nothing else in the process had loaded `stringio`.
|
|
14
|
+
* Add `bench/write_buffer_bench.rb`
|
|
15
|
+
|
|
1
16
|
## 6.3.4
|
|
2
17
|
|
|
3
18
|
* Fix a bug whereby `rollback!` would cause an exception without any entries having been written yet (rollback on first entry).
|
data/Rakefile
CHANGED
|
@@ -16,6 +16,14 @@ RSpec::Core::RakeTask.new(:spec)
|
|
|
16
16
|
task :generate_typedefs do
|
|
17
17
|
`bundle exec sord rbi/zip_kit.rbi`
|
|
18
18
|
`bundle exec sord rbi/zip_kit.rbs`
|
|
19
|
+
|
|
20
|
+
# Sord inlines the VERSION literal, which would make every version bump produce a typedef diff
|
|
21
|
+
rbi_path = "rbi/zip_kit.rbi"
|
|
22
|
+
rbi = File.read(rbi_path).sub(/^(\s*)VERSION = T\.let\(.+\)$/, '\1VERSION = T.let(T.unsafe(nil), String)')
|
|
23
|
+
File.write(rbi_path, rbi)
|
|
24
|
+
rbs_path = "rbi/zip_kit.rbs"
|
|
25
|
+
rbs = File.read(rbs_path).sub(/^(\s*)VERSION: untyped$/, '\1VERSION: String')
|
|
26
|
+
File.write(rbs_path, rbs)
|
|
19
27
|
end
|
|
20
28
|
|
|
21
29
|
task default: [:spec, :standard, :generate_typedefs]
|
|
@@ -0,0 +1,183 @@
|
|
|
1
|
+
# Benchmarks ZipKit::WriteBuffer with lots of tiny writes (like the XML fragments
|
|
2
|
+
# a library such as caxlsx produces) and with large writes, comparing it to the
|
|
3
|
+
# previous implementations of WriteBuffer.
|
|
4
|
+
#
|
|
5
|
+
# bundle exec ruby bench/write_buffer_bench.rb
|
|
6
|
+
require "bundler"
|
|
7
|
+
Bundler.setup
|
|
8
|
+
|
|
9
|
+
require "benchmark"
|
|
10
|
+
require "benchmark/ips"
|
|
11
|
+
require_relative "../lib/zip_kit"
|
|
12
|
+
|
|
13
|
+
# Initialization and flushing as in zip_kit 6.3.x. These are not subclasses of ZipKit::WriteBuffer
|
|
14
|
+
# on purpose: with Ruby 3.4, instances of a subclass can have a different object shape, which
|
|
15
|
+
# can make the instance variable caches of methods they share with ZipKit::WriteBuffer miss.
|
|
16
|
+
module WriteBuffer63x
|
|
17
|
+
def initialize(writable, buffer_size)
|
|
18
|
+
@buf = ("\0".b * (buffer_size * 2)).clear
|
|
19
|
+
@buffer_size = buffer_size
|
|
20
|
+
@writable = writable
|
|
21
|
+
end
|
|
22
|
+
|
|
23
|
+
def flush
|
|
24
|
+
unless @buf.empty?
|
|
25
|
+
@writable << @buf
|
|
26
|
+
@buf.clear
|
|
27
|
+
end
|
|
28
|
+
self
|
|
29
|
+
end
|
|
30
|
+
end
|
|
31
|
+
|
|
32
|
+
# WriteBuffer#<< as of zip_kit 6.3.2
|
|
33
|
+
class WriteBuffer632
|
|
34
|
+
include WriteBuffer63x
|
|
35
|
+
|
|
36
|
+
def <<(data)
|
|
37
|
+
if data.bytesize >= @buffer_size
|
|
38
|
+
flush unless @buf.empty?
|
|
39
|
+
@writable << data
|
|
40
|
+
else
|
|
41
|
+
@buf << data
|
|
42
|
+
flush if @buf.bytesize >= @buffer_size
|
|
43
|
+
end
|
|
44
|
+
self
|
|
45
|
+
end
|
|
46
|
+
end
|
|
47
|
+
|
|
48
|
+
# WriteBuffer#<< as of zip_kit 6.3.3/6.3.4, which copies every string with String#b
|
|
49
|
+
class WriteBuffer634
|
|
50
|
+
include WriteBuffer63x
|
|
51
|
+
|
|
52
|
+
def <<(string)
|
|
53
|
+
if string.bytesize >= @buffer_size
|
|
54
|
+
flush
|
|
55
|
+
@writable << string.b
|
|
56
|
+
else
|
|
57
|
+
@buf << string.b
|
|
58
|
+
flush if @buf.bytesize >= @buffer_size
|
|
59
|
+
end
|
|
60
|
+
self
|
|
61
|
+
end
|
|
62
|
+
end
|
|
63
|
+
|
|
64
|
+
# The simplest possible buffer, which does not deal with encodings or large writes
|
|
65
|
+
class NaiveBuffer
|
|
66
|
+
def initialize(io, buffer_size)
|
|
67
|
+
@io = io
|
|
68
|
+
@buffer_size = buffer_size
|
|
69
|
+
@buf = "".b
|
|
70
|
+
end
|
|
71
|
+
|
|
72
|
+
def <<(fragment)
|
|
73
|
+
@buf << fragment
|
|
74
|
+
flush if @buf.bytesize >= @buffer_size
|
|
75
|
+
self
|
|
76
|
+
end
|
|
77
|
+
|
|
78
|
+
def flush
|
|
79
|
+
return if @buf.empty?
|
|
80
|
+
@io << @buf
|
|
81
|
+
@buf.clear
|
|
82
|
+
end
|
|
83
|
+
end
|
|
84
|
+
|
|
85
|
+
BUFFER_SIZE = 64 * 1024
|
|
86
|
+
IMPLEMENTATIONS = {
|
|
87
|
+
"WriteBuffer" => ZipKit::WriteBuffer,
|
|
88
|
+
"WriteBuffer (6.3.2)" => WriteBuffer632,
|
|
89
|
+
"WriteBuffer (6.3.4, String#b)" => WriteBuffer634,
|
|
90
|
+
"Naive String buffer" => NaiveBuffer
|
|
91
|
+
}
|
|
92
|
+
|
|
93
|
+
# Fragments like the ones produced by caxlsx when writing out a worksheet: mostly frozen
|
|
94
|
+
# literals and short dynamic strings (UTF-8 or US-ASCII), mostly ASCII-only.
|
|
95
|
+
fragments = []
|
|
96
|
+
20_000.times do |row|
|
|
97
|
+
fragments << "<row r=\"" << (row + 1).to_s << "\">"
|
|
98
|
+
6.times do |col|
|
|
99
|
+
fragments << "<c r=\"" << "#{("A".ord + col).chr}#{row + 1}" << "\" s=\"" << "1" << "\""
|
|
100
|
+
fragments << ((col == 3) ? " t=\"inlineStr\"><is><t>Grüße, #{row}</t></is>" : "><v>#{row * col}</v>")
|
|
101
|
+
fragments << "</c>"
|
|
102
|
+
end
|
|
103
|
+
fragments << "</row>"
|
|
104
|
+
end
|
|
105
|
+
n_bytes = fragments.sum(&:bytesize)
|
|
106
|
+
|
|
107
|
+
puts RUBY_DESCRIPTION
|
|
108
|
+
puts "#{fragments.length} tiny writes (#{n_bytes} bytes, #{(n_bytes.to_f / fragments.length).round(1)} bytes per write on average)"
|
|
109
|
+
|
|
110
|
+
Benchmark.ips do |x|
|
|
111
|
+
x.config(time: 5, warmup: 2)
|
|
112
|
+
IMPLEMENTATIONS.each do |name, buffer_class|
|
|
113
|
+
x.report("#{name}, tiny writes into CRC32") do
|
|
114
|
+
buf = buffer_class.new(ZipKit::StreamCRC32.new, BUFFER_SIZE)
|
|
115
|
+
fragments.each { |fragment| buf << fragment }
|
|
116
|
+
buf.flush
|
|
117
|
+
end
|
|
118
|
+
end
|
|
119
|
+
x.compare!
|
|
120
|
+
end
|
|
121
|
+
|
|
122
|
+
Benchmark.ips do |x|
|
|
123
|
+
x.config(time: 5, warmup: 2)
|
|
124
|
+
IMPLEMENTATIONS.each do |name, buffer_class|
|
|
125
|
+
x.report("#{name}, tiny writes into write_deflated_file") do
|
|
126
|
+
ZipKit::Streamer.open(ZipKit::NullWriter) do |zip|
|
|
127
|
+
zip.write_deflated_file("sheet.xml") do |sink|
|
|
128
|
+
buf = buffer_class.new(sink, BUFFER_SIZE)
|
|
129
|
+
fragments.each { |fragment| buf << fragment }
|
|
130
|
+
buf.flush
|
|
131
|
+
end
|
|
132
|
+
end
|
|
133
|
+
end
|
|
134
|
+
end
|
|
135
|
+
x.report("No buffer, tiny writes into write_deflated_file") do
|
|
136
|
+
ZipKit::Streamer.open(ZipKit::NullWriter) do |zip|
|
|
137
|
+
zip.write_deflated_file("sheet.xml") do |sink|
|
|
138
|
+
fragments.each { |fragment| sink << fragment }
|
|
139
|
+
end
|
|
140
|
+
end
|
|
141
|
+
end
|
|
142
|
+
x.compare!
|
|
143
|
+
end
|
|
144
|
+
|
|
145
|
+
large_chunk = Random.new(42).bytes(1024 * 1024)
|
|
146
|
+
Benchmark.ips do |x|
|
|
147
|
+
x.config(time: 5, warmup: 2)
|
|
148
|
+
IMPLEMENTATIONS.each do |name, buffer_class|
|
|
149
|
+
x.report("#{name}, 64 writes of 1MB into CRC32") do
|
|
150
|
+
buf = buffer_class.new(ZipKit::StreamCRC32.new, BUFFER_SIZE)
|
|
151
|
+
64.times { buf << large_chunk }
|
|
152
|
+
buf.flush
|
|
153
|
+
end
|
|
154
|
+
end
|
|
155
|
+
x.compare!
|
|
156
|
+
end
|
|
157
|
+
|
|
158
|
+
__END__
|
|
159
|
+
|
|
160
|
+
Apple M1 Pro, macOS 15.7
|
|
161
|
+
|
|
162
|
+
ruby 3.4.1 (2024-12-25 revision 48d4efcb85) +PRISM [arm64-darwin24], without YJIT
|
|
163
|
+
920000 tiny writes (5189483 bytes, 5.6 bytes per write on average)
|
|
164
|
+
|
|
165
|
+
Comparison:
|
|
166
|
+
Naive String buffer, tiny writes into CRC32: 12.1 i/s
|
|
167
|
+
WriteBuffer, tiny writes into CRC32: 12.0 i/s - 1.01x slower
|
|
168
|
+
WriteBuffer (6.3.2), tiny writes into CRC32: 10.3 i/s - 1.17x slower
|
|
169
|
+
WriteBuffer (6.3.4, String#b), tiny writes into CRC32: 8.0 i/s - 1.51x slower
|
|
170
|
+
|
|
171
|
+
Comparison:
|
|
172
|
+
Naive String buffer, tiny writes into write_deflated_file: 7.6 i/s
|
|
173
|
+
WriteBuffer, tiny writes into write_deflated_file: 7.4 i/s - 1.02x slower
|
|
174
|
+
WriteBuffer (6.3.2), tiny writes into write_deflated_file: 6.8 i/s - 1.12x slower
|
|
175
|
+
WriteBuffer (6.3.4, String#b), tiny writes into write_deflated_file: 5.7 i/s - 1.33x slower
|
|
176
|
+
No buffer, tiny writes into write_deflated_file: 1.8 i/s - 4.18x slower
|
|
177
|
+
|
|
178
|
+
Comparison:
|
|
179
|
+
WriteBuffer (6.3.2), 64 writes of 1MB into CRC32: 464.4 i/s
|
|
180
|
+
WriteBuffer (6.3.4, String#b), 64 writes of 1MB into CRC32: 463.0 i/s - same-ish: difference falls within error
|
|
181
|
+
WriteBuffer, 64 writes of 1MB into CRC32: 462.9 i/s - same-ish: difference falls within error
|
|
182
|
+
Naive String buffer, 64 writes of 1MB into CRC32: 121.5 i/s - 3.82x slower
|
|
183
|
+
|
|
@@ -47,7 +47,7 @@
|
|
|
47
47
|
|
|
48
48
|
class ZipKit::BlockDeflate
|
|
49
49
|
DEFAULT_BLOCKSIZE = 1_024 * 1024 * 5
|
|
50
|
-
END_MARKER = [3, 0].pack("C*")
|
|
50
|
+
END_MARKER = [3, 0].pack("C*").freeze
|
|
51
51
|
# Zlib::NO_COMPRESSION..
|
|
52
52
|
VALID_COMPRESSIONS = (Zlib::DEFAULT_COMPRESSION..Zlib::BEST_COMPRESSION).to_a.freeze
|
|
53
53
|
# Write the end marker (\x3\x0) to the given IO.
|
data/lib/zip_kit/block_write.rb
CHANGED
|
@@ -27,11 +27,18 @@ class ZipKit::BlockWrite
|
|
|
27
27
|
@block = block
|
|
28
28
|
end
|
|
29
29
|
|
|
30
|
-
# Make sure those methods raise outright
|
|
31
|
-
|
|
32
|
-
|
|
33
|
-
|
|
34
|
-
|
|
30
|
+
# Make sure those methods raise outright. These are not created with define_method,
|
|
31
|
+
# because methods defined with a block can't be called from a non-main Ractor
|
|
32
|
+
def seek(*)
|
|
33
|
+
raise "seek not supported - this IO adapter is non-rewindable"
|
|
34
|
+
end
|
|
35
|
+
|
|
36
|
+
def pos=(*)
|
|
37
|
+
raise "pos= not supported - this IO adapter is non-rewindable"
|
|
38
|
+
end
|
|
39
|
+
|
|
40
|
+
def to_s(*)
|
|
41
|
+
raise "to_s not supported - this IO adapter is non-rewindable"
|
|
35
42
|
end
|
|
36
43
|
|
|
37
44
|
# Sends a string through to the block stored in the BlockWrite.
|
data/lib/zip_kit/streamer.rb
CHANGED
|
@@ -412,11 +412,10 @@ class ZipKit::Streamer
|
|
|
412
412
|
# Record the central directory offset, so that it can be written into the EOCD record
|
|
413
413
|
cdir_starts_at = @out.tell
|
|
414
414
|
|
|
415
|
-
# Write out the central directory entries, one for each file
|
|
416
|
-
|
|
417
|
-
|
|
418
|
-
|
|
419
|
-
|
|
415
|
+
# Write out the central directory entries, one for each file.
|
|
416
|
+
# Skip fillers which are standing in for broken/incomplete files
|
|
417
|
+
entries = @files.reject(&:filler?)
|
|
418
|
+
entries.each do |entry|
|
|
420
419
|
@writer.write_central_directory_file_header(io: @out,
|
|
421
420
|
local_file_header_location: entry.local_header_offset,
|
|
422
421
|
gp_flags: entry.gp_flags,
|
|
@@ -436,7 +435,7 @@ class ZipKit::Streamer
|
|
|
436
435
|
@writer.write_end_of_central_directory(io: @out,
|
|
437
436
|
start_of_central_directory_location: cdir_starts_at,
|
|
438
437
|
central_directory_size: cdir_size,
|
|
439
|
-
num_files_in_archive:
|
|
438
|
+
num_files_in_archive: entries.length)
|
|
440
439
|
|
|
441
440
|
# Clear the files so that GC will not have to trace all the way to here to deallocate them
|
|
442
441
|
@files.clear
|
data/lib/zip_kit/version.rb
CHANGED
data/lib/zip_kit/write_buffer.rb
CHANGED
|
@@ -12,10 +12,23 @@
|
|
|
12
12
|
# lots of very small writes, and some degree of speedup (about 20%) can be achieved
|
|
13
13
|
# with a buffer of a few KB.
|
|
14
14
|
#
|
|
15
|
-
#
|
|
16
|
-
#
|
|
17
|
-
#
|
|
18
|
-
#
|
|
15
|
+
# The WriteBuffer is also useful in front of a `write_file` / `write_deflated_file` writable
|
|
16
|
+
# if you are going to be appending lots of tiny strings (like XML fragments) to it. Every write
|
|
17
|
+
# into a writable goes through Zlib separately, so coalescing those writes into bigger chunks
|
|
18
|
+
# is much faster.
|
|
19
|
+
#
|
|
20
|
+
# All strings appended to the WriteBuffer are appended as bytes, and the buffer String
|
|
21
|
+
# given to the writable is always in binary encoding (`Encoding::BINARY`). You can therefore mix
|
|
22
|
+
# binary strings and strings in other encodings (for instance UTF-8 with non-ASCII characters)
|
|
23
|
+
# without getting an `Encoding::CompatibilityError`. No intermediate copies of the strings
|
|
24
|
+
# you append (like `String#b` would create) are made.
|
|
25
|
+
#
|
|
26
|
+
# Note that there is no guarantee that the write buffer is going to flush at exactly
|
|
27
|
+
# the given `buffer_size`. The buffer gets flushed when the next write would make it exceed
|
|
28
|
+
# `buffer_size`, so the chunks it outputs are usually a bit smaller than that (strings with
|
|
29
|
+
# multibyte characters can make it go slightly over). For writes of `buffer_size` or larger
|
|
30
|
+
# it will first `flush` and then write through the oversized chunk, without buffering it.
|
|
31
|
+
# This helps conserve memory. Also note that the buffer will *not* duplicate strings for you
|
|
19
32
|
# and *will* yield the same buffer String over and over, so if you are storing it in an
|
|
20
33
|
# Array you might need to duplicate it.
|
|
21
34
|
#
|
|
@@ -25,16 +38,19 @@
|
|
|
25
38
|
# to `<<`. Therefore, if you need to retain the output of the WriteBuffer in, say, an Array,
|
|
26
39
|
# you might need to `.dup` the `String` it gives you.
|
|
27
40
|
class ZipKit::WriteBuffer
|
|
41
|
+
# String#append_as_bytes (Ruby 3.4+) appends the bytes of the string without any encoding
|
|
42
|
+
# negotiation, so the buffer always stays binary. Without it we use String#<<, see `append_bytes`.
|
|
43
|
+
APPEND_AS_BYTES = String.instance_methods.include?(:append_as_bytes)
|
|
44
|
+
|
|
28
45
|
# Creates a new WriteBuffer bypassing into a given writable object
|
|
29
46
|
#
|
|
30
47
|
# @param writable[#<<] An object that responds to `#<<` with a String as argument
|
|
31
48
|
# @param buffer_size[Integer] How many bytes to buffer
|
|
32
49
|
def initialize(writable, buffer_size)
|
|
33
|
-
#
|
|
34
|
-
#
|
|
35
|
-
#
|
|
36
|
-
|
|
37
|
-
@buf = ("\0".b * (buffer_size * 2)).clear
|
|
50
|
+
# No capacity gets preallocated. String#clear releases the memory held by the String,
|
|
51
|
+
# so after the first flush the buffer would have to grow again anyway - and many
|
|
52
|
+
# WriteBuffers (like the ones used for the CRC32 of small ZIP entries) never fill up.
|
|
53
|
+
@buf = "".b
|
|
38
54
|
@buffer_size = buffer_size
|
|
39
55
|
@writable = writable
|
|
40
56
|
end
|
|
@@ -45,11 +61,15 @@ class ZipKit::WriteBuffer
|
|
|
45
61
|
# @param string[String] data to be written
|
|
46
62
|
# @return self
|
|
47
63
|
def <<(string)
|
|
48
|
-
if string.bytesize
|
|
49
|
-
|
|
64
|
+
if @buf.bytesize + string.bytesize < @buffer_size
|
|
65
|
+
APPEND_AS_BYTES ? @buf.append_as_bytes(string) : append_bytes(string)
|
|
66
|
+
elsif string.bytesize >= @buffer_size
|
|
67
|
+
flush
|
|
68
|
+
# String#b does not copy the bytes of a large String, the new String shares them
|
|
50
69
|
@writable << string.b
|
|
51
70
|
else
|
|
52
|
-
@buf
|
|
71
|
+
flush if @buf.bytesize + string.bytesize > @buffer_size
|
|
72
|
+
append_bytes(string)
|
|
53
73
|
flush if @buf.bytesize >= @buffer_size
|
|
54
74
|
end
|
|
55
75
|
self
|
|
@@ -60,7 +80,8 @@ class ZipKit::WriteBuffer
|
|
|
60
80
|
# @return self
|
|
61
81
|
def flush
|
|
62
82
|
unless @buf.empty?
|
|
63
|
-
|
|
83
|
+
# force_encoding does not copy the String, it only changes its encoding
|
|
84
|
+
@writable << @buf.force_encoding(Encoding::BINARY)
|
|
64
85
|
@buf.clear
|
|
65
86
|
end
|
|
66
87
|
self
|
|
@@ -68,4 +89,19 @@ class ZipKit::WriteBuffer
|
|
|
68
89
|
|
|
69
90
|
# `flush!` was renamed to `flush` but we preserve this method for backwards compatibility
|
|
70
91
|
alias_method :flush!, :flush
|
|
92
|
+
|
|
93
|
+
private
|
|
94
|
+
|
|
95
|
+
# Appends the bytes of the string without copying it. Without String#append_as_bytes (Ruby < 3.4)
|
|
96
|
+
# String#<< is used, which may change the encoding of the buffer or raise if the encodings are
|
|
97
|
+
# incompatible - in that case we append the bytes of the string instead. The buffer is forced
|
|
98
|
+
# back into binary before it is handed to the writable, see `flush`.
|
|
99
|
+
def append_bytes(string)
|
|
100
|
+
return @buf.append_as_bytes(string) if APPEND_AS_BYTES
|
|
101
|
+
|
|
102
|
+
@buf << string
|
|
103
|
+
rescue Encoding::CompatibilityError
|
|
104
|
+
@buf.force_encoding(Encoding::BINARY)
|
|
105
|
+
@buf << string.b
|
|
106
|
+
end
|
|
71
107
|
end
|
data/lib/zip_kit/zip_writer.rb
CHANGED
|
@@ -1,5 +1,7 @@
|
|
|
1
1
|
# frozen_string_literal: true
|
|
2
2
|
|
|
3
|
+
require "stringio"
|
|
4
|
+
|
|
3
5
|
# A low-level ZIP file data writer. You can use it to write out various headers and central directory elements
|
|
4
6
|
# separately. The class handles the actual encoding of the data according to the ZIP format APPNOTE document.
|
|
5
7
|
#
|
|
@@ -27,7 +29,7 @@
|
|
|
27
29
|
class ZipKit::ZipWriter
|
|
28
30
|
FOUR_BYTE_MAX_UINT = 0xFFFFFFFF
|
|
29
31
|
TWO_BYTE_MAX_UINT = 0xFFFF
|
|
30
|
-
ZIP_KIT_COMMENT = "Written using ZipKit %<version>s" % {version: ZipKit::VERSION}
|
|
32
|
+
ZIP_KIT_COMMENT = ("Written using ZipKit %<version>s" % {version: ZipKit::VERSION}).freeze
|
|
31
33
|
VERSION_MADE_BY = 52
|
|
32
34
|
VERSION_NEEDED_TO_EXTRACT = 20
|
|
33
35
|
VERSION_NEEDED_TO_EXTRACT_ZIP64 = 45
|
|
@@ -38,7 +40,7 @@ class ZipKit::ZipWriter
|
|
|
38
40
|
MADE_BY_SIGNATURE = begin
|
|
39
41
|
# A combination of the VERSION_MADE_BY low byte and the OS type high byte
|
|
40
42
|
os_type = 3 # UNIX
|
|
41
|
-
[VERSION_MADE_BY, os_type].pack("CC")
|
|
43
|
+
[VERSION_MADE_BY, os_type].pack("CC").freeze
|
|
42
44
|
end
|
|
43
45
|
|
|
44
46
|
C_UINT4 = "V" # Encode a 4-byte unsigned little-endian uint
|
data/rbi/zip_kit.rbi
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
# typed: strong
|
|
2
2
|
module ZipKit
|
|
3
|
-
VERSION = T.let(
|
|
3
|
+
VERSION = T.let(T.unsafe(nil), String)
|
|
4
4
|
|
|
5
5
|
class Railtie < Rails::Railtie
|
|
6
6
|
end
|
|
@@ -942,7 +942,7 @@ module ZipKit
|
|
|
942
942
|
class ZipWriter
|
|
943
943
|
FOUR_BYTE_MAX_UINT = T.let(0xFFFFFFFF, T.untyped)
|
|
944
944
|
TWO_BYTE_MAX_UINT = T.let(0xFFFF, T.untyped)
|
|
945
|
-
ZIP_KIT_COMMENT = T.let("Written using ZipKit %<version>s" % {version: ZipKit::VERSION}, T.untyped)
|
|
945
|
+
ZIP_KIT_COMMENT = T.let(("Written using ZipKit %<version>s" % {version: ZipKit::VERSION}).freeze, T.untyped)
|
|
946
946
|
VERSION_MADE_BY = T.let(52, T.untyped)
|
|
947
947
|
VERSION_NEEDED_TO_EXTRACT = T.let(20, T.untyped)
|
|
948
948
|
VERSION_NEEDED_TO_EXTRACT_ZIP64 = T.let(45, T.untyped)
|
|
@@ -953,7 +953,7 @@ module ZipKit
|
|
|
953
953
|
MADE_BY_SIGNATURE = T.let(begin
|
|
954
954
|
# A combination of the VERSION_MADE_BY low byte and the OS type high byte
|
|
955
955
|
os_type = 3 # UNIX
|
|
956
|
-
[VERSION_MADE_BY, os_type].pack("CC")
|
|
956
|
+
[VERSION_MADE_BY, os_type].pack("CC").freeze
|
|
957
957
|
end, T.untyped)
|
|
958
958
|
C_UINT4 = T.let("V", T.untyped)
|
|
959
959
|
C_UINT2 = T.let("v", T.untyped)
|
|
@@ -1164,6 +1164,20 @@ end, T.untyped)
|
|
|
1164
1164
|
sig { params(block: T.proc.params(bytes: String).void).void }
|
|
1165
1165
|
def initialize(&block); end
|
|
1166
1166
|
|
|
1167
|
+
# sord omit - no YARD return type given, using untyped
|
|
1168
|
+
# Make sure those methods raise outright. These are not created with define_method,
|
|
1169
|
+
# because methods defined with a block can't be called from a non-main Ractor
|
|
1170
|
+
sig { returns(T.untyped) }
|
|
1171
|
+
def seek; end
|
|
1172
|
+
|
|
1173
|
+
# sord omit - no YARD return type given, using untyped
|
|
1174
|
+
sig { returns(T.untyped) }
|
|
1175
|
+
def pos=; end
|
|
1176
|
+
|
|
1177
|
+
# sord omit - no YARD return type given, using untyped
|
|
1178
|
+
sig { returns(T.untyped) }
|
|
1179
|
+
def to_s; end
|
|
1180
|
+
|
|
1167
1181
|
# Sends a string through to the block stored in the BlockWrite.
|
|
1168
1182
|
#
|
|
1169
1183
|
# _@param_ `buf` — the string to write. Note that a zero-length String will not be forwarded to the block, as it has special meaning when used with chunked encoding (it indicates the end of the stream).
|
|
@@ -1737,10 +1751,23 @@ end, T.untyped)
|
|
|
1737
1751
|
# lots of very small writes, and some degree of speedup (about 20%) can be achieved
|
|
1738
1752
|
# with a buffer of a few KB.
|
|
1739
1753
|
#
|
|
1740
|
-
#
|
|
1741
|
-
#
|
|
1742
|
-
#
|
|
1743
|
-
#
|
|
1754
|
+
# The WriteBuffer is also useful in front of a `write_file` / `write_deflated_file` writable
|
|
1755
|
+
# if you are going to be appending lots of tiny strings (like XML fragments) to it. Every write
|
|
1756
|
+
# into a writable goes through Zlib separately, so coalescing those writes into bigger chunks
|
|
1757
|
+
# is much faster.
|
|
1758
|
+
#
|
|
1759
|
+
# All strings appended to the WriteBuffer are appended as bytes, and the buffer String
|
|
1760
|
+
# given to the writable is always in binary encoding (`Encoding::BINARY`). You can therefore mix
|
|
1761
|
+
# binary strings and strings in other encodings (for instance UTF-8 with non-ASCII characters)
|
|
1762
|
+
# without getting an `Encoding::CompatibilityError`. No intermediate copies of the strings
|
|
1763
|
+
# you append (like `String#b` would create) are made.
|
|
1764
|
+
#
|
|
1765
|
+
# Note that there is no guarantee that the write buffer is going to flush at exactly
|
|
1766
|
+
# the given `buffer_size`. The buffer gets flushed when the next write would make it exceed
|
|
1767
|
+
# `buffer_size`, so the chunks it outputs are usually a bit smaller than that (strings with
|
|
1768
|
+
# multibyte characters can make it go slightly over). For writes of `buffer_size` or larger
|
|
1769
|
+
# it will first `flush` and then write through the oversized chunk, without buffering it.
|
|
1770
|
+
# This helps conserve memory. Also note that the buffer will *not* duplicate strings for you
|
|
1744
1771
|
# and *will* yield the same buffer String over and over, so if you are storing it in an
|
|
1745
1772
|
# Array you might need to duplicate it.
|
|
1746
1773
|
#
|
|
@@ -1750,6 +1777,8 @@ end, T.untyped)
|
|
|
1750
1777
|
# to `<<`. Therefore, if you need to retain the output of the WriteBuffer in, say, an Array,
|
|
1751
1778
|
# you might need to `.dup` the `String` it gives you.
|
|
1752
1779
|
class WriteBuffer
|
|
1780
|
+
APPEND_AS_BYTES = T.let(String.instance_methods.include?(:append_as_bytes), T.untyped)
|
|
1781
|
+
|
|
1753
1782
|
# sord duck - #<< looks like a duck type, replacing with untyped
|
|
1754
1783
|
# Creates a new WriteBuffer bypassing into a given writable object
|
|
1755
1784
|
#
|
|
@@ -1773,6 +1802,15 @@ end, T.untyped)
|
|
|
1773
1802
|
# _@return_ — self
|
|
1774
1803
|
sig { returns(T.untyped) }
|
|
1775
1804
|
def flush; end
|
|
1805
|
+
|
|
1806
|
+
# sord omit - no YARD type given for "string", using untyped
|
|
1807
|
+
# sord omit - no YARD return type given, using untyped
|
|
1808
|
+
# Appends the bytes of the string without copying it. Without String#append_as_bytes (Ruby < 3.4)
|
|
1809
|
+
# String#<< is used, which may change the encoding of the buffer or raise if the encodings are
|
|
1810
|
+
# incompatible - in that case we append the bytes of the string instead. The buffer is forced
|
|
1811
|
+
# back into binary before it is handed to the writable, see `flush`.
|
|
1812
|
+
sig { params(string: T.untyped).returns(T.untyped) }
|
|
1813
|
+
def append_bytes(string); end
|
|
1776
1814
|
end
|
|
1777
1815
|
|
|
1778
1816
|
# A lot of objects in ZipKit accept bytes that may be sent
|
|
@@ -1855,7 +1893,7 @@ end, T.untyped)
|
|
|
1855
1893
|
# compressed_string = ZipKit::BlockDeflate.deflate_chunk(big_string)
|
|
1856
1894
|
class BlockDeflate
|
|
1857
1895
|
DEFAULT_BLOCKSIZE = T.let(1_024 * 1024 * 5, T.untyped)
|
|
1858
|
-
END_MARKER = T.let([3, 0].pack("C*"), T.untyped)
|
|
1896
|
+
END_MARKER = T.let([3, 0].pack("C*").freeze, T.untyped)
|
|
1859
1897
|
VALID_COMPRESSIONS = T.let((Zlib::DEFAULT_COMPRESSION..Zlib::BEST_COMPRESSION).to_a.freeze, T.untyped)
|
|
1860
1898
|
|
|
1861
1899
|
# Write the end marker (\x3\x0) to the given IO.
|
|
@@ -2201,7 +2239,7 @@ end, T.untyped)
|
|
|
2201
2239
|
# carry, so we copy it into our code.
|
|
2202
2240
|
class RackChunkedBody
|
|
2203
2241
|
TERM = T.let("\r\n", T.untyped)
|
|
2204
|
-
TAIL = T.let("0
|
|
2242
|
+
TAIL = T.let("0\r\n", T.untyped)
|
|
2205
2243
|
|
|
2206
2244
|
# sord duck - #each looks like a duck type, replacing with untyped
|
|
2207
2245
|
# _@param_ `body` — the enumerable that yields bytes, usually a `OutputEnumerator`
|
data/rbi/zip_kit.rbs
CHANGED
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
module ZipKit
|
|
2
|
-
VERSION:
|
|
2
|
+
VERSION: String
|
|
3
3
|
|
|
4
4
|
class Railtie < Rails::Railtie
|
|
5
5
|
end
|
|
@@ -1038,6 +1038,17 @@ module ZipKit
|
|
|
1038
1038
|
# _@param_ `block` — The block that will be called when this object receives the `<<` message
|
|
1039
1039
|
def initialize: () ?{ (String bytes) -> void } -> void
|
|
1040
1040
|
|
|
1041
|
+
# sord omit - no YARD return type given, using untyped
|
|
1042
|
+
# Make sure those methods raise outright. These are not created with define_method,
|
|
1043
|
+
# because methods defined with a block can't be called from a non-main Ractor
|
|
1044
|
+
def seek: () -> untyped
|
|
1045
|
+
|
|
1046
|
+
# sord omit - no YARD return type given, using untyped
|
|
1047
|
+
def pos=: () -> untyped
|
|
1048
|
+
|
|
1049
|
+
# sord omit - no YARD return type given, using untyped
|
|
1050
|
+
def to_s: () -> untyped
|
|
1051
|
+
|
|
1041
1052
|
# Sends a string through to the block stored in the BlockWrite.
|
|
1042
1053
|
#
|
|
1043
1054
|
# _@param_ `buf` — the string to write. Note that a zero-length String will not be forwarded to the block, as it has special meaning when used with chunked encoding (it indicates the end of the stream).
|
|
@@ -1510,10 +1521,23 @@ module ZipKit
|
|
|
1510
1521
|
# lots of very small writes, and some degree of speedup (about 20%) can be achieved
|
|
1511
1522
|
# with a buffer of a few KB.
|
|
1512
1523
|
#
|
|
1513
|
-
#
|
|
1514
|
-
#
|
|
1515
|
-
#
|
|
1516
|
-
#
|
|
1524
|
+
# The WriteBuffer is also useful in front of a `write_file` / `write_deflated_file` writable
|
|
1525
|
+
# if you are going to be appending lots of tiny strings (like XML fragments) to it. Every write
|
|
1526
|
+
# into a writable goes through Zlib separately, so coalescing those writes into bigger chunks
|
|
1527
|
+
# is much faster.
|
|
1528
|
+
#
|
|
1529
|
+
# All strings appended to the WriteBuffer are appended as bytes, and the buffer String
|
|
1530
|
+
# given to the writable is always in binary encoding (`Encoding::BINARY`). You can therefore mix
|
|
1531
|
+
# binary strings and strings in other encodings (for instance UTF-8 with non-ASCII characters)
|
|
1532
|
+
# without getting an `Encoding::CompatibilityError`. No intermediate copies of the strings
|
|
1533
|
+
# you append (like `String#b` would create) are made.
|
|
1534
|
+
#
|
|
1535
|
+
# Note that there is no guarantee that the write buffer is going to flush at exactly
|
|
1536
|
+
# the given `buffer_size`. The buffer gets flushed when the next write would make it exceed
|
|
1537
|
+
# `buffer_size`, so the chunks it outputs are usually a bit smaller than that (strings with
|
|
1538
|
+
# multibyte characters can make it go slightly over). For writes of `buffer_size` or larger
|
|
1539
|
+
# it will first `flush` and then write through the oversized chunk, without buffering it.
|
|
1540
|
+
# This helps conserve memory. Also note that the buffer will *not* duplicate strings for you
|
|
1517
1541
|
# and *will* yield the same buffer String over and over, so if you are storing it in an
|
|
1518
1542
|
# Array you might need to duplicate it.
|
|
1519
1543
|
#
|
|
@@ -1523,6 +1547,8 @@ module ZipKit
|
|
|
1523
1547
|
# to `<<`. Therefore, if you need to retain the output of the WriteBuffer in, say, an Array,
|
|
1524
1548
|
# you might need to `.dup` the `String` it gives you.
|
|
1525
1549
|
class WriteBuffer
|
|
1550
|
+
APPEND_AS_BYTES: untyped
|
|
1551
|
+
|
|
1526
1552
|
# sord duck - #<< looks like a duck type, replacing with untyped
|
|
1527
1553
|
# Creates a new WriteBuffer bypassing into a given writable object
|
|
1528
1554
|
#
|
|
@@ -1543,6 +1569,14 @@ module ZipKit
|
|
|
1543
1569
|
#
|
|
1544
1570
|
# _@return_ — self
|
|
1545
1571
|
def flush: () -> untyped
|
|
1572
|
+
|
|
1573
|
+
# sord omit - no YARD type given for "string", using untyped
|
|
1574
|
+
# sord omit - no YARD return type given, using untyped
|
|
1575
|
+
# Appends the bytes of the string without copying it. Without String#append_as_bytes (Ruby < 3.4)
|
|
1576
|
+
# String#<< is used, which may change the encoding of the buffer or raise if the encodings are
|
|
1577
|
+
# incompatible - in that case we append the bytes of the string instead. The buffer is forced
|
|
1578
|
+
# back into binary before it is handed to the writable, see `flush`.
|
|
1579
|
+
def append_bytes: (untyped string) -> untyped
|
|
1546
1580
|
end
|
|
1547
1581
|
|
|
1548
1582
|
# A lot of objects in ZipKit accept bytes that may be sent
|
metadata
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
--- !ruby/object:Gem::Specification
|
|
2
2
|
name: zip_kit
|
|
3
3
|
version: !ruby/object:Gem::Version
|
|
4
|
-
version: 6.3.
|
|
4
|
+
version: 6.3.6
|
|
5
5
|
platform: ruby
|
|
6
6
|
authors:
|
|
7
7
|
- Julik Tarkhanov
|
|
@@ -11,7 +11,7 @@ authors:
|
|
|
11
11
|
- Felix Bünemann
|
|
12
12
|
bindir: exe
|
|
13
13
|
cert_chain: []
|
|
14
|
-
date:
|
|
14
|
+
date: 2026-10-02 00:00:00.000000000 Z
|
|
15
15
|
dependencies:
|
|
16
16
|
- !ruby/object:Gem::Dependency
|
|
17
17
|
name: bundler
|
|
@@ -367,6 +367,7 @@ files:
|
|
|
367
367
|
- README.md
|
|
368
368
|
- RUBYZIP_DIFFERENCES.md
|
|
369
369
|
- Rakefile
|
|
370
|
+
- bench/write_buffer_bench.rb
|
|
370
371
|
- examples/archive_size_estimate.rb
|
|
371
372
|
- examples/config.ru
|
|
372
373
|
- examples/deferred_write.rb
|