zxing_ffi 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +7 -0
- data/CHANGELOG.md +27 -0
- data/LICENSE.txt +21 -0
- data/README.md +275 -0
- data/exe/zxing-scan +6 -0
- data/lib/zxing_ffi/barcode.rb +77 -0
- data/lib/zxing_ffi/cli.rb +152 -0
- data/lib/zxing_ffi/config.rb +119 -0
- data/lib/zxing_ffi/dedupe.rb +79 -0
- data/lib/zxing_ffi/diagnostics.rb +69 -0
- data/lib/zxing_ffi/dpi.rb +103 -0
- data/lib/zxing_ffi/errors.rb +69 -0
- data/lib/zxing_ffi/formats.rb +207 -0
- data/lib/zxing_ffi/geometry.rb +508 -0
- data/lib/zxing_ffi/header_probe.rb +98 -0
- data/lib/zxing_ffi/image.rb +123 -0
- data/lib/zxing_ffi/image_magick.rb +95 -0
- data/lib/zxing_ffi/library_defaults.rb +7 -0
- data/lib/zxing_ffi/library_loader.rb +176 -0
- data/lib/zxing_ffi/loaders/base.rb +155 -0
- data/lib/zxing_ffi/loaders/image_magick.rb +159 -0
- data/lib/zxing_ffi/loaders/pnm.rb +113 -0
- data/lib/zxing_ffi/loaders/poppler.rb +260 -0
- data/lib/zxing_ffi/loaders/registry.rb +54 -0
- data/lib/zxing_ffi/loaders/vips.rb +332 -0
- data/lib/zxing_ffi/loaders.rb +41 -0
- data/lib/zxing_ffi/native.rb +237 -0
- data/lib/zxing_ffi/pnm.rb +676 -0
- data/lib/zxing_ffi/reader.rb +271 -0
- data/lib/zxing_ffi/scanner.rb +295 -0
- data/lib/zxing_ffi/sniffer.rb +155 -0
- data/lib/zxing_ffi/source.rb +77 -0
- data/lib/zxing_ffi/strategy.rb +293 -0
- data/lib/zxing_ffi/subprocess.rb +416 -0
- data/lib/zxing_ffi/transformers/base.rb +69 -0
- data/lib/zxing_ffi/transformers/image_magick.rb +72 -0
- data/lib/zxing_ffi/transformers/vips.rb +85 -0
- data/lib/zxing_ffi/transformers.rb +36 -0
- data/lib/zxing_ffi/version.rb +5 -0
- data/lib/zxing_ffi.rb +82 -0
- metadata +108 -0
|
@@ -0,0 +1,676 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module ZXingFFI
|
|
4
|
+
# Pure-Ruby decoder for the Netpbm formats (PBM, PGM and PPM, both the ASCII "plain" variants +P1+–+P3+ and
|
|
5
|
+
# the binary +P4+–+P6+) producing 8-bit luminance, plus a minimal PGM writer.
|
|
6
|
+
#
|
|
7
|
+
# Besides reading user files it parses the PGM that +pdftoppm -gray+ and ImageMagick's +pgm:-+ write.
|
|
8
|
+
# That common case, a binary PGM with maxval 255, costs a header parse and a slice of the input
|
|
9
|
+
# with no per-pixel work. Other variants are converted with C-speed primitives where Ruby has them
|
|
10
|
+
# (+String#tr+ look-up tables, +unpack+/+pack+); only colour (PPM) input needs a per-pixel Ruby loop.
|
|
11
|
+
#
|
|
12
|
+
# Every variant except 8-bit PGM held in a String is converted in steps of 16K pixels (64 KiB of text for
|
|
13
|
+
# the ASCII variants) appended to a single output String, and each step's temporaries are freed before
|
|
14
|
+
# the next, so the memory needed beyond the input and the output stays under a few MB whatever the image
|
|
15
|
+
# size or content. An IO is read the same way, in chunks, and not past the raster (see {decode}).
|
|
16
|
+
#
|
|
17
|
+
# Conversions applied by {decode}:
|
|
18
|
+
# - samples are scaled to 0..255 with rounding, +(v * 255 + maxval / 2) / maxval+, so 16-bit input is
|
|
19
|
+
# scaled, never truncated; samples above maxval (invalid) are clamped to 255;
|
|
20
|
+
# - PBM bits become 255 (bit 0, white) and 0 (bit 1, black);
|
|
21
|
+
# - PPM pixels become zxing-cpp's luminance +(306 * r + 601 * g + 117 * b + 0x200) >> 10+ of the
|
|
22
|
+
# scaled channels.
|
|
23
|
+
#
|
|
24
|
+
# Header syntax follows the Netpbm reference implementation: tokens are separated by any mix of whitespace
|
|
25
|
+
# (space, TAB, LF, VT, FF, CR) and +#+ comments running to the end of the line, and exactly one whitespace
|
|
26
|
+
# byte ends the header (so +"255\r\n"+ leaves the LF as the first raster byte). A comment directly after
|
|
27
|
+
# the last token is allowed; the line break ending it is then that byte.
|
|
28
|
+
#
|
|
29
|
+
# @see https://netpbm.sourceforge.net/doc/pnm.html
|
|
30
|
+
module Pnm
|
|
31
|
+
# A parsed PNM header.
|
|
32
|
+
#
|
|
33
|
+
# @!attribute [r] kind
|
|
34
|
+
# @return [Symbol] +:p1+ .. +:p6+
|
|
35
|
+
# @!attribute [r] width
|
|
36
|
+
# @return [Integer]
|
|
37
|
+
# @!attribute [r] height
|
|
38
|
+
# @return [Integer]
|
|
39
|
+
# @!attribute [r] maxval
|
|
40
|
+
# @return [Integer] largest sample value (1..65535); always 1 for bitmaps (+P1+, +P4+)
|
|
41
|
+
# @!attribute [r] offset
|
|
42
|
+
# @return [Integer] byte index at which the raster starts
|
|
43
|
+
Header = Data.define(:kind, :width, :height, :maxval, :offset) do
|
|
44
|
+
# @return [Integer] width × height
|
|
45
|
+
def pixel_count
|
|
46
|
+
width * height
|
|
47
|
+
end
|
|
48
|
+
|
|
49
|
+
# Size of a binary raster (+P4+–+P6+) in bytes; +nil+ for the ASCII variants, whose size varies.
|
|
50
|
+
# @return [Integer, nil]
|
|
51
|
+
def raster_bytesize
|
|
52
|
+
sample_bytes = (maxval > 255) ? 2 : 1
|
|
53
|
+
case kind
|
|
54
|
+
when :p4 then (width + 7) / 8 * height
|
|
55
|
+
when :p5 then pixel_count * sample_bytes
|
|
56
|
+
when :p6 then pixel_count * 3 * sample_bytes
|
|
57
|
+
end
|
|
58
|
+
end
|
|
59
|
+
end
|
|
60
|
+
|
|
61
|
+
# A decoded image.
|
|
62
|
+
#
|
|
63
|
+
# @!attribute [r] width
|
|
64
|
+
# @return [Integer]
|
|
65
|
+
# @!attribute [r] height
|
|
66
|
+
# @return [Integer]
|
|
67
|
+
# @!attribute [r] pixels
|
|
68
|
+
# @return [String] BINARY, width × height bytes of 8-bit luminance, row-major
|
|
69
|
+
Decoded = Data.define(:width, :height, :pixels)
|
|
70
|
+
|
|
71
|
+
# Largest accepted width or height (libZXing takes +int+ dimensions).
|
|
72
|
+
MAX_DIMENSION = 2**31 - 1
|
|
73
|
+
# Largest maxval allowed by the Netpbm specification.
|
|
74
|
+
MAX_MAXVAL = 65_535
|
|
75
|
+
|
|
76
|
+
# Raised internally when an incomplete header may continue in data not read yet.
|
|
77
|
+
class NeedMoreData < StandardError; end
|
|
78
|
+
|
|
79
|
+
BINARY = Encoding::BINARY
|
|
80
|
+
MAGIC = {"P1" => :p1, "P2" => :p2, "P3" => :p3, "P4" => :p4, "P5" => :p5, "P6" => :p6}.freeze
|
|
81
|
+
BITMAP_KINDS = %i[p1 p4].freeze
|
|
82
|
+
WHITESPACE = " \t\n\v\f\r"
|
|
83
|
+
# Bytes allowed right after a header token: whitespace or the start of a comment.
|
|
84
|
+
DELIMITERS = "#{WHITESPACE}#".bytes.freeze
|
|
85
|
+
HASH = "#".ord
|
|
86
|
+
NON_WHITESPACE = /[^ \t\n\v\f\r]/
|
|
87
|
+
WHITESPACE_BYTE = /[ \t\n\v\f\r]/
|
|
88
|
+
LINE_END = /[\r\n]/
|
|
89
|
+
NON_DIGIT = /[^0-9]/
|
|
90
|
+
LEADING_ZEROS = /\A0+(?=[0-9])/
|
|
91
|
+
NONZERO_DIGIT = /[1-9]/
|
|
92
|
+
NON_BIT = /[^01]/
|
|
93
|
+
# Bytes other than digits and whitespace (String#count syntax).
|
|
94
|
+
NOT_SAMPLE_TEXT = "^0-9 \t\n\v\f\r"
|
|
95
|
+
# First read when parsing a header from an IO; doubled (up to the limit) until the header fits.
|
|
96
|
+
HEADER_CHUNK = 4096
|
|
97
|
+
# Longest header read from an IO, comments included: a longer one raises UnsupportedInput rather than being
|
|
98
|
+
# buffered without bound (an endless comment in an untrusted stream). String input is already in memory.
|
|
99
|
+
MAX_HEADER_BYTES = 1024 * 1024
|
|
100
|
+
# Pixels converted per step where conversion builds Arrays (16-bit P5, P6): < 1 MiB of temporaries.
|
|
101
|
+
# (Larger steps are no faster.)
|
|
102
|
+
CHUNK_PIXELS = 16_384
|
|
103
|
+
# Bytes per step where conversion is byte for byte (8-bit P5 read from an IO).
|
|
104
|
+
BYTE_CHUNK = 256 * 1024
|
|
105
|
+
# Bytes per step for P4 (16Ki pixels): whole rows when a row fits, otherwise parts of one.
|
|
106
|
+
BIT_CHUNK = CHUNK_PIXELS / 8
|
|
107
|
+
# Bytes of text per step for the ASCII variants.
|
|
108
|
+
TEXT_CHUNK = 64 * 1024
|
|
109
|
+
# Spaces that overwrite the comments of an ASCII raster, one chunk of text at a time.
|
|
110
|
+
BLANKS = (" " * TEXT_CHUNK).b.freeze
|
|
111
|
+
# Samples cut by chunk boundaries longer than this are shortened (see shorten_sample!).
|
|
112
|
+
LONG_SAMPLE = 64
|
|
113
|
+
# Output capacity reserved up front when the data has not yet shown that the header's size is real.
|
|
114
|
+
PREALLOCATION_LIMIT = 64 * 1024 * 1024
|
|
115
|
+
# String#tr source covering every byte, and each byte escaped for use in a tr replacement.
|
|
116
|
+
TR_ALL_BYTES = "\x00-\xFF".b.freeze
|
|
117
|
+
TR_CHARS = Array.new(256) { |v| ["-", "\\", "^"].include?(v.chr) ? "\\#{v.chr}".b.freeze : v.chr(BINARY).freeze }.freeze
|
|
118
|
+
# String#tr replacement for "01": bit 0 is white, bit 1 is black.
|
|
119
|
+
BIT_PIXELS = "\xFF\x00".b.freeze
|
|
120
|
+
# zxing-cpp's RGB weights, as tables (the rounding term 0x200 is folded into BLUE).
|
|
121
|
+
RED = Array.new(256) { |v| 306 * v }.freeze
|
|
122
|
+
GREEN = Array.new(256) { |v| 601 * v }.freeze
|
|
123
|
+
BLUE = Array.new(256) { |v| 117 * v + 0x200 }.freeze
|
|
124
|
+
|
|
125
|
+
# Reads a raster sequentially: from the data String itself, or from the bytes buffered while the header
|
|
126
|
+
# was parsed followed by the rest of an IO (read on demand, never beyond what is asked for).
|
|
127
|
+
class RasterReader
|
|
128
|
+
def initialize(buffer, offset, io)
|
|
129
|
+
@buffer = buffer
|
|
130
|
+
@position = offset
|
|
131
|
+
@io = io
|
|
132
|
+
end
|
|
133
|
+
|
|
134
|
+
# @return [String] BINARY, owned by the caller (who may clear it); +size+ bytes, fewer only at the end
|
|
135
|
+
# of the data
|
|
136
|
+
def read(size)
|
|
137
|
+
chunk = @buffer.byteslice(@position, size)
|
|
138
|
+
@position += chunk.bytesize
|
|
139
|
+
return chunk if @io.nil? || chunk.bytesize == size
|
|
140
|
+
|
|
141
|
+
while chunk.bytesize < size
|
|
142
|
+
more = @io.read(size - chunk.bytesize)
|
|
143
|
+
break if more.nil? || more.empty?
|
|
144
|
+
|
|
145
|
+
more = more.b if more.frozen? || more.encoding != Encoding::BINARY
|
|
146
|
+
if chunk.empty?
|
|
147
|
+
chunk = more
|
|
148
|
+
else
|
|
149
|
+
chunk << more
|
|
150
|
+
more.clear
|
|
151
|
+
end
|
|
152
|
+
end
|
|
153
|
+
chunk
|
|
154
|
+
end
|
|
155
|
+
end
|
|
156
|
+
|
|
157
|
+
private_constant :NeedMoreData, :BINARY, :MAGIC, :BITMAP_KINDS, :WHITESPACE, :DELIMITERS, :HASH,
|
|
158
|
+
:NON_WHITESPACE, :WHITESPACE_BYTE, :LINE_END, :NON_DIGIT, :LEADING_ZEROS, :NONZERO_DIGIT, :NON_BIT,
|
|
159
|
+
:NOT_SAMPLE_TEXT, :HEADER_CHUNK, :CHUNK_PIXELS, :BYTE_CHUNK, :BIT_CHUNK,
|
|
160
|
+
:TEXT_CHUNK, :BLANKS, :LONG_SAMPLE, :PREALLOCATION_LIMIT, :TR_ALL_BYTES, :TR_CHARS, :BIT_PIXELS, :RED,
|
|
161
|
+
:GREEN, :BLUE, :RasterReader
|
|
162
|
+
|
|
163
|
+
class << self
|
|
164
|
+
# Parses the header of a PNM image without looking at its raster.
|
|
165
|
+
#
|
|
166
|
+
# @param data [String, IO] the image bytes (any encoding; always treated as bytes), or an IO positioned
|
|
167
|
+
# at the start of the image. Only the bytes the header needs are read from an IO, and a seekable IO is
|
|
168
|
+
# moved back to where it was; +offset+ is then relative to that position.
|
|
169
|
+
# @return [Header]
|
|
170
|
+
# @raise [UnsupportedInput] if the data is not a PNM image (P7/PAM included) or its header is malformed
|
|
171
|
+
# or truncated
|
|
172
|
+
# @example
|
|
173
|
+
# ZXingFFI::Pnm.read_header("P5\n640 480\n255\n...") # => #<data Header kind=:p5, width=640, ...>
|
|
174
|
+
def read_header(data)
|
|
175
|
+
return parse_header(binary(data), final: true) unless io?(data)
|
|
176
|
+
|
|
177
|
+
position = io_position(data)
|
|
178
|
+
begin
|
|
179
|
+
read_io_header(data).first
|
|
180
|
+
ensure
|
|
181
|
+
data.seek(position) if position
|
|
182
|
+
end
|
|
183
|
+
end
|
|
184
|
+
|
|
185
|
+
# Decodes a PNM image to 8-bit luminance.
|
|
186
|
+
#
|
|
187
|
+
# Only the first image of a multi-image file is decoded: bytes after its raster are ignored.
|
|
188
|
+
#
|
|
189
|
+
# @param data [String, IO] the image bytes (any encoding; always treated as bytes), or an IO positioned
|
|
190
|
+
# at the start of the image. An IO is read in bounded chunks, never all at once: the header 4 KiB at a
|
|
191
|
+
# time (doubling for long headers), then the raster, exactly up to its end for the binary variants
|
|
192
|
+
# and in 64 KiB chunks for the ASCII ones. So what follows the image is read only by the first header
|
|
193
|
+
# chunk and the last ASCII chunk, and the IO's position afterwards is unspecified.
|
|
194
|
+
# @param max_pixels [Integer, nil] limit on width × height, checked against the header before any
|
|
195
|
+
# pixel data is read or converted
|
|
196
|
+
# @return [Decoded]
|
|
197
|
+
# @raise [UnsupportedInput] if the data is not a PNM image, or is malformed or truncated
|
|
198
|
+
# @raise [LimitExceeded] if width × height exceeds +max_pixels+ (+limit: :max_pixels+)
|
|
199
|
+
# @example
|
|
200
|
+
# decoded = File.open("page.pgm", "rb") { |file| ZXingFFI::Pnm.decode(file, max_pixels: 64_000_000) }
|
|
201
|
+
# decoded.pixels.bytesize == decoded.width * decoded.height # => true
|
|
202
|
+
def decode(data, max_pixels: nil)
|
|
203
|
+
if io?(data)
|
|
204
|
+
header, buffer = read_io_header(data)
|
|
205
|
+
check_max_pixels(header, max_pixels)
|
|
206
|
+
pixels = luminance(buffer, header, data)
|
|
207
|
+
else
|
|
208
|
+
buffer = binary(data)
|
|
209
|
+
header = parse_header(buffer, final: true)
|
|
210
|
+
check_max_pixels(header, max_pixels)
|
|
211
|
+
pixels = luminance(buffer, header, nil)
|
|
212
|
+
end
|
|
213
|
+
Decoded.new(width: header.width, height: header.height, pixels: pixels)
|
|
214
|
+
end
|
|
215
|
+
|
|
216
|
+
# Serializes 8-bit luminance as a binary PGM (+P5+, maxval 255).
|
|
217
|
+
#
|
|
218
|
+
# @param pixels [String] width × height bytes, row-major
|
|
219
|
+
# @param width [Integer]
|
|
220
|
+
# @param height [Integer]
|
|
221
|
+
# @return [String] BINARY +"P5\n<width> <height>\n255\n"+ followed by the pixels
|
|
222
|
+
# @raise [ArgumentError] if the dimensions are not positive Integers or do not match +pixels.bytesize+
|
|
223
|
+
# @raise [TypeError] if +pixels+ is not a String
|
|
224
|
+
def encode_pgm(pixels, width, height)
|
|
225
|
+
raise TypeError, "pixels must be a String, got #{pixels.class}" unless pixels.is_a?(String)
|
|
226
|
+
unless width.is_a?(Integer) && height.is_a?(Integer) && width.positive? && height.positive?
|
|
227
|
+
raise ArgumentError, "width and height must be positive Integers, got #{width.inspect}x#{height.inspect}"
|
|
228
|
+
end
|
|
229
|
+
unless pixels.bytesize == width * height
|
|
230
|
+
raise ArgumentError,
|
|
231
|
+
"expected #{width * height} bytes of pixels for #{width}x#{height}, got #{pixels.bytesize}"
|
|
232
|
+
end
|
|
233
|
+
|
|
234
|
+
header = "P5\n#{width} #{height}\n255\n"
|
|
235
|
+
String.new(capacity: header.bytesize + pixels.bytesize, encoding: BINARY) << header << pixels.b
|
|
236
|
+
end
|
|
237
|
+
|
|
238
|
+
private
|
|
239
|
+
|
|
240
|
+
def io?(data)
|
|
241
|
+
!data.is_a?(String) && data.respond_to?(:read)
|
|
242
|
+
end
|
|
243
|
+
|
|
244
|
+
def binary(data)
|
|
245
|
+
raise TypeError, "expected PNM data as a String or an IO, got #{data.class}" unless data.is_a?(String)
|
|
246
|
+
|
|
247
|
+
(data.encoding == BINARY) ? data : data.b
|
|
248
|
+
end
|
|
249
|
+
|
|
250
|
+
# --- Header ------------------------------------------------------------------------------------------
|
|
251
|
+
|
|
252
|
+
# Parses the header at the start of +buffer+ (BINARY). Unless +final+, the buffer may be a prefix of the
|
|
253
|
+
# data and running out of bytes raises NeedMoreData instead of UnsupportedInput.
|
|
254
|
+
def parse_header(buffer, final:)
|
|
255
|
+
kind = parse_magic(buffer, final)
|
|
256
|
+
names = BITMAP_KINDS.include?(kind) ? %i[width height] : %i[width height maxval]
|
|
257
|
+
values = {maxval: 1}
|
|
258
|
+
position = 2
|
|
259
|
+
names.each_with_index do |name, index|
|
|
260
|
+
position = skip_separators(buffer, position) || truncated!("truncated PNM header: missing #{name}", final)
|
|
261
|
+
stop = buffer.byteindex(NON_DIGIT, position) || buffer.bytesize
|
|
262
|
+
malformed!("expected #{name}", buffer, position) if stop == position
|
|
263
|
+
# A token must be followed by a delimiter, otherwise more digits could follow.
|
|
264
|
+
if stop == buffer.bytesize
|
|
265
|
+
missing = names[index + 1] || "the whitespace byte that ends the header"
|
|
266
|
+
truncated!("truncated PNM header: missing #{missing}", final)
|
|
267
|
+
end
|
|
268
|
+
malformed!("expected whitespace after #{name}", buffer, stop) unless DELIMITERS.include?(buffer.getbyte(stop))
|
|
269
|
+
values[name] = check_range(name, buffer.byteslice(position, stop - position))
|
|
270
|
+
position = stop
|
|
271
|
+
end
|
|
272
|
+
Header.new(kind: kind, offset: raster_offset(buffer, position, final), **values)
|
|
273
|
+
end
|
|
274
|
+
|
|
275
|
+
def parse_magic(buffer, final)
|
|
276
|
+
magic = buffer.byteslice(0, 2)
|
|
277
|
+
truncated!("not a PNM image: no data", final) if magic.empty?
|
|
278
|
+
unless magic.start_with?("P")
|
|
279
|
+
raise UnsupportedInput, "not a PNM image: expected magic number P1-P6, got #{magic.inspect}"
|
|
280
|
+
end
|
|
281
|
+
truncated!("truncated PNM header: incomplete magic number #{magic.inspect}", final) if magic.bytesize < 2
|
|
282
|
+
|
|
283
|
+
kind = MAGIC[magic]
|
|
284
|
+
unless kind
|
|
285
|
+
raise UnsupportedInput, "PAM images (P7) are not supported" if magic == "P7"
|
|
286
|
+
raise UnsupportedInput, "not a PNM image: expected magic number P1-P6, got #{magic.inspect}"
|
|
287
|
+
end
|
|
288
|
+
separator = buffer.getbyte(2)
|
|
289
|
+
truncated!("truncated PNM header: missing width", final) unless separator
|
|
290
|
+
malformed!("expected whitespace after magic number #{magic}", buffer, 2) unless DELIMITERS.include?(separator)
|
|
291
|
+
kind
|
|
292
|
+
end
|
|
293
|
+
|
|
294
|
+
# Position of the next token after whitespace and comments, or nil if the data ends first.
|
|
295
|
+
def skip_separators(buffer, position)
|
|
296
|
+
loop do
|
|
297
|
+
position = buffer.byteindex(NON_WHITESPACE, position)
|
|
298
|
+
return nil if position.nil?
|
|
299
|
+
return position unless buffer.getbyte(position) == HASH
|
|
300
|
+
|
|
301
|
+
position = buffer.byteindex(LINE_END, position)
|
|
302
|
+
return nil if position.nil?
|
|
303
|
+
end
|
|
304
|
+
end
|
|
305
|
+
|
|
306
|
+
# +position+ is at the byte after the last token: a whitespace byte or a comment whose line break then
|
|
307
|
+
# ends the header.
|
|
308
|
+
def raster_offset(buffer, position, final)
|
|
309
|
+
return position + 1 unless buffer.getbyte(position) == HASH
|
|
310
|
+
|
|
311
|
+
line_end = buffer.byteindex(LINE_END, position) ||
|
|
312
|
+
truncated!("truncated PNM header: unterminated comment after the last header value", final)
|
|
313
|
+
line_end + 1
|
|
314
|
+
end
|
|
315
|
+
|
|
316
|
+
def check_range(name, digits)
|
|
317
|
+
max = (name == :maxval) ? MAX_MAXVAL : MAX_DIMENSION
|
|
318
|
+
significant = digits.sub(LEADING_ZEROS, "")
|
|
319
|
+
# More than 10 significant digits cannot be in range; don't build a huge Integer for them.
|
|
320
|
+
value = significant.to_i if significant.bytesize <= 10
|
|
321
|
+
return value if value&.between?(1, max)
|
|
322
|
+
|
|
323
|
+
shown = (significant.bytesize > 20) ? "#{significant.byteslice(0, 20)}..." : significant
|
|
324
|
+
raise UnsupportedInput, "invalid PNM header: #{name} must be 1..#{max}, got #{shown}"
|
|
325
|
+
end
|
|
326
|
+
|
|
327
|
+
def truncated!(message, final)
|
|
328
|
+
raise NeedMoreData unless final
|
|
329
|
+
|
|
330
|
+
raise UnsupportedInput, message
|
|
331
|
+
end
|
|
332
|
+
|
|
333
|
+
def malformed!(expectation, buffer, position)
|
|
334
|
+
found = buffer.byteslice(position, 10)
|
|
335
|
+
found = found.empty? ? "end of data" : found.inspect
|
|
336
|
+
raise UnsupportedInput, "malformed PNM header: #{expectation}, got #{found}"
|
|
337
|
+
end
|
|
338
|
+
|
|
339
|
+
def check_max_pixels(header, max_pixels)
|
|
340
|
+
return if max_pixels.nil? || header.pixel_count <= max_pixels
|
|
341
|
+
|
|
342
|
+
raise LimitExceeded.new(
|
|
343
|
+
"PNM image is #{header.width}x#{header.height} (#{header.pixel_count} pixels), " \
|
|
344
|
+
"more than max_pixels (#{max_pixels})",
|
|
345
|
+
limit: :max_pixels, value: header.pixel_count
|
|
346
|
+
)
|
|
347
|
+
end
|
|
348
|
+
|
|
349
|
+
# --- IO ----------------------------------------------------------------------------------------------
|
|
350
|
+
|
|
351
|
+
def io_position(io)
|
|
352
|
+
return nil unless io.respond_to?(:pos) && io.respond_to?(:seek)
|
|
353
|
+
|
|
354
|
+
io.pos
|
|
355
|
+
rescue SystemCallError, IOError # pipes, sockets and other unseekable streams
|
|
356
|
+
nil
|
|
357
|
+
end
|
|
358
|
+
|
|
359
|
+
# Reads from +io+ until the header parses. Returns the header and every byte read so far (BINARY).
|
|
360
|
+
def read_io_header(io)
|
|
361
|
+
buffer = String.new(encoding: BINARY)
|
|
362
|
+
chunk_size = HEADER_CHUNK
|
|
363
|
+
loop do
|
|
364
|
+
chunk = io.read(chunk_size)
|
|
365
|
+
buffer << chunk.b if chunk
|
|
366
|
+
final = chunk.nil? || chunk.bytesize < chunk_size # IO#read(n) only returns less at EOF
|
|
367
|
+
begin
|
|
368
|
+
return [parse_header(buffer, final: final), buffer]
|
|
369
|
+
rescue NeedMoreData
|
|
370
|
+
if buffer.bytesize >= MAX_HEADER_BYTES
|
|
371
|
+
raise UnsupportedInput, "PNM header longer than #{MAX_HEADER_BYTES} bytes (long comments?) is not supported"
|
|
372
|
+
end
|
|
373
|
+
|
|
374
|
+
chunk_size = [chunk_size * 2, MAX_HEADER_BYTES - buffer.bytesize].min
|
|
375
|
+
end
|
|
376
|
+
end
|
|
377
|
+
end
|
|
378
|
+
|
|
379
|
+
# --- Raster ------------------------------------------------------------------------------------------
|
|
380
|
+
|
|
381
|
+
# Converts the raster to 8-bit luminance. For String input +io+ is nil and +buffer+ is the whole data;
|
|
382
|
+
# for IO input +buffer+ holds the bytes read with the header and the rest is read from +io+ on demand.
|
|
383
|
+
def luminance(buffer, header, io)
|
|
384
|
+
reader = RasterReader.new(buffer, header.offset, io)
|
|
385
|
+
# ASCII rasters need at least one byte per pixel, so the data bounds the output of String input.
|
|
386
|
+
text_capacity = io ? PREALLOCATION_LIMIT : buffer.bytesize - header.offset
|
|
387
|
+
case header.kind
|
|
388
|
+
when :p1 then plain_bits(reader, header, text_capacity)
|
|
389
|
+
when :p2, :p3 then plain_samples(reader, header, text_capacity)
|
|
390
|
+
else
|
|
391
|
+
return binary_luminance(reader, header, PREALLOCATION_LIMIT) if io
|
|
392
|
+
|
|
393
|
+
available = buffer.bytesize - header.offset
|
|
394
|
+
truncated_raster!(header.raster_bytesize, available, "bytes") if available < header.raster_bytesize
|
|
395
|
+
return gray_luminance(buffer, header) if header.kind == :p5 && header.maxval <= 255
|
|
396
|
+
|
|
397
|
+
binary_luminance(reader, header, header.pixel_count)
|
|
398
|
+
end
|
|
399
|
+
end
|
|
400
|
+
|
|
401
|
+
# 8-bit PGM held in a String, the hot path: a slice of the input (shared when the raster ends the data)
|
|
402
|
+
# or a single String#tr.
|
|
403
|
+
def gray_luminance(buffer, header)
|
|
404
|
+
raster = buffer.byteslice(header.offset, header.raster_bytesize)
|
|
405
|
+
(header.maxval == 255) ? raster : raster.tr(TR_ALL_BYTES, tr_table(header.maxval))
|
|
406
|
+
end
|
|
407
|
+
|
|
408
|
+
# P4/P5/P6 in steps of whole pixels (and of whole rows for P4 when they fit). Output capacity is reserved
|
|
409
|
+
# once the first step shows data, up to +capacity+.
|
|
410
|
+
#
|
|
411
|
+
# Each step's temporaries are released with #clear as soon as they are used: CRuby then frees their
|
|
412
|
+
# buffers at once, so memory stays flat instead of accumulating garbage up to the GC's malloc limit.
|
|
413
|
+
def binary_luminance(reader, header, capacity)
|
|
414
|
+
step_bytes, convert = binary_step(header)
|
|
415
|
+
total = header.raster_bytesize
|
|
416
|
+
remaining = total
|
|
417
|
+
output = nil
|
|
418
|
+
while remaining.positive?
|
|
419
|
+
size = [remaining, step_bytes].min
|
|
420
|
+
chunk = reader.read(size)
|
|
421
|
+
truncated_raster!(total, total - remaining + chunk.bytesize, "bytes") if chunk.bytesize < size
|
|
422
|
+
output ||= String.new(capacity: [header.pixel_count, capacity].min, encoding: BINARY)
|
|
423
|
+
convert.call(chunk, output)
|
|
424
|
+
chunk.clear
|
|
425
|
+
remaining -= size
|
|
426
|
+
end
|
|
427
|
+
output
|
|
428
|
+
end
|
|
429
|
+
|
|
430
|
+
# Bytes per step and the conversion appending one step's luminance to the output.
|
|
431
|
+
def binary_step(header)
|
|
432
|
+
case header.kind
|
|
433
|
+
when :p4
|
|
434
|
+
row_bytes = (header.width + 7) / 8
|
|
435
|
+
step = (row_bytes < BIT_CHUNK) ? BIT_CHUNK / row_bytes * row_bytes : BIT_CHUNK
|
|
436
|
+
[step, bits_converter(header.width, row_bytes)]
|
|
437
|
+
when :p5
|
|
438
|
+
scale = sample_scaler(header.maxval)
|
|
439
|
+
convert = lambda do |chunk, output|
|
|
440
|
+
samples = scale.call(chunk)
|
|
441
|
+
output << samples
|
|
442
|
+
samples.clear
|
|
443
|
+
end
|
|
444
|
+
[(header.maxval > 255) ? 2 * CHUNK_PIXELS : BYTE_CHUNK, convert]
|
|
445
|
+
else
|
|
446
|
+
scale = sample_scaler(header.maxval)
|
|
447
|
+
convert = lambda do |chunk, output|
|
|
448
|
+
samples = scale.call(chunk)
|
|
449
|
+
append_rgb_luminance(samples, output)
|
|
450
|
+
samples.clear
|
|
451
|
+
end
|
|
452
|
+
[((header.maxval > 255) ? 6 : 3) * CHUNK_PIXELS, convert]
|
|
453
|
+
end
|
|
454
|
+
end
|
|
455
|
+
|
|
456
|
+
# Scales binary samples (8- or 16-bit) to 8-bit samples. The result is +raw+ itself (converted in
|
|
457
|
+
# place) for 8-bit input, a new String for 16-bit input.
|
|
458
|
+
def sample_scaler(maxval)
|
|
459
|
+
if maxval == 255
|
|
460
|
+
->(raw) { raw }
|
|
461
|
+
elsif maxval < 255
|
|
462
|
+
table = tr_table(maxval)
|
|
463
|
+
->(raw) { raw.tr!(TR_ALL_BYTES, table) || raw }
|
|
464
|
+
else
|
|
465
|
+
table = scale_table(maxval, 65_536)
|
|
466
|
+
lambda do |raw|
|
|
467
|
+
values = raw.unpack("n*").map! { |v| table[v] }
|
|
468
|
+
samples = values.pack("C*")
|
|
469
|
+
values.clear
|
|
470
|
+
samples
|
|
471
|
+
end
|
|
472
|
+
end
|
|
473
|
+
end
|
|
474
|
+
|
|
475
|
+
# Maps sample values 0...size to 0..255, clamping values above maxval.
|
|
476
|
+
def scale_table(maxval, size)
|
|
477
|
+
half = maxval / 2
|
|
478
|
+
Array.new(size) { |v| (v > maxval) ? 255 : (v * 255 + half) / maxval }
|
|
479
|
+
end
|
|
480
|
+
|
|
481
|
+
# String#tr replacement mapping every byte through scale_table(maxval, 256).
|
|
482
|
+
def tr_table(maxval)
|
|
483
|
+
scale_table(maxval, 256).map { |v| TR_CHARS[v] }.join
|
|
484
|
+
end
|
|
485
|
+
|
|
486
|
+
# Appends the luminance of 8-bit RGB samples to +output+; a trailing incomplete pixel is ignored.
|
|
487
|
+
def append_rgb_luminance(samples, output)
|
|
488
|
+
values = samples.unpack("C*")
|
|
489
|
+
pixels = Array.new(values.size / 3) do |i|
|
|
490
|
+
j = i * 3
|
|
491
|
+
(RED[values[j]] + GREEN[values[j + 1]] + BLUE[values[j + 2]]) >> 10
|
|
492
|
+
end
|
|
493
|
+
pixels.pack("C*", buffer: output)
|
|
494
|
+
values.clear
|
|
495
|
+
pixels.clear
|
|
496
|
+
end
|
|
497
|
+
|
|
498
|
+
# The conversion of P4 steps. Rows are padded to a byte boundary; the padding bits are dropped. A row
|
|
499
|
+
# wider than a step is converted in parts, +column+ counting the bytes of it already converted.
|
|
500
|
+
def bits_converter(width, row_bytes)
|
|
501
|
+
return ->(chunk, output) { append_bit_pixels(chunk.unpack1("B*"), output) } if width == row_bytes * 8
|
|
502
|
+
|
|
503
|
+
row_template = "B#{width}"
|
|
504
|
+
column = 0
|
|
505
|
+
lambda do |chunk, output|
|
|
506
|
+
position = 0
|
|
507
|
+
while position < chunk.bytesize
|
|
508
|
+
size = [row_bytes - column, chunk.bytesize - position].min
|
|
509
|
+
template =
|
|
510
|
+
if size == row_bytes then row_template
|
|
511
|
+
elsif column + size == row_bytes then "B#{width - column * 8}" # the end of a row
|
|
512
|
+
else "B#{size * 8}"
|
|
513
|
+
end
|
|
514
|
+
append_bit_pixels(chunk.unpack1(template, offset: position), output)
|
|
515
|
+
column = (column + size) % row_bytes
|
|
516
|
+
position += size
|
|
517
|
+
end
|
|
518
|
+
end
|
|
519
|
+
end
|
|
520
|
+
|
|
521
|
+
# Appends a String of "0"/"1" as white/black pixels to +output+, consuming it.
|
|
522
|
+
def append_bit_pixels(bits, output)
|
|
523
|
+
bits.tr!("01", BIT_PIXELS)
|
|
524
|
+
output << bits
|
|
525
|
+
bits.clear
|
|
526
|
+
end
|
|
527
|
+
|
|
528
|
+
# P1: one "0" or "1" per pixel, separated by whitespace or not at all. As in a single pass over the
|
|
529
|
+
# whole raster, a short raster is reported as truncated even when it also holds invalid characters.
|
|
530
|
+
def plain_bits(reader, header, capacity)
|
|
531
|
+
count = header.pixel_count
|
|
532
|
+
output = String.new(capacity: [count, capacity].min, encoding: BINARY)
|
|
533
|
+
found = 0
|
|
534
|
+
invalid = nil
|
|
535
|
+
each_plain_piece(reader, whole_tokens: false) do |text|
|
|
536
|
+
bits = text.delete(WHITESPACE)
|
|
537
|
+
text.clear
|
|
538
|
+
found += bits.bytesize
|
|
539
|
+
unless invalid
|
|
540
|
+
wanted = (found > count) ? bits.byteslice(0, bits.bytesize - (found - count)) : bits
|
|
541
|
+
if wanted.count("01") == wanted.bytesize
|
|
542
|
+
append_bit_pixels(wanted, output)
|
|
543
|
+
else
|
|
544
|
+
invalid = wanted.byteslice(wanted.byteindex(NON_BIT), 1)
|
|
545
|
+
end
|
|
546
|
+
end
|
|
547
|
+
bits.clear
|
|
548
|
+
break if found >= count
|
|
549
|
+
end
|
|
550
|
+
truncated_raster!(count, found, "bits") if found < count
|
|
551
|
+
raise UnsupportedInput, "malformed PNM raster: expected 0 or 1 in a P1 image, got #{invalid.inspect}" if invalid
|
|
552
|
+
|
|
553
|
+
output
|
|
554
|
+
end
|
|
555
|
+
|
|
556
|
+
# P2/P3: whitespace-separated decimal samples. Error precedence as in plain_bits.
|
|
557
|
+
def plain_samples(reader, header, capacity)
|
|
558
|
+
channels = (header.kind == :p3) ? 3 : 1
|
|
559
|
+
count = header.pixel_count * channels
|
|
560
|
+
maxval = header.maxval
|
|
561
|
+
table = scale_table(maxval, maxval + 1)
|
|
562
|
+
output = String.new(capacity: [header.pixel_count, capacity].min, encoding: BINARY)
|
|
563
|
+
partial = "".b # P3: scaled samples of a pixel split between two pieces
|
|
564
|
+
found = 0
|
|
565
|
+
invalid = nil
|
|
566
|
+
each_plain_piece(reader, whole_tokens: true) do |text|
|
|
567
|
+
tokens = text.split(" ")
|
|
568
|
+
size = tokens.size
|
|
569
|
+
unless invalid
|
|
570
|
+
tokens.pop(size - (count - found)) if size > count - found # whatever follows the raster
|
|
571
|
+
invalid = first_invalid_sample(text, tokens)
|
|
572
|
+
unless invalid
|
|
573
|
+
tokens.map! do |token|
|
|
574
|
+
value = token.to_i
|
|
575
|
+
(value > maxval) ? 255 : table[value]
|
|
576
|
+
end
|
|
577
|
+
if channels == 1
|
|
578
|
+
tokens.pack("C*", buffer: output)
|
|
579
|
+
else
|
|
580
|
+
tokens.pack("C*", buffer: partial)
|
|
581
|
+
append_rgb_luminance(partial, output) # whole pixels only
|
|
582
|
+
rest = partial.byteslice(partial.bytesize / 3 * 3, 2)
|
|
583
|
+
partial.clear
|
|
584
|
+
partial = rest
|
|
585
|
+
end
|
|
586
|
+
end
|
|
587
|
+
end
|
|
588
|
+
text.clear
|
|
589
|
+
tokens.clear
|
|
590
|
+
found += size
|
|
591
|
+
break if found >= count
|
|
592
|
+
end
|
|
593
|
+
truncated_raster!(count, found, "samples") if found < count
|
|
594
|
+
raise UnsupportedInput, "malformed PNM raster: expected decimal samples, got #{invalid.inspect}" if invalid
|
|
595
|
+
|
|
596
|
+
output
|
|
597
|
+
end
|
|
598
|
+
|
|
599
|
+
# The first 20 bytes of the first token that is not a decimal number, or nil.
|
|
600
|
+
def first_invalid_sample(text, tokens)
|
|
601
|
+
return nil if text.count(NOT_SAMPLE_TEXT).zero? # the usual case: digits and whitespace only
|
|
602
|
+
|
|
603
|
+
tokens.find { |token| token.match?(NON_DIGIT) }&.byteslice(0, 20)
|
|
604
|
+
end
|
|
605
|
+
|
|
606
|
+
# Yields the ASCII raster in pieces of about TEXT_CHUNK bytes with comments blanked out (Netpbm's reader
|
|
607
|
+
# skips them in the raster too). With +whole_tokens+ (samples) no token is split between two pieces.
|
|
608
|
+
def each_plain_piece(reader, whole_tokens:)
|
|
609
|
+
carry = "".b # the start of a token cut by the end of the previous chunk
|
|
610
|
+
in_comment = false
|
|
611
|
+
loop do
|
|
612
|
+
text = reader.read(TEXT_CHUNK)
|
|
613
|
+
last = text.bytesize < TEXT_CHUNK
|
|
614
|
+
in_comment = blank_comments!(text, in_comment)
|
|
615
|
+
if whole_tokens && !last
|
|
616
|
+
cut = text.byterindex(WHITESPACE_BYTE)
|
|
617
|
+
if cut.nil? # the chunk continues a single token
|
|
618
|
+
shorten_sample!(carry << text)
|
|
619
|
+
text.clear
|
|
620
|
+
next
|
|
621
|
+
end
|
|
622
|
+
piece = text.byteslice(0, cut + 1)
|
|
623
|
+
piece = carry << piece unless carry.empty?
|
|
624
|
+
carry = text.byteslice(cut + 1, text.bytesize - cut - 1)
|
|
625
|
+
else
|
|
626
|
+
piece = carry.empty? ? text : carry << text
|
|
627
|
+
carry = "".b
|
|
628
|
+
end
|
|
629
|
+
yield piece
|
|
630
|
+
break if last
|
|
631
|
+
end
|
|
632
|
+
end
|
|
633
|
+
|
|
634
|
+
# Overwrites the comments in +chunk+ with spaces, in place. A comment ends at a line break or at the end
|
|
635
|
+
# of the data, so this is the same as removing it, and unlike String#gsub it allocates nothing per
|
|
636
|
+
# comment (a raster may hold one per pixel). +in_comment+ tells whether the previous chunk ended inside
|
|
637
|
+
# a comment; returns the same for this chunk.
|
|
638
|
+
def blank_comments!(chunk, in_comment)
|
|
639
|
+
start = in_comment ? 0 : chunk.byteindex("#")
|
|
640
|
+
line_feed = carriage_return = -1 # the next LF and CR from +start+ on (chunk.bytesize if none)
|
|
641
|
+
while start
|
|
642
|
+
line_feed = chunk.byteindex("\n", start) || chunk.bytesize if line_feed < start
|
|
643
|
+
carriage_return = chunk.byteindex("\r", start) || chunk.bytesize if carriage_return < start
|
|
644
|
+
stop = (line_feed < carriage_return) ? line_feed : carriage_return
|
|
645
|
+
chunk.bytesplice(start, stop - start, BLANKS, 0, stop - start)
|
|
646
|
+
return true if stop == chunk.bytesize
|
|
647
|
+
|
|
648
|
+
start = chunk.byteindex("#", stop)
|
|
649
|
+
end
|
|
650
|
+
false
|
|
651
|
+
end
|
|
652
|
+
|
|
653
|
+
# Replaces a +token+ longer than LONG_SAMPLE with a short one that plain_samples decodes the same way,
|
|
654
|
+
# whatever follows it: with the same first 20 bytes (shown in error messages) and the same value, or a
|
|
655
|
+
# value above 65535 if that one is (both clamp to 255). Keeps a token cut by many chunk boundaries (a
|
|
656
|
+
# long run of digits, or of NUL bytes after a truncated raster) from growing with the data.
|
|
657
|
+
def shorten_sample!(token)
|
|
658
|
+
return if token.bytesize <= LONG_SAMPLE
|
|
659
|
+
|
|
660
|
+
head = token.byteslice(0, 20)
|
|
661
|
+
if token.match?(NON_DIGIT)
|
|
662
|
+
head << "x" # invalid, whatever follows
|
|
663
|
+
else
|
|
664
|
+
zeros = token.byteindex(NONZERO_DIGIT) || token.bytesize - 1 # a zero keeps its last digit
|
|
665
|
+
significant = token.bytesize - zeros # at most 5 digits only if +head+ is all zeros
|
|
666
|
+
head << ((significant > 5) ? "9999999" : token.byteslice(zeros, significant))
|
|
667
|
+
end
|
|
668
|
+
token.replace(head)
|
|
669
|
+
end
|
|
670
|
+
|
|
671
|
+
def truncated_raster!(expected, got, unit)
|
|
672
|
+
raise UnsupportedInput, "truncated PNM: expected #{expected} #{unit} of pixel data, got #{got}"
|
|
673
|
+
end
|
|
674
|
+
end
|
|
675
|
+
end
|
|
676
|
+
end
|