zxing_ffi 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (41) hide show
  1. checksums.yaml +7 -0
  2. data/CHANGELOG.md +27 -0
  3. data/LICENSE.txt +21 -0
  4. data/README.md +275 -0
  5. data/exe/zxing-scan +6 -0
  6. data/lib/zxing_ffi/barcode.rb +77 -0
  7. data/lib/zxing_ffi/cli.rb +152 -0
  8. data/lib/zxing_ffi/config.rb +119 -0
  9. data/lib/zxing_ffi/dedupe.rb +79 -0
  10. data/lib/zxing_ffi/diagnostics.rb +69 -0
  11. data/lib/zxing_ffi/dpi.rb +103 -0
  12. data/lib/zxing_ffi/errors.rb +69 -0
  13. data/lib/zxing_ffi/formats.rb +207 -0
  14. data/lib/zxing_ffi/geometry.rb +508 -0
  15. data/lib/zxing_ffi/header_probe.rb +98 -0
  16. data/lib/zxing_ffi/image.rb +123 -0
  17. data/lib/zxing_ffi/image_magick.rb +95 -0
  18. data/lib/zxing_ffi/library_defaults.rb +7 -0
  19. data/lib/zxing_ffi/library_loader.rb +176 -0
  20. data/lib/zxing_ffi/loaders/base.rb +155 -0
  21. data/lib/zxing_ffi/loaders/image_magick.rb +159 -0
  22. data/lib/zxing_ffi/loaders/pnm.rb +113 -0
  23. data/lib/zxing_ffi/loaders/poppler.rb +260 -0
  24. data/lib/zxing_ffi/loaders/registry.rb +54 -0
  25. data/lib/zxing_ffi/loaders/vips.rb +332 -0
  26. data/lib/zxing_ffi/loaders.rb +41 -0
  27. data/lib/zxing_ffi/native.rb +237 -0
  28. data/lib/zxing_ffi/pnm.rb +676 -0
  29. data/lib/zxing_ffi/reader.rb +271 -0
  30. data/lib/zxing_ffi/scanner.rb +295 -0
  31. data/lib/zxing_ffi/sniffer.rb +155 -0
  32. data/lib/zxing_ffi/source.rb +77 -0
  33. data/lib/zxing_ffi/strategy.rb +293 -0
  34. data/lib/zxing_ffi/subprocess.rb +416 -0
  35. data/lib/zxing_ffi/transformers/base.rb +69 -0
  36. data/lib/zxing_ffi/transformers/image_magick.rb +72 -0
  37. data/lib/zxing_ffi/transformers/vips.rb +85 -0
  38. data/lib/zxing_ffi/transformers.rb +36 -0
  39. data/lib/zxing_ffi/version.rb +5 -0
  40. data/lib/zxing_ffi.rb +82 -0
  41. metadata +108 -0
@@ -0,0 +1,676 @@
1
+ # frozen_string_literal: true
2
+
3
+ module ZXingFFI
4
+ # Pure-Ruby decoder for the Netpbm formats (PBM, PGM and PPM, both the ASCII "plain" variants +P1+–+P3+ and
5
+ # the binary +P4+–+P6+) producing 8-bit luminance, plus a minimal PGM writer.
6
+ #
7
+ # Besides reading user files it parses the PGM that +pdftoppm -gray+ and ImageMagick's +pgm:-+ write.
8
+ # That common case, a binary PGM with maxval 255, costs a header parse and a slice of the input
9
+ # with no per-pixel work. Other variants are converted with C-speed primitives where Ruby has them
10
+ # (+String#tr+ look-up tables, +unpack+/+pack+); only colour (PPM) input needs a per-pixel Ruby loop.
11
+ #
12
+ # Every variant except 8-bit PGM held in a String is converted in steps of 16K pixels (64 KiB of text for
13
+ # the ASCII variants) appended to a single output String, and each step's temporaries are freed before
14
+ # the next, so the memory needed beyond the input and the output stays under a few MB whatever the image
15
+ # size or content. An IO is read the same way, in chunks, and not past the raster (see {decode}).
16
+ #
17
+ # Conversions applied by {decode}:
18
+ # - samples are scaled to 0..255 with rounding, +(v * 255 + maxval / 2) / maxval+, so 16-bit input is
19
+ # scaled, never truncated; samples above maxval (invalid) are clamped to 255;
20
+ # - PBM bits become 255 (bit 0, white) and 0 (bit 1, black);
21
+ # - PPM pixels become zxing-cpp's luminance +(306 * r + 601 * g + 117 * b + 0x200) >> 10+ of the
22
+ # scaled channels.
23
+ #
24
+ # Header syntax follows the Netpbm reference implementation: tokens are separated by any mix of whitespace
25
+ # (space, TAB, LF, VT, FF, CR) and +#+ comments running to the end of the line, and exactly one whitespace
26
+ # byte ends the header (so +"255\r\n"+ leaves the LF as the first raster byte). A comment directly after
27
+ # the last token is allowed; the line break ending it is then that byte.
28
+ #
29
+ # @see https://netpbm.sourceforge.net/doc/pnm.html
30
+ module Pnm
31
+ # A parsed PNM header.
32
+ #
33
+ # @!attribute [r] kind
34
+ # @return [Symbol] +:p1+ .. +:p6+
35
+ # @!attribute [r] width
36
+ # @return [Integer]
37
+ # @!attribute [r] height
38
+ # @return [Integer]
39
+ # @!attribute [r] maxval
40
+ # @return [Integer] largest sample value (1..65535); always 1 for bitmaps (+P1+, +P4+)
41
+ # @!attribute [r] offset
42
+ # @return [Integer] byte index at which the raster starts
43
+ Header = Data.define(:kind, :width, :height, :maxval, :offset) do
44
+ # @return [Integer] width × height
45
+ def pixel_count
46
+ width * height
47
+ end
48
+
49
+ # Size of a binary raster (+P4+–+P6+) in bytes; +nil+ for the ASCII variants, whose size varies.
50
+ # @return [Integer, nil]
51
+ def raster_bytesize
52
+ sample_bytes = (maxval > 255) ? 2 : 1
53
+ case kind
54
+ when :p4 then (width + 7) / 8 * height
55
+ when :p5 then pixel_count * sample_bytes
56
+ when :p6 then pixel_count * 3 * sample_bytes
57
+ end
58
+ end
59
+ end
60
+
61
+ # A decoded image.
62
+ #
63
+ # @!attribute [r] width
64
+ # @return [Integer]
65
+ # @!attribute [r] height
66
+ # @return [Integer]
67
+ # @!attribute [r] pixels
68
+ # @return [String] BINARY, width × height bytes of 8-bit luminance, row-major
69
+ Decoded = Data.define(:width, :height, :pixels)
70
+
71
+ # Largest accepted width or height (libZXing takes +int+ dimensions).
72
+ MAX_DIMENSION = 2**31 - 1
73
+ # Largest maxval allowed by the Netpbm specification.
74
+ MAX_MAXVAL = 65_535
75
+
76
+ # Raised internally when an incomplete header may continue in data not read yet.
77
+ class NeedMoreData < StandardError; end
78
+
79
+ BINARY = Encoding::BINARY
80
+ MAGIC = {"P1" => :p1, "P2" => :p2, "P3" => :p3, "P4" => :p4, "P5" => :p5, "P6" => :p6}.freeze
81
+ BITMAP_KINDS = %i[p1 p4].freeze
82
+ WHITESPACE = " \t\n\v\f\r"
83
+ # Bytes allowed right after a header token: whitespace or the start of a comment.
84
+ DELIMITERS = "#{WHITESPACE}#".bytes.freeze
85
+ HASH = "#".ord
86
+ NON_WHITESPACE = /[^ \t\n\v\f\r]/
87
+ WHITESPACE_BYTE = /[ \t\n\v\f\r]/
88
+ LINE_END = /[\r\n]/
89
+ NON_DIGIT = /[^0-9]/
90
+ LEADING_ZEROS = /\A0+(?=[0-9])/
91
+ NONZERO_DIGIT = /[1-9]/
92
+ NON_BIT = /[^01]/
93
+ # Bytes other than digits and whitespace (String#count syntax).
94
+ NOT_SAMPLE_TEXT = "^0-9 \t\n\v\f\r"
95
+ # First read when parsing a header from an IO; doubled (up to the limit) until the header fits.
96
+ HEADER_CHUNK = 4096
97
+ # Longest header read from an IO, comments included: a longer one raises UnsupportedInput rather than being
98
+ # buffered without bound (an endless comment in an untrusted stream). String input is already in memory.
99
+ MAX_HEADER_BYTES = 1024 * 1024
100
+ # Pixels converted per step where conversion builds Arrays (16-bit P5, P6): < 1 MiB of temporaries.
101
+ # (Larger steps are no faster.)
102
+ CHUNK_PIXELS = 16_384
103
+ # Bytes per step where conversion is byte for byte (8-bit P5 read from an IO).
104
+ BYTE_CHUNK = 256 * 1024
105
+ # Bytes per step for P4 (16Ki pixels): whole rows when a row fits, otherwise parts of one.
106
+ BIT_CHUNK = CHUNK_PIXELS / 8
107
+ # Bytes of text per step for the ASCII variants.
108
+ TEXT_CHUNK = 64 * 1024
109
+ # Spaces that overwrite the comments of an ASCII raster, one chunk of text at a time.
110
+ BLANKS = (" " * TEXT_CHUNK).b.freeze
111
+ # Samples cut by chunk boundaries longer than this are shortened (see shorten_sample!).
112
+ LONG_SAMPLE = 64
113
+ # Output capacity reserved up front when the data has not yet shown that the header's size is real.
114
+ PREALLOCATION_LIMIT = 64 * 1024 * 1024
115
+ # String#tr source covering every byte, and each byte escaped for use in a tr replacement.
116
+ TR_ALL_BYTES = "\x00-\xFF".b.freeze
117
+ TR_CHARS = Array.new(256) { |v| ["-", "\\", "^"].include?(v.chr) ? "\\#{v.chr}".b.freeze : v.chr(BINARY).freeze }.freeze
118
+ # String#tr replacement for "01": bit 0 is white, bit 1 is black.
119
+ BIT_PIXELS = "\xFF\x00".b.freeze
120
+ # zxing-cpp's RGB weights, as tables (the rounding term 0x200 is folded into BLUE).
121
+ RED = Array.new(256) { |v| 306 * v }.freeze
122
+ GREEN = Array.new(256) { |v| 601 * v }.freeze
123
+ BLUE = Array.new(256) { |v| 117 * v + 0x200 }.freeze
124
+
125
+ # Reads a raster sequentially: from the data String itself, or from the bytes buffered while the header
126
+ # was parsed followed by the rest of an IO (read on demand, never beyond what is asked for).
127
+ class RasterReader
128
+ def initialize(buffer, offset, io)
129
+ @buffer = buffer
130
+ @position = offset
131
+ @io = io
132
+ end
133
+
134
+ # @return [String] BINARY, owned by the caller (who may clear it); +size+ bytes, fewer only at the end
135
+ # of the data
136
+ def read(size)
137
+ chunk = @buffer.byteslice(@position, size)
138
+ @position += chunk.bytesize
139
+ return chunk if @io.nil? || chunk.bytesize == size
140
+
141
+ while chunk.bytesize < size
142
+ more = @io.read(size - chunk.bytesize)
143
+ break if more.nil? || more.empty?
144
+
145
+ more = more.b if more.frozen? || more.encoding != Encoding::BINARY
146
+ if chunk.empty?
147
+ chunk = more
148
+ else
149
+ chunk << more
150
+ more.clear
151
+ end
152
+ end
153
+ chunk
154
+ end
155
+ end
156
+
157
+ private_constant :NeedMoreData, :BINARY, :MAGIC, :BITMAP_KINDS, :WHITESPACE, :DELIMITERS, :HASH,
158
+ :NON_WHITESPACE, :WHITESPACE_BYTE, :LINE_END, :NON_DIGIT, :LEADING_ZEROS, :NONZERO_DIGIT, :NON_BIT,
159
+ :NOT_SAMPLE_TEXT, :HEADER_CHUNK, :CHUNK_PIXELS, :BYTE_CHUNK, :BIT_CHUNK,
160
+ :TEXT_CHUNK, :BLANKS, :LONG_SAMPLE, :PREALLOCATION_LIMIT, :TR_ALL_BYTES, :TR_CHARS, :BIT_PIXELS, :RED,
161
+ :GREEN, :BLUE, :RasterReader
162
+
163
+ class << self
164
+ # Parses the header of a PNM image without looking at its raster.
165
+ #
166
+ # @param data [String, IO] the image bytes (any encoding; always treated as bytes), or an IO positioned
167
+ # at the start of the image. Only the bytes the header needs are read from an IO, and a seekable IO is
168
+ # moved back to where it was; +offset+ is then relative to that position.
169
+ # @return [Header]
170
+ # @raise [UnsupportedInput] if the data is not a PNM image (P7/PAM included) or its header is malformed
171
+ # or truncated
172
+ # @example
173
+ # ZXingFFI::Pnm.read_header("P5\n640 480\n255\n...") # => #<data Header kind=:p5, width=640, ...>
174
+ def read_header(data)
175
+ return parse_header(binary(data), final: true) unless io?(data)
176
+
177
+ position = io_position(data)
178
+ begin
179
+ read_io_header(data).first
180
+ ensure
181
+ data.seek(position) if position
182
+ end
183
+ end
184
+
185
+ # Decodes a PNM image to 8-bit luminance.
186
+ #
187
+ # Only the first image of a multi-image file is decoded: bytes after its raster are ignored.
188
+ #
189
+ # @param data [String, IO] the image bytes (any encoding; always treated as bytes), or an IO positioned
190
+ # at the start of the image. An IO is read in bounded chunks, never all at once: the header 4 KiB at a
191
+ # time (doubling for long headers), then the raster, exactly up to its end for the binary variants
192
+ # and in 64 KiB chunks for the ASCII ones. So what follows the image is read only by the first header
193
+ # chunk and the last ASCII chunk, and the IO's position afterwards is unspecified.
194
+ # @param max_pixels [Integer, nil] limit on width × height, checked against the header before any
195
+ # pixel data is read or converted
196
+ # @return [Decoded]
197
+ # @raise [UnsupportedInput] if the data is not a PNM image, or is malformed or truncated
198
+ # @raise [LimitExceeded] if width × height exceeds +max_pixels+ (+limit: :max_pixels+)
199
+ # @example
200
+ # decoded = File.open("page.pgm", "rb") { |file| ZXingFFI::Pnm.decode(file, max_pixels: 64_000_000) }
201
+ # decoded.pixels.bytesize == decoded.width * decoded.height # => true
202
+ def decode(data, max_pixels: nil)
203
+ if io?(data)
204
+ header, buffer = read_io_header(data)
205
+ check_max_pixels(header, max_pixels)
206
+ pixels = luminance(buffer, header, data)
207
+ else
208
+ buffer = binary(data)
209
+ header = parse_header(buffer, final: true)
210
+ check_max_pixels(header, max_pixels)
211
+ pixels = luminance(buffer, header, nil)
212
+ end
213
+ Decoded.new(width: header.width, height: header.height, pixels: pixels)
214
+ end
215
+
216
+ # Serializes 8-bit luminance as a binary PGM (+P5+, maxval 255).
217
+ #
218
+ # @param pixels [String] width × height bytes, row-major
219
+ # @param width [Integer]
220
+ # @param height [Integer]
221
+ # @return [String] BINARY +"P5\n<width> <height>\n255\n"+ followed by the pixels
222
+ # @raise [ArgumentError] if the dimensions are not positive Integers or do not match +pixels.bytesize+
223
+ # @raise [TypeError] if +pixels+ is not a String
224
+ def encode_pgm(pixels, width, height)
225
+ raise TypeError, "pixels must be a String, got #{pixels.class}" unless pixels.is_a?(String)
226
+ unless width.is_a?(Integer) && height.is_a?(Integer) && width.positive? && height.positive?
227
+ raise ArgumentError, "width and height must be positive Integers, got #{width.inspect}x#{height.inspect}"
228
+ end
229
+ unless pixels.bytesize == width * height
230
+ raise ArgumentError,
231
+ "expected #{width * height} bytes of pixels for #{width}x#{height}, got #{pixels.bytesize}"
232
+ end
233
+
234
+ header = "P5\n#{width} #{height}\n255\n"
235
+ String.new(capacity: header.bytesize + pixels.bytesize, encoding: BINARY) << header << pixels.b
236
+ end
237
+
238
+ private
239
+
240
+ def io?(data)
241
+ !data.is_a?(String) && data.respond_to?(:read)
242
+ end
243
+
244
+ def binary(data)
245
+ raise TypeError, "expected PNM data as a String or an IO, got #{data.class}" unless data.is_a?(String)
246
+
247
+ (data.encoding == BINARY) ? data : data.b
248
+ end
249
+
250
+ # --- Header ------------------------------------------------------------------------------------------
251
+
252
+ # Parses the header at the start of +buffer+ (BINARY). Unless +final+, the buffer may be a prefix of the
253
+ # data and running out of bytes raises NeedMoreData instead of UnsupportedInput.
254
+ def parse_header(buffer, final:)
255
+ kind = parse_magic(buffer, final)
256
+ names = BITMAP_KINDS.include?(kind) ? %i[width height] : %i[width height maxval]
257
+ values = {maxval: 1}
258
+ position = 2
259
+ names.each_with_index do |name, index|
260
+ position = skip_separators(buffer, position) || truncated!("truncated PNM header: missing #{name}", final)
261
+ stop = buffer.byteindex(NON_DIGIT, position) || buffer.bytesize
262
+ malformed!("expected #{name}", buffer, position) if stop == position
263
+ # A token must be followed by a delimiter, otherwise more digits could follow.
264
+ if stop == buffer.bytesize
265
+ missing = names[index + 1] || "the whitespace byte that ends the header"
266
+ truncated!("truncated PNM header: missing #{missing}", final)
267
+ end
268
+ malformed!("expected whitespace after #{name}", buffer, stop) unless DELIMITERS.include?(buffer.getbyte(stop))
269
+ values[name] = check_range(name, buffer.byteslice(position, stop - position))
270
+ position = stop
271
+ end
272
+ Header.new(kind: kind, offset: raster_offset(buffer, position, final), **values)
273
+ end
274
+
275
+ def parse_magic(buffer, final)
276
+ magic = buffer.byteslice(0, 2)
277
+ truncated!("not a PNM image: no data", final) if magic.empty?
278
+ unless magic.start_with?("P")
279
+ raise UnsupportedInput, "not a PNM image: expected magic number P1-P6, got #{magic.inspect}"
280
+ end
281
+ truncated!("truncated PNM header: incomplete magic number #{magic.inspect}", final) if magic.bytesize < 2
282
+
283
+ kind = MAGIC[magic]
284
+ unless kind
285
+ raise UnsupportedInput, "PAM images (P7) are not supported" if magic == "P7"
286
+ raise UnsupportedInput, "not a PNM image: expected magic number P1-P6, got #{magic.inspect}"
287
+ end
288
+ separator = buffer.getbyte(2)
289
+ truncated!("truncated PNM header: missing width", final) unless separator
290
+ malformed!("expected whitespace after magic number #{magic}", buffer, 2) unless DELIMITERS.include?(separator)
291
+ kind
292
+ end
293
+
294
+ # Position of the next token after whitespace and comments, or nil if the data ends first.
295
+ def skip_separators(buffer, position)
296
+ loop do
297
+ position = buffer.byteindex(NON_WHITESPACE, position)
298
+ return nil if position.nil?
299
+ return position unless buffer.getbyte(position) == HASH
300
+
301
+ position = buffer.byteindex(LINE_END, position)
302
+ return nil if position.nil?
303
+ end
304
+ end
305
+
306
+ # +position+ is at the byte after the last token: a whitespace byte or a comment whose line break then
307
+ # ends the header.
308
+ def raster_offset(buffer, position, final)
309
+ return position + 1 unless buffer.getbyte(position) == HASH
310
+
311
+ line_end = buffer.byteindex(LINE_END, position) ||
312
+ truncated!("truncated PNM header: unterminated comment after the last header value", final)
313
+ line_end + 1
314
+ end
315
+
316
+ def check_range(name, digits)
317
+ max = (name == :maxval) ? MAX_MAXVAL : MAX_DIMENSION
318
+ significant = digits.sub(LEADING_ZEROS, "")
319
+ # More than 10 significant digits cannot be in range; don't build a huge Integer for them.
320
+ value = significant.to_i if significant.bytesize <= 10
321
+ return value if value&.between?(1, max)
322
+
323
+ shown = (significant.bytesize > 20) ? "#{significant.byteslice(0, 20)}..." : significant
324
+ raise UnsupportedInput, "invalid PNM header: #{name} must be 1..#{max}, got #{shown}"
325
+ end
326
+
327
+ def truncated!(message, final)
328
+ raise NeedMoreData unless final
329
+
330
+ raise UnsupportedInput, message
331
+ end
332
+
333
+ def malformed!(expectation, buffer, position)
334
+ found = buffer.byteslice(position, 10)
335
+ found = found.empty? ? "end of data" : found.inspect
336
+ raise UnsupportedInput, "malformed PNM header: #{expectation}, got #{found}"
337
+ end
338
+
339
+ def check_max_pixels(header, max_pixels)
340
+ return if max_pixels.nil? || header.pixel_count <= max_pixels
341
+
342
+ raise LimitExceeded.new(
343
+ "PNM image is #{header.width}x#{header.height} (#{header.pixel_count} pixels), " \
344
+ "more than max_pixels (#{max_pixels})",
345
+ limit: :max_pixels, value: header.pixel_count
346
+ )
347
+ end
348
+
349
+ # --- IO ----------------------------------------------------------------------------------------------
350
+
351
+ def io_position(io)
352
+ return nil unless io.respond_to?(:pos) && io.respond_to?(:seek)
353
+
354
+ io.pos
355
+ rescue SystemCallError, IOError # pipes, sockets and other unseekable streams
356
+ nil
357
+ end
358
+
359
+ # Reads from +io+ until the header parses. Returns the header and every byte read so far (BINARY).
360
+ def read_io_header(io)
361
+ buffer = String.new(encoding: BINARY)
362
+ chunk_size = HEADER_CHUNK
363
+ loop do
364
+ chunk = io.read(chunk_size)
365
+ buffer << chunk.b if chunk
366
+ final = chunk.nil? || chunk.bytesize < chunk_size # IO#read(n) only returns less at EOF
367
+ begin
368
+ return [parse_header(buffer, final: final), buffer]
369
+ rescue NeedMoreData
370
+ if buffer.bytesize >= MAX_HEADER_BYTES
371
+ raise UnsupportedInput, "PNM header longer than #{MAX_HEADER_BYTES} bytes (long comments?) is not supported"
372
+ end
373
+
374
+ chunk_size = [chunk_size * 2, MAX_HEADER_BYTES - buffer.bytesize].min
375
+ end
376
+ end
377
+ end
378
+
379
+ # --- Raster ------------------------------------------------------------------------------------------
380
+
381
+ # Converts the raster to 8-bit luminance. For String input +io+ is nil and +buffer+ is the whole data;
382
+ # for IO input +buffer+ holds the bytes read with the header and the rest is read from +io+ on demand.
383
+ def luminance(buffer, header, io)
384
+ reader = RasterReader.new(buffer, header.offset, io)
385
+ # ASCII rasters need at least one byte per pixel, so the data bounds the output of String input.
386
+ text_capacity = io ? PREALLOCATION_LIMIT : buffer.bytesize - header.offset
387
+ case header.kind
388
+ when :p1 then plain_bits(reader, header, text_capacity)
389
+ when :p2, :p3 then plain_samples(reader, header, text_capacity)
390
+ else
391
+ return binary_luminance(reader, header, PREALLOCATION_LIMIT) if io
392
+
393
+ available = buffer.bytesize - header.offset
394
+ truncated_raster!(header.raster_bytesize, available, "bytes") if available < header.raster_bytesize
395
+ return gray_luminance(buffer, header) if header.kind == :p5 && header.maxval <= 255
396
+
397
+ binary_luminance(reader, header, header.pixel_count)
398
+ end
399
+ end
400
+
401
+ # 8-bit PGM held in a String, the hot path: a slice of the input (shared when the raster ends the data)
402
+ # or a single String#tr.
403
+ def gray_luminance(buffer, header)
404
+ raster = buffer.byteslice(header.offset, header.raster_bytesize)
405
+ (header.maxval == 255) ? raster : raster.tr(TR_ALL_BYTES, tr_table(header.maxval))
406
+ end
407
+
408
+ # P4/P5/P6 in steps of whole pixels (and of whole rows for P4 when they fit). Output capacity is reserved
409
+ # once the first step shows data, up to +capacity+.
410
+ #
411
+ # Each step's temporaries are released with #clear as soon as they are used: CRuby then frees their
412
+ # buffers at once, so memory stays flat instead of accumulating garbage up to the GC's malloc limit.
413
+ def binary_luminance(reader, header, capacity)
414
+ step_bytes, convert = binary_step(header)
415
+ total = header.raster_bytesize
416
+ remaining = total
417
+ output = nil
418
+ while remaining.positive?
419
+ size = [remaining, step_bytes].min
420
+ chunk = reader.read(size)
421
+ truncated_raster!(total, total - remaining + chunk.bytesize, "bytes") if chunk.bytesize < size
422
+ output ||= String.new(capacity: [header.pixel_count, capacity].min, encoding: BINARY)
423
+ convert.call(chunk, output)
424
+ chunk.clear
425
+ remaining -= size
426
+ end
427
+ output
428
+ end
429
+
430
+ # Bytes per step and the conversion appending one step's luminance to the output.
431
+ def binary_step(header)
432
+ case header.kind
433
+ when :p4
434
+ row_bytes = (header.width + 7) / 8
435
+ step = (row_bytes < BIT_CHUNK) ? BIT_CHUNK / row_bytes * row_bytes : BIT_CHUNK
436
+ [step, bits_converter(header.width, row_bytes)]
437
+ when :p5
438
+ scale = sample_scaler(header.maxval)
439
+ convert = lambda do |chunk, output|
440
+ samples = scale.call(chunk)
441
+ output << samples
442
+ samples.clear
443
+ end
444
+ [(header.maxval > 255) ? 2 * CHUNK_PIXELS : BYTE_CHUNK, convert]
445
+ else
446
+ scale = sample_scaler(header.maxval)
447
+ convert = lambda do |chunk, output|
448
+ samples = scale.call(chunk)
449
+ append_rgb_luminance(samples, output)
450
+ samples.clear
451
+ end
452
+ [((header.maxval > 255) ? 6 : 3) * CHUNK_PIXELS, convert]
453
+ end
454
+ end
455
+
456
+ # Scales binary samples (8- or 16-bit) to 8-bit samples. The result is +raw+ itself (converted in
457
+ # place) for 8-bit input, a new String for 16-bit input.
458
+ def sample_scaler(maxval)
459
+ if maxval == 255
460
+ ->(raw) { raw }
461
+ elsif maxval < 255
462
+ table = tr_table(maxval)
463
+ ->(raw) { raw.tr!(TR_ALL_BYTES, table) || raw }
464
+ else
465
+ table = scale_table(maxval, 65_536)
466
+ lambda do |raw|
467
+ values = raw.unpack("n*").map! { |v| table[v] }
468
+ samples = values.pack("C*")
469
+ values.clear
470
+ samples
471
+ end
472
+ end
473
+ end
474
+
475
+ # Maps sample values 0...size to 0..255, clamping values above maxval.
476
+ def scale_table(maxval, size)
477
+ half = maxval / 2
478
+ Array.new(size) { |v| (v > maxval) ? 255 : (v * 255 + half) / maxval }
479
+ end
480
+
481
+ # String#tr replacement mapping every byte through scale_table(maxval, 256).
482
+ def tr_table(maxval)
483
+ scale_table(maxval, 256).map { |v| TR_CHARS[v] }.join
484
+ end
485
+
486
+ # Appends the luminance of 8-bit RGB samples to +output+; a trailing incomplete pixel is ignored.
487
+ def append_rgb_luminance(samples, output)
488
+ values = samples.unpack("C*")
489
+ pixels = Array.new(values.size / 3) do |i|
490
+ j = i * 3
491
+ (RED[values[j]] + GREEN[values[j + 1]] + BLUE[values[j + 2]]) >> 10
492
+ end
493
+ pixels.pack("C*", buffer: output)
494
+ values.clear
495
+ pixels.clear
496
+ end
497
+
498
+ # The conversion of P4 steps. Rows are padded to a byte boundary; the padding bits are dropped. A row
499
+ # wider than a step is converted in parts, +column+ counting the bytes of it already converted.
500
+ def bits_converter(width, row_bytes)
501
+ return ->(chunk, output) { append_bit_pixels(chunk.unpack1("B*"), output) } if width == row_bytes * 8
502
+
503
+ row_template = "B#{width}"
504
+ column = 0
505
+ lambda do |chunk, output|
506
+ position = 0
507
+ while position < chunk.bytesize
508
+ size = [row_bytes - column, chunk.bytesize - position].min
509
+ template =
510
+ if size == row_bytes then row_template
511
+ elsif column + size == row_bytes then "B#{width - column * 8}" # the end of a row
512
+ else "B#{size * 8}"
513
+ end
514
+ append_bit_pixels(chunk.unpack1(template, offset: position), output)
515
+ column = (column + size) % row_bytes
516
+ position += size
517
+ end
518
+ end
519
+ end
520
+
521
+ # Appends a String of "0"/"1" as white/black pixels to +output+, consuming it.
522
+ def append_bit_pixels(bits, output)
523
+ bits.tr!("01", BIT_PIXELS)
524
+ output << bits
525
+ bits.clear
526
+ end
527
+
528
+ # P1: one "0" or "1" per pixel, separated by whitespace or not at all. As in a single pass over the
529
+ # whole raster, a short raster is reported as truncated even when it also holds invalid characters.
530
+ def plain_bits(reader, header, capacity)
531
+ count = header.pixel_count
532
+ output = String.new(capacity: [count, capacity].min, encoding: BINARY)
533
+ found = 0
534
+ invalid = nil
535
+ each_plain_piece(reader, whole_tokens: false) do |text|
536
+ bits = text.delete(WHITESPACE)
537
+ text.clear
538
+ found += bits.bytesize
539
+ unless invalid
540
+ wanted = (found > count) ? bits.byteslice(0, bits.bytesize - (found - count)) : bits
541
+ if wanted.count("01") == wanted.bytesize
542
+ append_bit_pixels(wanted, output)
543
+ else
544
+ invalid = wanted.byteslice(wanted.byteindex(NON_BIT), 1)
545
+ end
546
+ end
547
+ bits.clear
548
+ break if found >= count
549
+ end
550
+ truncated_raster!(count, found, "bits") if found < count
551
+ raise UnsupportedInput, "malformed PNM raster: expected 0 or 1 in a P1 image, got #{invalid.inspect}" if invalid
552
+
553
+ output
554
+ end
555
+
556
+ # P2/P3: whitespace-separated decimal samples. Error precedence as in plain_bits.
557
+ def plain_samples(reader, header, capacity)
558
+ channels = (header.kind == :p3) ? 3 : 1
559
+ count = header.pixel_count * channels
560
+ maxval = header.maxval
561
+ table = scale_table(maxval, maxval + 1)
562
+ output = String.new(capacity: [header.pixel_count, capacity].min, encoding: BINARY)
563
+ partial = "".b # P3: scaled samples of a pixel split between two pieces
564
+ found = 0
565
+ invalid = nil
566
+ each_plain_piece(reader, whole_tokens: true) do |text|
567
+ tokens = text.split(" ")
568
+ size = tokens.size
569
+ unless invalid
570
+ tokens.pop(size - (count - found)) if size > count - found # whatever follows the raster
571
+ invalid = first_invalid_sample(text, tokens)
572
+ unless invalid
573
+ tokens.map! do |token|
574
+ value = token.to_i
575
+ (value > maxval) ? 255 : table[value]
576
+ end
577
+ if channels == 1
578
+ tokens.pack("C*", buffer: output)
579
+ else
580
+ tokens.pack("C*", buffer: partial)
581
+ append_rgb_luminance(partial, output) # whole pixels only
582
+ rest = partial.byteslice(partial.bytesize / 3 * 3, 2)
583
+ partial.clear
584
+ partial = rest
585
+ end
586
+ end
587
+ end
588
+ text.clear
589
+ tokens.clear
590
+ found += size
591
+ break if found >= count
592
+ end
593
+ truncated_raster!(count, found, "samples") if found < count
594
+ raise UnsupportedInput, "malformed PNM raster: expected decimal samples, got #{invalid.inspect}" if invalid
595
+
596
+ output
597
+ end
598
+
599
+ # The first 20 bytes of the first token that is not a decimal number, or nil.
600
+ def first_invalid_sample(text, tokens)
601
+ return nil if text.count(NOT_SAMPLE_TEXT).zero? # the usual case: digits and whitespace only
602
+
603
+ tokens.find { |token| token.match?(NON_DIGIT) }&.byteslice(0, 20)
604
+ end
605
+
606
+ # Yields the ASCII raster in pieces of about TEXT_CHUNK bytes with comments blanked out (Netpbm's reader
607
+ # skips them in the raster too). With +whole_tokens+ (samples) no token is split between two pieces.
608
+ def each_plain_piece(reader, whole_tokens:)
609
+ carry = "".b # the start of a token cut by the end of the previous chunk
610
+ in_comment = false
611
+ loop do
612
+ text = reader.read(TEXT_CHUNK)
613
+ last = text.bytesize < TEXT_CHUNK
614
+ in_comment = blank_comments!(text, in_comment)
615
+ if whole_tokens && !last
616
+ cut = text.byterindex(WHITESPACE_BYTE)
617
+ if cut.nil? # the chunk continues a single token
618
+ shorten_sample!(carry << text)
619
+ text.clear
620
+ next
621
+ end
622
+ piece = text.byteslice(0, cut + 1)
623
+ piece = carry << piece unless carry.empty?
624
+ carry = text.byteslice(cut + 1, text.bytesize - cut - 1)
625
+ else
626
+ piece = carry.empty? ? text : carry << text
627
+ carry = "".b
628
+ end
629
+ yield piece
630
+ break if last
631
+ end
632
+ end
633
+
634
+ # Overwrites the comments in +chunk+ with spaces, in place. A comment ends at a line break or at the end
635
+ # of the data, so this is the same as removing it, and unlike String#gsub it allocates nothing per
636
+ # comment (a raster may hold one per pixel). +in_comment+ tells whether the previous chunk ended inside
637
+ # a comment; returns the same for this chunk.
638
+ def blank_comments!(chunk, in_comment)
639
+ start = in_comment ? 0 : chunk.byteindex("#")
640
+ line_feed = carriage_return = -1 # the next LF and CR from +start+ on (chunk.bytesize if none)
641
+ while start
642
+ line_feed = chunk.byteindex("\n", start) || chunk.bytesize if line_feed < start
643
+ carriage_return = chunk.byteindex("\r", start) || chunk.bytesize if carriage_return < start
644
+ stop = (line_feed < carriage_return) ? line_feed : carriage_return
645
+ chunk.bytesplice(start, stop - start, BLANKS, 0, stop - start)
646
+ return true if stop == chunk.bytesize
647
+
648
+ start = chunk.byteindex("#", stop)
649
+ end
650
+ false
651
+ end
652
+
653
+ # Replaces a +token+ longer than LONG_SAMPLE with a short one that plain_samples decodes the same way,
654
+ # whatever follows it: with the same first 20 bytes (shown in error messages) and the same value, or a
655
+ # value above 65535 if that one is (both clamp to 255). Keeps a token cut by many chunk boundaries (a
656
+ # long run of digits, or of NUL bytes after a truncated raster) from growing with the data.
657
+ def shorten_sample!(token)
658
+ return if token.bytesize <= LONG_SAMPLE
659
+
660
+ head = token.byteslice(0, 20)
661
+ if token.match?(NON_DIGIT)
662
+ head << "x" # invalid, whatever follows
663
+ else
664
+ zeros = token.byteindex(NONZERO_DIGIT) || token.bytesize - 1 # a zero keeps its last digit
665
+ significant = token.bytesize - zeros # at most 5 digits only if +head+ is all zeros
666
+ head << ((significant > 5) ? "9999999" : token.byteslice(zeros, significant))
667
+ end
668
+ token.replace(head)
669
+ end
670
+
671
+ def truncated_raster!(expected, got, unit)
672
+ raise UnsupportedInput, "truncated PNM: expected #{expected} #{unit} of pixel data, got #{got}"
673
+ end
674
+ end
675
+ end
676
+ end