pdf-reader 2.15.0 → 2.16.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG +29 -2
- data/Rakefile +1 -1
- data/lib/pdf/reader/bounding_rectangle_runs_filter.rb +1 -1
- data/lib/pdf/reader/buffer.rb +82 -30
- data/lib/pdf/reader/cmap.rb +14 -7
- data/lib/pdf/reader/encoding.rb +42 -26
- data/lib/pdf/reader/encoding_utils.rb +128 -0
- data/lib/pdf/reader/error.rb +16 -6
- data/lib/pdf/reader/font.rb +40 -21
- data/lib/pdf/reader/form_xobject.rb +5 -2
- data/lib/pdf/reader/key_builder_v5.rb +0 -1
- data/lib/pdf/reader/object_cache.rb +11 -7
- data/lib/pdf/reader/object_hash.rb +14 -9
- data/lib/pdf/reader/page.rb +2 -2
- data/lib/pdf/reader/page_layout.rb +1 -1
- data/lib/pdf/reader/page_state.rb +73 -29
- data/lib/pdf/reader/page_text_receiver.rb +43 -13
- data/lib/pdf/reader/pages_strategy.rb +1 -1
- data/lib/pdf/reader/parser.rb +52 -18
- data/lib/pdf/reader/point.rb +6 -0
- data/lib/pdf/reader/rc4.rb +43 -0
- data/lib/pdf/reader/rc4_security_handler.rb +2 -2
- data/lib/pdf/reader/rectangle.rb +13 -2
- data/lib/pdf/reader/reference.rb +1 -1
- data/lib/pdf/reader/standard_key_builder.rb +9 -10
- data/lib/pdf/reader/text_run.rb +13 -8
- data/lib/pdf/reader/transformation_matrix.rb +11 -0
- data/lib/pdf/reader/type_check.rb +1 -1
- data/lib/pdf/reader/xref.rb +6 -3
- data/lib/pdf/reader.rb +3 -53
- data/rbi/pdf-reader.rbi +129 -36
- metadata +6 -18
checksums.yaml
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
---
|
|
2
2
|
SHA256:
|
|
3
|
-
metadata.gz:
|
|
4
|
-
data.tar.gz:
|
|
3
|
+
metadata.gz: a9f0772356c29687ffafffb6af1cdc5d08ae7ac54f2d3da60d426f07f5338045
|
|
4
|
+
data.tar.gz: c8df04fcae9e48cd6bfd3ca22819e5108abc75f3084f9c6a2cbc86db07f9a387
|
|
5
5
|
SHA512:
|
|
6
|
-
metadata.gz:
|
|
7
|
-
data.tar.gz:
|
|
6
|
+
metadata.gz: f9bf3dc84620d087f2593af2c6027984ae51998ec6706a7f12cc5bee8eabc77589fbb5804d878e1cf43def5566eccc72c9cb39913b27a66b0353243a107cccd7
|
|
7
|
+
data.tar.gz: 2f35f101c0753c23094879e8bca382ca4daf6344a2922f096459ef31126e8180ec47a7584c2535807b347122539af131c6348a3aca48f9e775652cd208192a01
|
data/CHANGELOG
CHANGED
|
@@ -1,5 +1,32 @@
|
|
|
1
|
-
v2.
|
|
1
|
+
v2.16.0 (9th August 2026)
|
|
2
|
+
- Add new method that use Point(x, y) instead of Array[Integer], for improved typing with sorbet (https://github.com/yob/pdf-reader/pull/578)
|
|
3
|
+
- Support extracting text via ActualText annotations (https://github.com/yob/pdf-reader/pull/587)
|
|
4
|
+
- Accept any IO-like object in PDF::Reader.new() (https://github.com/yob/pdf-reader/pull/607)
|
|
5
|
+
- Expanded type annotations
|
|
6
|
+
- Add dynamically generated RBS type signaturesd to published gem (https://github.com/yob/pdf-reader/pull/580)
|
|
7
|
+
- A few bug fixes to CMap parsing and Unicode extraction
|
|
8
|
+
- https://github.com/yob/pdf-reader/pull/582
|
|
9
|
+
- https://github.com/yob/pdf-reader/pull/593
|
|
10
|
+
- Fix occasional exception when parsing PDFDoc strings with invalid bytes (https://github.com/yob/pdf-reader/pull/590)
|
|
11
|
+
- Fox occasional exception parsing differnce tables with an unmatched surrgate pair
|
|
12
|
+
- https://github.com/yob/pdf-reader/pull/604
|
|
13
|
+
- https://github.com/yob/pdf-reader/pull/605
|
|
14
|
+
- Reduce object allocations when parsing many PDFs
|
|
15
|
+
- https://github.com/yob/pdf-reader/pull/595
|
|
16
|
+
- https://github.com/yob/pdf-reader/pull/596
|
|
17
|
+
- https://github.com/yob/pdf-reader/pull/598
|
|
18
|
+
- https://github.com/yob/pdf-reader/pull/599
|
|
19
|
+
- https://github.com/yob/pdf-reader/pull/600
|
|
20
|
+
- https://github.com/yob/pdf-reader/pull/601
|
|
21
|
+
- Add truffleruby to the test matrix in CI (https://github.com/yob/pdf-reader/pull/583)
|
|
22
|
+
- Add JRuby 10.1 to the test matrix (https://github.com/yob/pdf-reader/pull/597)
|
|
23
|
+
|
|
24
|
+
v2.15.1 (28th December 2025)
|
|
25
|
+
- Add ruby 4.0 to the CI matrix (https://github.com/yob/pdf-reader/pull/575)
|
|
26
|
+
- Avoiding raising an error when ToUnicode poins to the wrong object type (https://github.com/yob/pdf-reader/pull/573)
|
|
27
|
+
- Skip invalid UTF-16 surrogate pairs in CMaps (https://github.com/yob/pdf-reader/pull/574)
|
|
2
28
|
|
|
29
|
+
v2.15.0 (13th August 2025)
|
|
3
30
|
- Overhaul sorbet types, moving from an external RBI file to inline comments in RBS syntax
|
|
4
31
|
- multiple PRs, but mainly https://github.com/yob/pdf-reader/pull/562
|
|
5
32
|
- See https://railsatscale.com/2025-04-23-rbs-support-for-sorbet/
|
|
@@ -88,7 +115,7 @@ v2.6.0 (12th November 2021)
|
|
|
88
115
|
|
|
89
116
|
v2.5.0 (6th June 2021)
|
|
90
117
|
- bump minimum ruby version to 2.0
|
|
91
|
-
- Correctly handle trascoding to UTF-8 from some fonts that use a difference table [#344](https://github.com/yob/pdf-reader/pull/344/)
|
|
118
|
+
- Correctly handle trascoding to UTF-8 from some fonts that use a difference table [#344](https://github.com/yob/pdf-reader/pull/344/)
|
|
92
119
|
- Fix some character spacing issues with the TJ operator [#343](https://github.com/yob/pdf-reader/pull/343)
|
|
93
120
|
- Fix crash with some encrypted PDFs [#348](https://github.com/yob/pdf-reader/pull/348/)
|
|
94
121
|
- Fix positions of text on some PDFs with pages rotated 90° [#350](https://github.com/yob/pdf-reader/pull/350/)
|
data/Rakefile
CHANGED
|
@@ -14,7 +14,7 @@ desc "Run cane to check quality metrics"
|
|
|
14
14
|
Cane::RakeTask.new(:quality) do |cane|
|
|
15
15
|
cane.abc_max = 20
|
|
16
16
|
cane.style_measure = 100
|
|
17
|
-
cane.max_violations =
|
|
17
|
+
cane.max_violations = 32
|
|
18
18
|
|
|
19
19
|
cane.use Morecane::EncodingCheck, :encoding_glob => "{app,lib,spec}/**/*.rb"
|
|
20
20
|
end
|
|
@@ -10,7 +10,7 @@ class PDF::Reader
|
|
|
10
10
|
|
|
11
11
|
#: (Array[PDF::Reader::TextRun], PDF::Reader::Rectangle) -> Array[PDF::Reader::TextRun]
|
|
12
12
|
def self.runs_within_rect(runs, rect)
|
|
13
|
-
runs.select { |run| rect.
|
|
13
|
+
runs.select { |run| rect.contains_xy?(run.x, run.y) }
|
|
14
14
|
end
|
|
15
15
|
end
|
|
16
16
|
end
|
data/lib/pdf/reader/buffer.rb
CHANGED
|
@@ -62,6 +62,9 @@ class PDF::Reader
|
|
|
62
62
|
# must match whole tokens
|
|
63
63
|
DIGITS_ONLY = %r{\A\d+\z} #: Regexp
|
|
64
64
|
|
|
65
|
+
# bytes that terminate a PDF name token (used in prepare_regular_token)
|
|
66
|
+
NAME_TERMINATOR_BYTES = ([nil, 0x20, 0x0A] + TOKEN_DELIMITER).freeze #: Array[Integer?]
|
|
67
|
+
|
|
65
68
|
#: Integer
|
|
66
69
|
attr_reader :pos
|
|
67
70
|
|
|
@@ -255,13 +258,25 @@ class PDF::Reader
|
|
|
255
258
|
|
|
256
259
|
token_one = @tokens[0]
|
|
257
260
|
token_two = @tokens[1]
|
|
258
|
-
if token_one.is_a?(String) && token_two.is_a?(String) &&
|
|
261
|
+
if token_one.is_a?(String) && token_two.is_a?(String) && match?(token_one, DIGITS_ONLY) && match?(token_two, DIGITS_ONLY)
|
|
259
262
|
@tokens[0] = PDF::Reader::Reference.new(token_one.to_i, token_two.to_i)
|
|
260
263
|
@tokens.delete_at(2)
|
|
261
264
|
@tokens.delete_at(1)
|
|
262
265
|
end
|
|
263
266
|
end
|
|
264
267
|
|
|
268
|
+
# Once min ruby version is >= 2.4, we can drop this method and just use String#match?
|
|
269
|
+
#
|
|
270
|
+
#: (String, Regexp) -> bool
|
|
271
|
+
def match?(str, regexp)
|
|
272
|
+
# We prefer match? because fewer objects are allocated, and this code path is hot
|
|
273
|
+
if str.respond_to?(:match?)
|
|
274
|
+
str.match?(regexp)
|
|
275
|
+
else
|
|
276
|
+
str.match(regexp) != nil
|
|
277
|
+
end
|
|
278
|
+
end
|
|
279
|
+
|
|
265
280
|
# Extract data between ID and EI
|
|
266
281
|
# If the EI follows white-space the space is dropped from the data
|
|
267
282
|
# The EI must followed by white-space or end of buffer
|
|
@@ -327,8 +342,12 @@ class PDF::Reader
|
|
|
327
342
|
# ignore it
|
|
328
343
|
else
|
|
329
344
|
@tokens << str if str.size > 0
|
|
330
|
-
|
|
331
|
-
|
|
345
|
+
if byte == 0x3E
|
|
346
|
+
@tokens << ">"
|
|
347
|
+
else
|
|
348
|
+
@tokens << ">"
|
|
349
|
+
@tokens << byte.chr
|
|
350
|
+
end
|
|
332
351
|
break
|
|
333
352
|
end
|
|
334
353
|
end
|
|
@@ -383,6 +402,7 @@ class PDF::Reader
|
|
|
383
402
|
|
|
384
403
|
case byte
|
|
385
404
|
when nil
|
|
405
|
+
@tokens << tok if tok.size > 0
|
|
386
406
|
break
|
|
387
407
|
when 0x25
|
|
388
408
|
# comment, ignore everything until the next EOL char
|
|
@@ -390,63 +410,95 @@ class PDF::Reader
|
|
|
390
410
|
commentbyte = @io.getbyte
|
|
391
411
|
break if commentbyte.nil? || commentbyte == 0x0A || commentbyte == 0x0D
|
|
392
412
|
end
|
|
393
|
-
when
|
|
413
|
+
when 0x00, 0x09, 0x0A, 0x0C, 0x0D, 0x20 # TOKEN_WHITESPACE
|
|
394
414
|
# white space, token finished
|
|
395
|
-
|
|
396
|
-
|
|
397
|
-
|
|
398
|
-
|
|
399
|
-
|
|
415
|
+
if tok.size > 0
|
|
416
|
+
@tokens << tok
|
|
417
|
+
else
|
|
418
|
+
#If the token was empty, chomp the rest of the whitespace too
|
|
419
|
+
while TOKEN_WHITESPACE.include?(peek_byte)
|
|
420
|
+
@io.getbyte
|
|
421
|
+
end
|
|
400
422
|
end
|
|
401
|
-
tok = "".dup
|
|
402
423
|
break
|
|
403
424
|
when 0x3C
|
|
404
425
|
# opening delimiter '<', start of new token
|
|
405
|
-
|
|
426
|
+
if tok.size > 0
|
|
427
|
+
@tokens << tok
|
|
428
|
+
end
|
|
406
429
|
if peek_byte == 0x3C # check if token is actually '<<'
|
|
407
430
|
@io.getbyte
|
|
408
431
|
@tokens << "<<"
|
|
409
432
|
else
|
|
410
433
|
@tokens << "<"
|
|
411
434
|
end
|
|
412
|
-
tok = "".dup
|
|
413
435
|
break
|
|
414
436
|
when 0x3E
|
|
415
437
|
# closing delimiter '>', start of new token
|
|
416
|
-
|
|
438
|
+
if tok.size > 0
|
|
439
|
+
@tokens << tok
|
|
440
|
+
end
|
|
417
441
|
if peek_byte == 0x3E # check if token is actually '>>'
|
|
418
442
|
@io.getbyte
|
|
419
443
|
@tokens << ">>"
|
|
420
444
|
else
|
|
421
445
|
@tokens << ">"
|
|
422
446
|
end
|
|
423
|
-
tok = "".dup
|
|
424
447
|
break
|
|
425
|
-
when 0x28
|
|
426
|
-
# opening delimiter, start of new token
|
|
427
|
-
|
|
428
|
-
|
|
429
|
-
|
|
448
|
+
when 0x28
|
|
449
|
+
# opening delimiter '(', start of new token
|
|
450
|
+
if tok.size > 0
|
|
451
|
+
@tokens << tok
|
|
452
|
+
end
|
|
453
|
+
@tokens << "("
|
|
430
454
|
break
|
|
431
|
-
when
|
|
432
|
-
#
|
|
433
|
-
|
|
434
|
-
|
|
435
|
-
|
|
455
|
+
when 0x5B
|
|
456
|
+
# opening delimiter '[', start of new token
|
|
457
|
+
if tok.size > 0
|
|
458
|
+
@tokens << tok
|
|
459
|
+
end
|
|
460
|
+
@tokens << "["
|
|
461
|
+
break
|
|
462
|
+
when 0x7B
|
|
463
|
+
# opening delimiter '{', start of new token
|
|
464
|
+
if tok.size > 0
|
|
465
|
+
@tokens << tok
|
|
466
|
+
end
|
|
467
|
+
@tokens << "{"
|
|
468
|
+
break
|
|
469
|
+
when 0x29
|
|
470
|
+
# closing delimiter ')'
|
|
471
|
+
if tok.size > 0
|
|
472
|
+
@tokens << tok
|
|
473
|
+
end
|
|
474
|
+
@tokens << ")"
|
|
475
|
+
break
|
|
476
|
+
when 0x5D
|
|
477
|
+
# closing delimiter ']'
|
|
478
|
+
if tok.size > 0
|
|
479
|
+
@tokens << tok
|
|
480
|
+
end
|
|
481
|
+
@tokens << "]"
|
|
482
|
+
break
|
|
483
|
+
when 0x7D
|
|
484
|
+
# closing delimiter '}'
|
|
485
|
+
if tok.size > 0
|
|
486
|
+
@tokens << tok
|
|
487
|
+
end
|
|
488
|
+
@tokens << "}"
|
|
436
489
|
break
|
|
437
490
|
when 0x2F
|
|
438
491
|
# PDF name, start of new token
|
|
439
|
-
|
|
440
|
-
|
|
441
|
-
|
|
442
|
-
|
|
492
|
+
if tok.size > 0
|
|
493
|
+
@tokens << tok
|
|
494
|
+
end
|
|
495
|
+
@tokens << "/"
|
|
496
|
+
@tokens << "" if NAME_TERMINATOR_BYTES.include?(peek_byte)
|
|
443
497
|
break
|
|
444
498
|
else
|
|
445
499
|
tok << byte
|
|
446
500
|
end
|
|
447
501
|
end
|
|
448
|
-
|
|
449
|
-
@tokens << tok if tok.size > 0
|
|
450
502
|
end
|
|
451
503
|
|
|
452
504
|
# peek at the next character in the io stream, leaving the stream position
|
data/lib/pdf/reader/cmap.rb
CHANGED
|
@@ -46,6 +46,12 @@ class PDF::Reader
|
|
|
46
46
|
"def" => :noop
|
|
47
47
|
} #: Hash[String, Symbol]
|
|
48
48
|
|
|
49
|
+
# Indicates the start of a UTF-16 surrogate pair, see
|
|
50
|
+
# https://en.wikipedia.org/wiki/Universal_Character_Set_characters
|
|
51
|
+
HIGH_SURROGATE_RANGE = (0xD800..0xDBFF) #: Range[Integer]
|
|
52
|
+
|
|
53
|
+
EMPTY_CODEPOINTS = [].freeze #: Array[Integer]
|
|
54
|
+
|
|
49
55
|
#: Hash[Integer, Array[Integer]]
|
|
50
56
|
attr_reader :map
|
|
51
57
|
|
|
@@ -66,7 +72,7 @@ class PDF::Reader
|
|
|
66
72
|
#
|
|
67
73
|
#: (Integer) -> Array[Integer]
|
|
68
74
|
def decode(c)
|
|
69
|
-
@map.fetch(c,
|
|
75
|
+
@map.fetch(c, EMPTY_CODEPOINTS)
|
|
70
76
|
end
|
|
71
77
|
|
|
72
78
|
private
|
|
@@ -77,7 +83,7 @@ class PDF::Reader
|
|
|
77
83
|
mode = initial_mode
|
|
78
84
|
instructions = []
|
|
79
85
|
|
|
80
|
-
while token = parser.parse_token
|
|
86
|
+
while token = parser.parse_token
|
|
81
87
|
if token.is_a?(String) || token.is_a?(Array)
|
|
82
88
|
if token == "beginbfchar"
|
|
83
89
|
mode = :char
|
|
@@ -104,7 +110,7 @@ class PDF::Reader
|
|
|
104
110
|
#: (String) -> PDF::Reader::Parser
|
|
105
111
|
def build_parser(instructions)
|
|
106
112
|
buffer = Buffer.new(StringIO.new(instructions))
|
|
107
|
-
Parser.new(buffer)
|
|
113
|
+
Parser.new(buffer, operators: CMAP_KEYWORDS, relaxed_dictionaries: true)
|
|
108
114
|
end
|
|
109
115
|
|
|
110
116
|
# The following includes some manual decoding of UTF-16BE strings into unicode codepoints. In
|
|
@@ -124,15 +130,16 @@ class PDF::Reader
|
|
|
124
130
|
end
|
|
125
131
|
result = []
|
|
126
132
|
while unpacked_string.any? do
|
|
127
|
-
if unpacked_string.size >= 2 &&
|
|
128
|
-
unpacked_string.first.to_i >= 0xD800 &&
|
|
129
|
-
unpacked_string.first.to_i <= 0xDBFF
|
|
133
|
+
if unpacked_string.size >= 2 && HIGH_SURROGATE_RANGE.include?(unpacked_string.first.to_i)
|
|
130
134
|
# this is a Unicode UTF-16 "Surrogate Pair" see Unicode Spec. Chapter 3.7
|
|
131
135
|
# lets convert to a UTF-32. (the high bit is between 0xD800-0xDBFF, the
|
|
132
136
|
# low bit is between 0xDC00-0xDFFF) for example: U+1D44E (U+D835 U+DC4E)
|
|
133
137
|
point_one = unpacked_string.shift.to_i
|
|
134
138
|
point_two = unpacked_string.shift.to_i
|
|
135
139
|
result << (point_one - 0xD800) * 0x400 + (point_two - 0xDC00) + 0x10000
|
|
140
|
+
elsif unpacked_string.size == 1 && HIGH_SURROGATE_RANGE.include?(unpacked_string.first.to_i)
|
|
141
|
+
# the start of a surrogate pair but the pair is missing. Skip it
|
|
142
|
+
unpacked_string.shift
|
|
136
143
|
else
|
|
137
144
|
result << unpacked_string.shift
|
|
138
145
|
end
|
|
@@ -170,7 +177,7 @@ class PDF::Reader
|
|
|
170
177
|
end_code = str_to_int(end_code).first
|
|
171
178
|
dst = str_to_int(dst)
|
|
172
179
|
|
|
173
|
-
return if start_code.nil? || end_code.nil?
|
|
180
|
+
return if start_code.nil? || end_code.nil? || dst.empty?
|
|
174
181
|
|
|
175
182
|
# add all values in the range to our mapping
|
|
176
183
|
(start_code..end_code).each_with_index do |val, idx|
|
data/lib/pdf/reader/encoding.rb
CHANGED
|
@@ -34,15 +34,19 @@ class PDF::Reader
|
|
|
34
34
|
CONTROL_CHARS = [0,1,2,3,4,5,6,7,8,11,12,14,15,16,17,18,19,20,21,22,23,
|
|
35
35
|
24,25,26,27,28,29,30,31] #: Array[Integer]
|
|
36
36
|
UNKNOWN_CHAR = 0x25AF #: Integer # ▯
|
|
37
|
+
DEFAULT_MAPPING = (0..255).each_with_object({}) { |i, h|
|
|
38
|
+
h[i] = CONTROL_CHARS.include?(i) ? UNKNOWN_CHAR : i
|
|
39
|
+
}.freeze #: Hash[Integer, Integer]
|
|
40
|
+
|
|
41
|
+
# Cache mapping files to avoid re-reading and re-parsing the same file for
|
|
42
|
+
# every Encoding that uses it (e.g. many fonts sharing StandardEncoding).
|
|
43
|
+
FILE_MAPPINGS = {} #: Hash[String, Hash[Integer, Integer]]
|
|
37
44
|
|
|
38
45
|
#: String
|
|
39
46
|
attr_reader :unpack
|
|
40
47
|
|
|
41
48
|
#: (Hash[Symbol, untyped] | Symbol | nil) -> void
|
|
42
49
|
def initialize(enc)
|
|
43
|
-
# maps from character codes to Unicode codepoints
|
|
44
|
-
@mapping = default_mapping #: Hash[Integer, Integer]
|
|
45
|
-
|
|
46
50
|
# maps from character codes to UTF-8 strings.
|
|
47
51
|
@string_cache = {} #: Hash[Integer, String]
|
|
48
52
|
|
|
@@ -58,7 +62,15 @@ class PDF::Reader
|
|
|
58
62
|
@differences = nil #: Hash[Integer, Integer] | nil
|
|
59
63
|
@glyphlist = nil #: PDF::Reader::GlyphHash | nil
|
|
60
64
|
|
|
61
|
-
|
|
65
|
+
# maps from character codes to Unicode codepoints
|
|
66
|
+
# If we have a mapping file, use the cached parsed version (loaded once per file)
|
|
67
|
+
if @map_file
|
|
68
|
+
@mapping = (
|
|
69
|
+
FILE_MAPPINGS[@map_file] ||= build_file_mapping(@map_file)
|
|
70
|
+
).dup #: Hash[Integer, Integer]
|
|
71
|
+
else
|
|
72
|
+
@mapping = DEFAULT_MAPPING.dup
|
|
73
|
+
end #: Hash[Integer, Integer]
|
|
62
74
|
|
|
63
75
|
if enc.is_a?(Hash) && enc[:Differences]
|
|
64
76
|
self.differences = enc[:Differences]
|
|
@@ -149,25 +161,15 @@ class PDF::Reader
|
|
|
149
161
|
private
|
|
150
162
|
|
|
151
163
|
# returns a hash that:
|
|
152
|
-
# - maps control chars and nil to the unicode "unknown character"
|
|
153
|
-
# - leaves all other bytes <= 255 unchaged
|
|
154
|
-
#
|
|
155
|
-
# Each specific encoding will change this default as required for their glyphs
|
|
156
|
-
#: () -> Hash[Integer, Integer]
|
|
157
|
-
def default_mapping
|
|
158
|
-
all_bytes = (0..255).to_a
|
|
159
|
-
tuples = all_bytes.map {|i|
|
|
160
|
-
CONTROL_CHARS.include?(i) ? [i, UNKNOWN_CHAR] : [i,i]
|
|
161
|
-
}
|
|
162
|
-
mapping = Hash[tuples]
|
|
163
|
-
mapping
|
|
164
|
-
end
|
|
165
|
-
|
|
166
164
|
#: (Integer) -> String
|
|
167
165
|
def internal_int_to_utf8_string(glyph_code)
|
|
168
|
-
|
|
169
|
-
|
|
170
|
-
|
|
166
|
+
codepoint = @mapping[glyph_code.to_i] || glyph_code.to_i
|
|
167
|
+
if valid_unicode_scalar?(codepoint.to_i)
|
|
168
|
+
codepoints = [codepoint]
|
|
169
|
+
else
|
|
170
|
+
codepoints = [UNKNOWN_CHAR]
|
|
171
|
+
end
|
|
172
|
+
ret = codepoints.pack("U*")
|
|
171
173
|
ret.force_encoding("UTF-8")
|
|
172
174
|
ret
|
|
173
175
|
end
|
|
@@ -187,11 +189,22 @@ class PDF::Reader
|
|
|
187
189
|
|
|
188
190
|
#: (String) -> String
|
|
189
191
|
def convert_to_utf8(str)
|
|
190
|
-
ret = str.unpack(unpack).map! { |c|
|
|
192
|
+
ret = str.unpack(unpack).map! { |c|
|
|
193
|
+
codepoint = @mapping[c.to_i] || c
|
|
194
|
+
valid_unicode_scalar?(codepoint.to_i) ? codepoint : UNKNOWN_CHAR
|
|
195
|
+
}.pack("U*")
|
|
191
196
|
ret.force_encoding("UTF-8")
|
|
192
197
|
ret
|
|
193
198
|
end
|
|
194
199
|
|
|
200
|
+
# true if the integer is a Unicode scalar value that can be encoded as
|
|
201
|
+
# UTF-8. Surrogates (U+D800..U+DFFF) and code points above U+10FFFF are
|
|
202
|
+
# excluded, as packing them produces invalid UTF-8 byte sequences.
|
|
203
|
+
#: (Integer) -> bool
|
|
204
|
+
def valid_unicode_scalar?(codepoint)
|
|
205
|
+
codepoint.between?(0, 0xD7FF) || codepoint.between?(0xE000, 0x10FFFF)
|
|
206
|
+
end
|
|
207
|
+
|
|
195
208
|
#: (Symbol) -> String
|
|
196
209
|
def get_unpack(enc)
|
|
197
210
|
case enc
|
|
@@ -229,14 +242,17 @@ class PDF::Reader
|
|
|
229
242
|
@glyphlist ||= PDF::Reader::GlyphHash.new
|
|
230
243
|
end
|
|
231
244
|
|
|
232
|
-
#: (String) ->
|
|
233
|
-
def
|
|
245
|
+
#: (String) -> Hash[Integer, Integer]
|
|
246
|
+
def build_file_mapping(file)
|
|
247
|
+
mapping = DEFAULT_MAPPING.dup
|
|
234
248
|
File.open(file, "r:BINARY") do |f|
|
|
235
249
|
f.each do |l|
|
|
236
|
-
|
|
237
|
-
|
|
250
|
+
if l =~ /\A([0-9A-Za-z]+);([0-9A-F]{4})/
|
|
251
|
+
mapping[$1.to_i(16)] = $2.to_i(16)
|
|
252
|
+
end
|
|
238
253
|
end
|
|
239
254
|
end
|
|
255
|
+
mapping.freeze
|
|
240
256
|
end
|
|
241
257
|
|
|
242
258
|
end
|
|
@@ -0,0 +1,128 @@
|
|
|
1
|
+
# coding: utf-8
|
|
2
|
+
# typed: strict
|
|
3
|
+
# frozen_string_literal: true
|
|
4
|
+
|
|
5
|
+
module PDF
|
|
6
|
+
class Reader
|
|
7
|
+
# Utiliy class for some conversions to utf8 (the standard output encoding for pdf-reader).
|
|
8
|
+
# This is not used for general purpose of encoding management while parsing PDFs, that lives
|
|
9
|
+
# in PDF::Reader::Encoding
|
|
10
|
+
class EncodingUtils
|
|
11
|
+
|
|
12
|
+
UTF16_BOM = [254, 255] #: Array[Integer]
|
|
13
|
+
|
|
14
|
+
# PDFDocEncoding bytes that map to Unicode codepoints differing from their
|
|
15
|
+
# byte value. All other bytes map directly to their Unicode codepoint value
|
|
16
|
+
# (0x00-0x17/0x20-0x7F: ASCII; 0xA1-0xFF: Latin-1, which PDFDocEncoding
|
|
17
|
+
# matches exactly).
|
|
18
|
+
PDFDOC_CODEPOINTS = {
|
|
19
|
+
0x18 => 0x02D8, # BREVE
|
|
20
|
+
0x19 => 0x02C7, # CARON
|
|
21
|
+
0x1A => 0x02C6, # MODIFIER LETTER CIRCUMFLEX ACCENT
|
|
22
|
+
0x1B => 0x02D9, # DOT ABOVE
|
|
23
|
+
0x1C => 0x02DD, # DOUBLE ACUTE ACCENT
|
|
24
|
+
0x1D => 0x02DB, # OGONEK
|
|
25
|
+
0x1E => 0x02DA, # RING ABOVE
|
|
26
|
+
0x1F => 0x02DC, # SMALL TILDE
|
|
27
|
+
0x80 => 0x2022, # BULLET
|
|
28
|
+
0x81 => 0x2020, # DAGGER
|
|
29
|
+
0x82 => 0x2021, # DOUBLE DAGGER
|
|
30
|
+
0x83 => 0x2026, # HORIZONTAL ELLIPSIS
|
|
31
|
+
0x84 => 0x2014, # EM DASH
|
|
32
|
+
0x85 => 0x2013, # EN DASH
|
|
33
|
+
0x86 => 0x0192, # LATIN SMALL LETTER F WITH HOOK
|
|
34
|
+
0x87 => 0x2044, # FRACTION SLASH
|
|
35
|
+
0x88 => 0x2039, # SINGLE LEFT-POINTING ANGLE QUOTATION MARK
|
|
36
|
+
0x89 => 0x203A, # SINGLE RIGHT-POINTING ANGLE QUOTATION MARK
|
|
37
|
+
0x8A => 0x2212, # MINUS SIGN
|
|
38
|
+
0x8B => 0x2030, # PER MILLE SIGN
|
|
39
|
+
0x8C => 0x201E, # DOUBLE LOW-9 QUOTATION MARK
|
|
40
|
+
0x8D => 0x201C, # LEFT DOUBLE QUOTATION MARK
|
|
41
|
+
0x8E => 0x201D, # RIGHT DOUBLE QUOTATION MARK
|
|
42
|
+
0x8F => 0x2018, # LEFT SINGLE QUOTATION MARK
|
|
43
|
+
0x90 => 0x2019, # RIGHT SINGLE QUOTATION MARK
|
|
44
|
+
0x91 => 0x201A, # SINGLE LOW-9 QUOTATION MARK
|
|
45
|
+
0x92 => 0x2122, # TRADE MARK SIGN
|
|
46
|
+
0x93 => 0xFB01, # LATIN SMALL LIGATURE FI
|
|
47
|
+
0x94 => 0xFB02, # LATIN SMALL LIGATURE FL
|
|
48
|
+
0x95 => 0x0141, # LATIN CAPITAL LETTER L WITH STROKE
|
|
49
|
+
0x96 => 0x0152, # LATIN CAPITAL LIGATURE OE
|
|
50
|
+
0x97 => 0x0160, # LATIN CAPITAL LETTER S WITH CARON
|
|
51
|
+
0x98 => 0x0178, # LATIN CAPITAL LETTER Y WITH DIAERESIS
|
|
52
|
+
0x99 => 0x017D, # LATIN CAPITAL LETTER Z WITH CARON
|
|
53
|
+
0x9A => 0x0131, # LATIN SMALL LETTER DOTLESS I
|
|
54
|
+
0x9B => 0x0142, # LATIN SMALL LETTER L WITH STROKE
|
|
55
|
+
0x9C => 0x0153, # LATIN SMALL LIGATURE OE
|
|
56
|
+
0x9D => 0x0161, # LATIN SMALL LETTER S WITH CARON
|
|
57
|
+
0x9E => 0x017E, # LATIN SMALL LETTER Z WITH CARON
|
|
58
|
+
0xA0 => 0x20AC, # EURO SIGN
|
|
59
|
+
}.freeze #: Hash[Integer, Integer]
|
|
60
|
+
|
|
61
|
+
#: (untyped) -> untyped
|
|
62
|
+
def self.obj_to_utf8(obj)
|
|
63
|
+
new.obj_to_utf8(obj)
|
|
64
|
+
end
|
|
65
|
+
|
|
66
|
+
#: (String) -> String
|
|
67
|
+
def self.string_to_utf8(str)
|
|
68
|
+
new.string_to_utf8(str)
|
|
69
|
+
end
|
|
70
|
+
|
|
71
|
+
# Recursively convert Hashes, Arrays, and Strings to UTF-8
|
|
72
|
+
#
|
|
73
|
+
#: (untyped) -> untyped
|
|
74
|
+
def obj_to_utf8(obj)
|
|
75
|
+
case obj
|
|
76
|
+
when ::Hash then
|
|
77
|
+
{}.tap { |new_hash|
|
|
78
|
+
obj.each do |key, value|
|
|
79
|
+
new_hash[key] = obj_to_utf8(value)
|
|
80
|
+
end
|
|
81
|
+
}
|
|
82
|
+
when Array then
|
|
83
|
+
obj.map { |item| obj_to_utf8(item) }
|
|
84
|
+
when String then
|
|
85
|
+
string_to_utf8(obj)
|
|
86
|
+
else
|
|
87
|
+
obj
|
|
88
|
+
end
|
|
89
|
+
end
|
|
90
|
+
|
|
91
|
+
# Convert a String to UTF-8
|
|
92
|
+
#
|
|
93
|
+
#: (String) -> String
|
|
94
|
+
def string_to_utf8(str)
|
|
95
|
+
if has_utf16_bom?(str)
|
|
96
|
+
utf16_to_utf8(str)
|
|
97
|
+
else
|
|
98
|
+
pdfdoc_to_utf8(str)
|
|
99
|
+
end
|
|
100
|
+
end
|
|
101
|
+
|
|
102
|
+
private
|
|
103
|
+
|
|
104
|
+
#: (String) -> bool
|
|
105
|
+
def has_utf16_bom?(str)
|
|
106
|
+
first_bytes = str[0,2]
|
|
107
|
+
|
|
108
|
+
return false if first_bytes.nil?
|
|
109
|
+
|
|
110
|
+
first_bytes.unpack("C*") == UTF16_BOM
|
|
111
|
+
end
|
|
112
|
+
|
|
113
|
+
#: (String) -> String
|
|
114
|
+
def pdfdoc_to_utf8(str)
|
|
115
|
+
str.bytes.map { |b| PDFDOC_CODEPOINTS[b] || b }.pack("U*")
|
|
116
|
+
end
|
|
117
|
+
|
|
118
|
+
#: (String) -> String
|
|
119
|
+
def utf16_to_utf8(obj)
|
|
120
|
+
obj.dup.force_encoding(
|
|
121
|
+
::Encoding::UTF_16
|
|
122
|
+
).encode(
|
|
123
|
+
::Encoding::UTF_8, invalid: :replace, replace: "\uFFFD"
|
|
124
|
+
)
|
|
125
|
+
end
|
|
126
|
+
end
|
|
127
|
+
end
|
|
128
|
+
end
|
data/lib/pdf/reader/error.rb
CHANGED
|
@@ -33,16 +33,26 @@ class PDF::Reader
|
|
|
33
33
|
################################################################################
|
|
34
34
|
#: (untyped, untyped, ?untyped) -> untyped
|
|
35
35
|
def self.str_assert(lvalue, rvalue, chars=nil)
|
|
36
|
-
|
|
36
|
+
if chars and !lvalue.kind_of?(String)
|
|
37
|
+
raise MalformedPDFError, "PDF malformed, expected string but found #{lvalue.class} instead"
|
|
38
|
+
end
|
|
37
39
|
lvalue = lvalue[0,chars] if chars
|
|
38
|
-
|
|
40
|
+
|
|
41
|
+
if lvalue != rvalue
|
|
42
|
+
raise MalformedPDFError, "PDF malformed, expected '#{rvalue}' but found '#{lvalue}' instead"
|
|
43
|
+
end
|
|
39
44
|
end
|
|
40
45
|
################################################################################
|
|
41
46
|
#: (untyped, untyped, ?untyped) -> untyped
|
|
42
47
|
def self.str_assert_not(lvalue, rvalue, chars=nil)
|
|
43
|
-
|
|
48
|
+
if chars and !lvalue.kind_of?(String)
|
|
49
|
+
raise MalformedPDFError, "PDF malformed, expected string but found #{lvalue.class} instead"
|
|
50
|
+
end
|
|
44
51
|
lvalue = lvalue[0,chars] if chars
|
|
45
|
-
|
|
52
|
+
|
|
53
|
+
if lvalue == rvalue
|
|
54
|
+
raise MalformedPDFError, "PDF malformed, expected '#{rvalue}' but found '#{lvalue}' instead"
|
|
55
|
+
end
|
|
46
56
|
end
|
|
47
57
|
################################################################################
|
|
48
58
|
#: (untyped, untyped) -> untyped
|
|
@@ -50,12 +60,12 @@ class PDF::Reader
|
|
|
50
60
|
raise MalformedPDFError, "PDF malformed, expected '#{rvalue}' but found '#{lvalue}' instead" if lvalue != rvalue
|
|
51
61
|
end
|
|
52
62
|
################################################################################
|
|
53
|
-
#: (Object, String, Module) -> void
|
|
63
|
+
#: (Object, String, Module[untyped]) -> void
|
|
54
64
|
def self.validate_type(object, name, klass)
|
|
55
65
|
raise ArgumentError, "#{name} (#{object}) must be a #{klass}" unless object.is_a?(klass)
|
|
56
66
|
end
|
|
57
67
|
################################################################################
|
|
58
|
-
#: (Object, String, Module) -> void
|
|
68
|
+
#: (Object, String, Module[untyped]) -> void
|
|
59
69
|
def self.validate_type_as_malformed(object, name, klass)
|
|
60
70
|
raise MalformedPDFError, "#{name} (#{object}) must be a #{klass}" unless object.is_a?(klass)
|
|
61
71
|
end
|