qpdf-compress 0.6.0 → 0.6.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +15 -0
- package/binding.gyp +24 -9
- package/package.json +3 -2
- package/scripts/download-harfbuzz.mjs +166 -0
- package/scripts/download-mozjpeg.mjs +1 -0
- package/scripts/download-qpdf.mjs +8 -1
- package/scripts/install.mjs +1 -0
- package/src/content.cc +291 -0
- package/src/content.h +6 -0
- package/src/encoding_tables.h +42 -0
- package/src/font_subset.cc +128 -425
- package/src/font_subset.h +17 -7
- package/src/fonts.cc +1209 -0
- package/src/fonts.h +5 -0
- package/src/hash_utils.h +15 -0
- package/src/images.cc +156 -132
- package/src/images.h +20 -2
- package/src/jpeg.cc +6 -0
- package/src/jpeg.h +1 -0
- package/src/optimize.h +5 -13
- package/src/qpdf_addon.cc +137 -55
- package/src/strip.cc +391 -0
- package/src/strip.h +9 -0
- package/src/structure.cc +346 -0
- package/src/structure.h +7 -0
- package/src/optimize.cc +0 -1048
package/src/fonts.cc
ADDED
|
@@ -0,0 +1,1209 @@
|
|
|
1
|
+
#include "fonts.h"
|
|
2
|
+
#include "encoding_tables.h"
|
|
3
|
+
#include "font_subset.h"
|
|
4
|
+
|
|
5
|
+
#include <algorithm>
|
|
6
|
+
#include <cstdint>
|
|
7
|
+
#include <map>
|
|
8
|
+
#include <set>
|
|
9
|
+
#include <string>
|
|
10
|
+
#include <utility>
|
|
11
|
+
#include <vector>
|
|
12
|
+
|
|
13
|
+
#include <qpdf/QPDFObjectHandle.hh>
|
|
14
|
+
#include <qpdf/QPDFPageDocumentHelper.hh>
|
|
15
|
+
#include <qpdf/QPDFPageObjectHelper.hh>
|
|
16
|
+
|
|
17
|
+
namespace {
|
|
18
|
+
|
|
19
|
+
/// Checks if a font BaseFont name has a subset prefix (e.g., "ABCDEF+ArialMT").
|
|
20
|
+
bool isAlreadySubset(QPDFObjectHandle fontObj) {
|
|
21
|
+
auto baseFont = fontObj.getKey("/BaseFont");
|
|
22
|
+
if (!baseFont.isName())
|
|
23
|
+
return false;
|
|
24
|
+
auto name = baseFont.getName();
|
|
25
|
+
if (!name.empty() && name[0] == '/')
|
|
26
|
+
name = name.substr(1);
|
|
27
|
+
if (name.size() <= 7 || name[6] != '+')
|
|
28
|
+
return false;
|
|
29
|
+
for (int i = 0; i < 6; ++i) {
|
|
30
|
+
if (name[i] < 'A' || name[i] > 'Z')
|
|
31
|
+
return false;
|
|
32
|
+
}
|
|
33
|
+
return true;
|
|
34
|
+
}
|
|
35
|
+
|
|
36
|
+
} // namespace
|
|
37
|
+
|
|
38
|
+
// ---------------------------------------------------------------------------
|
|
39
|
+
// Combined parser callback: collects font names referenced by Tf operators
|
|
40
|
+
// AND character codes used by each font in a single pass
|
|
41
|
+
// ---------------------------------------------------------------------------
|
|
42
|
+
|
|
43
|
+
class FontUsageCollector : public QPDFObjectHandle::ParserCallbacks {
|
|
44
|
+
public:
|
|
45
|
+
std::map<std::string, std::set<uint16_t>> fontCodes;
|
|
46
|
+
std::set<std::string> cidFonts; // set before parsing
|
|
47
|
+
|
|
48
|
+
void handleObject(QPDFObjectHandle obj) override {
|
|
49
|
+
if (obj.isOperator()) {
|
|
50
|
+
std::string op = obj.getOperatorValue();
|
|
51
|
+
if (op == "Tf" && operands.size() >= 2) {
|
|
52
|
+
auto nameObj = operands[operands.size() - 2];
|
|
53
|
+
if (nameObj.isName()) {
|
|
54
|
+
currentFont = nameObj.getName();
|
|
55
|
+
}
|
|
56
|
+
}
|
|
57
|
+
if (!currentFont.empty()) {
|
|
58
|
+
bool isCID = cidFonts.count(currentFont) > 0;
|
|
59
|
+
if (op == "Tj" || op == "'" || op == "\"") {
|
|
60
|
+
if (!operands.empty() && operands.back().isString())
|
|
61
|
+
collectFromString(operands.back(), isCID);
|
|
62
|
+
} else if (op == "TJ") {
|
|
63
|
+
if (!operands.empty() && operands.back().isArray()) {
|
|
64
|
+
auto arr = operands.back();
|
|
65
|
+
for (int i = 0; i < arr.getArrayNItems(); ++i) {
|
|
66
|
+
auto item = arr.getArrayItem(i);
|
|
67
|
+
if (item.isString())
|
|
68
|
+
collectFromString(item, isCID);
|
|
69
|
+
}
|
|
70
|
+
}
|
|
71
|
+
}
|
|
72
|
+
}
|
|
73
|
+
operands.clear();
|
|
74
|
+
} else {
|
|
75
|
+
operands.push_back(obj);
|
|
76
|
+
}
|
|
77
|
+
}
|
|
78
|
+
void handleEOF() override {}
|
|
79
|
+
|
|
80
|
+
private:
|
|
81
|
+
std::string currentFont;
|
|
82
|
+
std::vector<QPDFObjectHandle> operands;
|
|
83
|
+
|
|
84
|
+
void collectFromString(QPDFObjectHandle strObj, bool isCID) {
|
|
85
|
+
std::string raw = strObj.getStringValue();
|
|
86
|
+
if (isCID) {
|
|
87
|
+
for (size_t i = 0; i + 1 < raw.size(); i += 2) {
|
|
88
|
+
uint16_t code = (static_cast<uint8_t>(raw[i]) << 8) |
|
|
89
|
+
static_cast<uint8_t>(raw[i + 1]);
|
|
90
|
+
fontCodes[currentFont].insert(code);
|
|
91
|
+
}
|
|
92
|
+
} else {
|
|
93
|
+
for (unsigned char c : raw)
|
|
94
|
+
fontCodes[currentFont].insert(c);
|
|
95
|
+
}
|
|
96
|
+
}
|
|
97
|
+
};
|
|
98
|
+
|
|
99
|
+
// helper: run FontUsageCollector on a stream, swallowing parse errors
|
|
100
|
+
static void collectFontUsageFromStream(QPDFObjectHandle stream,
|
|
101
|
+
FontUsageCollector &collector) {
|
|
102
|
+
try {
|
|
103
|
+
QPDFObjectHandle::parseContentStream(stream, &collector);
|
|
104
|
+
} catch (...) {
|
|
105
|
+
}
|
|
106
|
+
}
|
|
107
|
+
|
|
108
|
+
// ---------------------------------------------------------------------------
|
|
109
|
+
// /ToUnicode CMap parser — converts character codes to Unicode using the
|
|
110
|
+
// font's /ToUnicode stream. handles beginbfchar, beginbfrange (scalar and
|
|
111
|
+
// array forms). this is the most reliable way to map codes to Unicode.
|
|
112
|
+
// ---------------------------------------------------------------------------
|
|
113
|
+
|
|
114
|
+
static std::set<uint16_t> parseToUnicode(QPDFObjectHandle toUnicodeStream,
|
|
115
|
+
const std::set<uint16_t> &charCodes) {
|
|
116
|
+
std::set<uint16_t> unicodeCodes;
|
|
117
|
+
|
|
118
|
+
try {
|
|
119
|
+
auto data = toUnicodeStream.getStreamData(qpdf_dl_all);
|
|
120
|
+
std::string cmap(reinterpret_cast<const char *>(data->getBuffer()),
|
|
121
|
+
data->getSize());
|
|
122
|
+
|
|
123
|
+
// helper: find <hex> value at/after pos, return parsed value and position
|
|
124
|
+
// after closing >
|
|
125
|
+
auto findHexValue = [&](size_t pos,
|
|
126
|
+
size_t end) -> std::pair<uint16_t, size_t> {
|
|
127
|
+
size_t start = cmap.find('<', pos);
|
|
128
|
+
if (start == std::string::npos || start >= end)
|
|
129
|
+
return {0, std::string::npos};
|
|
130
|
+
size_t stop = cmap.find('>', start);
|
|
131
|
+
if (stop == std::string::npos || stop >= end)
|
|
132
|
+
return {0, std::string::npos};
|
|
133
|
+
std::string hex = cmap.substr(start + 1, stop - start - 1);
|
|
134
|
+
if (hex.empty())
|
|
135
|
+
return {0, stop + 1};
|
|
136
|
+
return {static_cast<uint16_t>(std::stoul(hex, nullptr, 16)), stop + 1};
|
|
137
|
+
};
|
|
138
|
+
|
|
139
|
+
// process beginbfchar sections: <srcCode> <dstUnicode>
|
|
140
|
+
size_t pos = 0;
|
|
141
|
+
while (true) {
|
|
142
|
+
size_t start = cmap.find("beginbfchar", pos);
|
|
143
|
+
if (start == std::string::npos)
|
|
144
|
+
break;
|
|
145
|
+
start += 11;
|
|
146
|
+
size_t end = cmap.find("endbfchar", start);
|
|
147
|
+
if (end == std::string::npos)
|
|
148
|
+
break;
|
|
149
|
+
|
|
150
|
+
size_t p = start;
|
|
151
|
+
while (p < end) {
|
|
152
|
+
auto [src, after1] = findHexValue(p, end);
|
|
153
|
+
if (after1 == std::string::npos)
|
|
154
|
+
break;
|
|
155
|
+
auto [dst, after2] = findHexValue(after1, end);
|
|
156
|
+
if (after2 == std::string::npos)
|
|
157
|
+
break;
|
|
158
|
+
if (charCodes.count(src))
|
|
159
|
+
unicodeCodes.insert(dst);
|
|
160
|
+
p = after2;
|
|
161
|
+
}
|
|
162
|
+
pos = end + 9;
|
|
163
|
+
}
|
|
164
|
+
|
|
165
|
+
// process beginbfrange sections: <lo> <hi> <dstStart> or <lo> <hi> [...]
|
|
166
|
+
pos = 0;
|
|
167
|
+
while (true) {
|
|
168
|
+
size_t start = cmap.find("beginbfrange", pos);
|
|
169
|
+
if (start == std::string::npos)
|
|
170
|
+
break;
|
|
171
|
+
start += 12;
|
|
172
|
+
size_t end = cmap.find("endbfrange", start);
|
|
173
|
+
if (end == std::string::npos)
|
|
174
|
+
break;
|
|
175
|
+
|
|
176
|
+
size_t p = start;
|
|
177
|
+
while (p < end) {
|
|
178
|
+
auto [lo, after1] = findHexValue(p, end);
|
|
179
|
+
if (after1 == std::string::npos)
|
|
180
|
+
break;
|
|
181
|
+
auto [hi, after2] = findHexValue(after1, end);
|
|
182
|
+
if (after2 == std::string::npos)
|
|
183
|
+
break;
|
|
184
|
+
|
|
185
|
+
// check for array form vs scalar form
|
|
186
|
+
size_t next = after2;
|
|
187
|
+
while (next < end &&
|
|
188
|
+
std::isspace(static_cast<unsigned char>(cmap[next])))
|
|
189
|
+
++next;
|
|
190
|
+
|
|
191
|
+
if (next < end && cmap[next] == '[') {
|
|
192
|
+
// array form: <lo> <hi> [<v1> <v2> ...]
|
|
193
|
+
size_t arrEnd = cmap.find(']', next);
|
|
194
|
+
if (arrEnd == std::string::npos || arrEnd >= end)
|
|
195
|
+
break;
|
|
196
|
+
uint16_t code = lo;
|
|
197
|
+
size_t ap = next + 1;
|
|
198
|
+
while (ap < arrEnd && code <= hi) {
|
|
199
|
+
auto [val, afterVal] = findHexValue(ap, arrEnd);
|
|
200
|
+
if (afterVal == std::string::npos)
|
|
201
|
+
break;
|
|
202
|
+
if (charCodes.count(code))
|
|
203
|
+
unicodeCodes.insert(val);
|
|
204
|
+
++code;
|
|
205
|
+
ap = afterVal;
|
|
206
|
+
}
|
|
207
|
+
p = arrEnd + 1;
|
|
208
|
+
} else {
|
|
209
|
+
// scalar form: <lo> <hi> <dstStart>
|
|
210
|
+
auto [dstStart, after3] = findHexValue(after2, end);
|
|
211
|
+
if (after3 == std::string::npos)
|
|
212
|
+
break;
|
|
213
|
+
for (uint16_t c = lo; c <= hi; ++c) {
|
|
214
|
+
if (charCodes.count(c))
|
|
215
|
+
unicodeCodes.insert(static_cast<uint16_t>(dstStart + (c - lo)));
|
|
216
|
+
}
|
|
217
|
+
p = after3;
|
|
218
|
+
}
|
|
219
|
+
}
|
|
220
|
+
pos = end + 10;
|
|
221
|
+
}
|
|
222
|
+
} catch (...) {
|
|
223
|
+
// parsing failed — return what we have so far
|
|
224
|
+
}
|
|
225
|
+
|
|
226
|
+
return unicodeCodes;
|
|
227
|
+
}
|
|
228
|
+
|
|
229
|
+
// ---------------------------------------------------------------------------
|
|
230
|
+
// /Encoding /Differences parser — extract glyph names for character codes
|
|
231
|
+
// that have been remapped via /Differences entries.
|
|
232
|
+
// format: [code1 /name1 /name2 ... code2 /name3 ...]
|
|
233
|
+
// integers set current position; names are assigned sequentially.
|
|
234
|
+
// ---------------------------------------------------------------------------
|
|
235
|
+
|
|
236
|
+
static std::vector<std::string>
|
|
237
|
+
getGlyphNamesFromEncoding(const std::set<uint16_t> &codes,
|
|
238
|
+
QPDFObjectHandle fontObj) {
|
|
239
|
+
std::vector<std::string> names;
|
|
240
|
+
|
|
241
|
+
auto encoding = fontObj.getKey("/Encoding");
|
|
242
|
+
if (!encoding.isDictionary())
|
|
243
|
+
return names;
|
|
244
|
+
|
|
245
|
+
auto diffs = encoding.getKey("/Differences");
|
|
246
|
+
if (!diffs.isArray())
|
|
247
|
+
return names;
|
|
248
|
+
|
|
249
|
+
// parse /Differences: integers set position, names are glyph names
|
|
250
|
+
std::map<uint16_t, std::string> codeToName;
|
|
251
|
+
int currentCode = 0;
|
|
252
|
+
for (int i = 0; i < diffs.getArrayNItems(); ++i) {
|
|
253
|
+
auto item = diffs.getArrayItem(i);
|
|
254
|
+
if (item.isInteger()) {
|
|
255
|
+
currentCode = static_cast<int>(item.getIntValue());
|
|
256
|
+
} else if (item.isName()) {
|
|
257
|
+
std::string name = item.getName();
|
|
258
|
+
// strip leading / from PDF name
|
|
259
|
+
if (!name.empty() && name[0] == '/')
|
|
260
|
+
name = name.substr(1);
|
|
261
|
+
codeToName[static_cast<uint16_t>(currentCode)] = name;
|
|
262
|
+
++currentCode;
|
|
263
|
+
}
|
|
264
|
+
}
|
|
265
|
+
|
|
266
|
+
// return glyph names for used codes that have /Differences entries
|
|
267
|
+
for (uint16_t code : codes) {
|
|
268
|
+
auto it = codeToName.find(code);
|
|
269
|
+
if (it != codeToName.end())
|
|
270
|
+
names.push_back(it->second);
|
|
271
|
+
}
|
|
272
|
+
|
|
273
|
+
return names;
|
|
274
|
+
}
|
|
275
|
+
|
|
276
|
+
// ---------------------------------------------------------------------------
|
|
277
|
+
// Encoding helpers — convert encoding-specific byte codes to Unicode
|
|
278
|
+
// for correct cmap lookups during font subsetting
|
|
279
|
+
// ---------------------------------------------------------------------------
|
|
280
|
+
|
|
281
|
+
// convert Adobe Glyph List (AGL) name to Unicode codepoint. handles:
|
|
282
|
+
// - "uniXXXX" convention (4 hex digits after "uni")
|
|
283
|
+
// - "uXXXX"/"uXXXXX" convention (4-6 hex digits after "u")
|
|
284
|
+
// - standard AGL names for common Latin/currency/typographic characters
|
|
285
|
+
static uint16_t glyphNameToUnicode(const std::string &name) {
|
|
286
|
+
// strip any variant suffix (e.g., "Euro.oldstyle" → "Euro")
|
|
287
|
+
std::string base = name;
|
|
288
|
+
auto dot = base.find('.');
|
|
289
|
+
if (dot != std::string::npos)
|
|
290
|
+
base = base.substr(0, dot);
|
|
291
|
+
|
|
292
|
+
// "uniXXXX" — standard Unicode naming convention
|
|
293
|
+
if (base.size() == 7 && base[0] == 'u' && base[1] == 'n' && base[2] == 'i') {
|
|
294
|
+
char *end = nullptr;
|
|
295
|
+
unsigned long val = strtoul(base.c_str() + 3, &end, 16);
|
|
296
|
+
if (end == base.c_str() + 7 && val > 0 && val <= 0xFFFF)
|
|
297
|
+
return static_cast<uint16_t>(val);
|
|
298
|
+
}
|
|
299
|
+
|
|
300
|
+
// "uXXXX" or "uXXXXX" — alternate Unicode naming convention
|
|
301
|
+
if (base.size() >= 5 && base.size() <= 7 && base[0] == 'u' &&
|
|
302
|
+
base[1] != 'n') {
|
|
303
|
+
char *end = nullptr;
|
|
304
|
+
unsigned long val = strtoul(base.c_str() + 1, &end, 16);
|
|
305
|
+
if (end == base.c_str() + static_cast<ptrdiff_t>(base.size()) && val > 0 &&
|
|
306
|
+
val <= 0xFFFF)
|
|
307
|
+
return static_cast<uint16_t>(val);
|
|
308
|
+
}
|
|
309
|
+
|
|
310
|
+
// single-character names map to their ASCII value
|
|
311
|
+
if (base.size() == 1 && base[0] >= 0x20)
|
|
312
|
+
return static_cast<uint16_t>(static_cast<unsigned char>(base[0]));
|
|
313
|
+
|
|
314
|
+
// standard AGL names — covers WinAnsi, Latin Extended, and common symbols
|
|
315
|
+
// sourced from the Adobe Glyph List (agl-aglfn) specification
|
|
316
|
+
static const std::map<std::string, uint16_t> agl = {
|
|
317
|
+
// basic ASCII names
|
|
318
|
+
{"space", 0x0020},
|
|
319
|
+
{"exclam", 0x0021},
|
|
320
|
+
{"quotedbl", 0x0022},
|
|
321
|
+
{"numbersign", 0x0023},
|
|
322
|
+
{"dollar", 0x0024},
|
|
323
|
+
{"percent", 0x0025},
|
|
324
|
+
{"ampersand", 0x0026},
|
|
325
|
+
{"quotesingle", 0x0027},
|
|
326
|
+
{"parenleft", 0x0028},
|
|
327
|
+
{"parenright", 0x0029},
|
|
328
|
+
{"asterisk", 0x002A},
|
|
329
|
+
{"plus", 0x002B},
|
|
330
|
+
{"comma", 0x002C},
|
|
331
|
+
{"hyphen", 0x002D},
|
|
332
|
+
{"period", 0x002E},
|
|
333
|
+
{"slash", 0x002F},
|
|
334
|
+
{"zero", 0x0030},
|
|
335
|
+
{"one", 0x0031},
|
|
336
|
+
{"two", 0x0032},
|
|
337
|
+
{"three", 0x0033},
|
|
338
|
+
{"four", 0x0034},
|
|
339
|
+
{"five", 0x0035},
|
|
340
|
+
{"six", 0x0036},
|
|
341
|
+
{"seven", 0x0037},
|
|
342
|
+
{"eight", 0x0038},
|
|
343
|
+
{"nine", 0x0039},
|
|
344
|
+
{"colon", 0x003A},
|
|
345
|
+
{"semicolon", 0x003B},
|
|
346
|
+
{"less", 0x003C},
|
|
347
|
+
{"equal", 0x003D},
|
|
348
|
+
{"greater", 0x003E},
|
|
349
|
+
{"question", 0x003F},
|
|
350
|
+
{"at", 0x0040},
|
|
351
|
+
{"bracketleft", 0x005B},
|
|
352
|
+
{"backslash", 0x005C},
|
|
353
|
+
{"bracketright", 0x005D},
|
|
354
|
+
{"asciicircum", 0x005E},
|
|
355
|
+
{"underscore", 0x005F},
|
|
356
|
+
{"grave", 0x0060},
|
|
357
|
+
{"braceleft", 0x007B},
|
|
358
|
+
{"bar", 0x007C},
|
|
359
|
+
{"braceright", 0x007D},
|
|
360
|
+
{"asciitilde", 0x007E},
|
|
361
|
+
// WinAnsi 0x80-0x9F range
|
|
362
|
+
{"Euro", 0x20AC},
|
|
363
|
+
{"quotesinglbase", 0x201A},
|
|
364
|
+
{"florin", 0x0192},
|
|
365
|
+
{"quotedblbase", 0x201E},
|
|
366
|
+
{"ellipsis", 0x2026},
|
|
367
|
+
{"dagger", 0x2020},
|
|
368
|
+
{"daggerdbl", 0x2021},
|
|
369
|
+
{"circumflex", 0x02C6},
|
|
370
|
+
{"perthousand", 0x2030},
|
|
371
|
+
{"Scaron", 0x0160},
|
|
372
|
+
{"guilsinglleft", 0x2039},
|
|
373
|
+
{"OE", 0x0152},
|
|
374
|
+
{"Zcaron", 0x017D},
|
|
375
|
+
{"quoteleft", 0x2018},
|
|
376
|
+
{"quoteright", 0x2019},
|
|
377
|
+
{"quotedblleft", 0x201C},
|
|
378
|
+
{"quotedblright", 0x201D},
|
|
379
|
+
{"bullet", 0x2022},
|
|
380
|
+
{"endash", 0x2013},
|
|
381
|
+
{"emdash", 0x2014},
|
|
382
|
+
{"tilde", 0x02DC},
|
|
383
|
+
{"trademark", 0x2122},
|
|
384
|
+
{"scaron", 0x0161},
|
|
385
|
+
{"guilsinglright", 0x203A},
|
|
386
|
+
{"oe", 0x0153},
|
|
387
|
+
{"zcaron", 0x017E},
|
|
388
|
+
{"Ydieresis", 0x0178},
|
|
389
|
+
// Latin-1 Supplement (0xA0-0xFF)
|
|
390
|
+
{"nbspace", 0x00A0},
|
|
391
|
+
{"exclamdown", 0x00A1},
|
|
392
|
+
{"cent", 0x00A2},
|
|
393
|
+
{"sterling", 0x00A3},
|
|
394
|
+
{"currency", 0x00A4},
|
|
395
|
+
{"yen", 0x00A5},
|
|
396
|
+
{"brokenbar", 0x00A6},
|
|
397
|
+
{"section", 0x00A7},
|
|
398
|
+
{"dieresis", 0x00A8},
|
|
399
|
+
{"copyright", 0x00A9},
|
|
400
|
+
{"ordfeminine", 0x00AA},
|
|
401
|
+
{"guillemotleft", 0x00AB},
|
|
402
|
+
{"logicalnot", 0x00AC},
|
|
403
|
+
{"softhyphen", 0x00AD},
|
|
404
|
+
{"registered", 0x00AE},
|
|
405
|
+
{"macron", 0x00AF},
|
|
406
|
+
{"degree", 0x00B0},
|
|
407
|
+
{"plusminus", 0x00B1},
|
|
408
|
+
{"twosuperior", 0x00B2},
|
|
409
|
+
{"threesuperior", 0x00B3},
|
|
410
|
+
{"acute", 0x00B4},
|
|
411
|
+
{"mu", 0x00B5},
|
|
412
|
+
{"paragraph", 0x00B6},
|
|
413
|
+
{"periodcentered", 0x00B7},
|
|
414
|
+
{"cedilla", 0x00B8},
|
|
415
|
+
{"onesuperior", 0x00B9},
|
|
416
|
+
{"ordmasculine", 0x00BA},
|
|
417
|
+
{"guillemotright", 0x00BB},
|
|
418
|
+
{"onequarter", 0x00BC},
|
|
419
|
+
{"onehalf", 0x00BD},
|
|
420
|
+
{"threequarters", 0x00BE},
|
|
421
|
+
{"questiondown", 0x00BF},
|
|
422
|
+
{"Agrave", 0x00C0},
|
|
423
|
+
{"Aacute", 0x00C1},
|
|
424
|
+
{"Acircumflex", 0x00C2},
|
|
425
|
+
{"Atilde", 0x00C3},
|
|
426
|
+
{"Adieresis", 0x00C4},
|
|
427
|
+
{"Aring", 0x00C5},
|
|
428
|
+
{"AE", 0x00C6},
|
|
429
|
+
{"Ccedilla", 0x00C7},
|
|
430
|
+
{"Egrave", 0x00C8},
|
|
431
|
+
{"Eacute", 0x00C9},
|
|
432
|
+
{"Ecircumflex", 0x00CA},
|
|
433
|
+
{"Edieresis", 0x00CB},
|
|
434
|
+
{"Igrave", 0x00CC},
|
|
435
|
+
{"Iacute", 0x00CD},
|
|
436
|
+
{"Icircumflex", 0x00CE},
|
|
437
|
+
{"Idieresis", 0x00CF},
|
|
438
|
+
{"Eth", 0x00D0},
|
|
439
|
+
{"Ntilde", 0x00D1},
|
|
440
|
+
{"Ograve", 0x00D2},
|
|
441
|
+
{"Oacute", 0x00D3},
|
|
442
|
+
{"Ocircumflex", 0x00D4},
|
|
443
|
+
{"Otilde", 0x00D5},
|
|
444
|
+
{"Odieresis", 0x00D6},
|
|
445
|
+
{"multiply", 0x00D7},
|
|
446
|
+
{"Oslash", 0x00D8},
|
|
447
|
+
{"Ugrave", 0x00D9},
|
|
448
|
+
{"Uacute", 0x00DA},
|
|
449
|
+
{"Ucircumflex", 0x00DB},
|
|
450
|
+
{"Udieresis", 0x00DC},
|
|
451
|
+
{"Yacute", 0x00DD},
|
|
452
|
+
{"Thorn", 0x00DE},
|
|
453
|
+
{"germandbls", 0x00DF},
|
|
454
|
+
{"agrave", 0x00E0},
|
|
455
|
+
{"aacute", 0x00E1},
|
|
456
|
+
{"acircumflex", 0x00E2},
|
|
457
|
+
{"atilde", 0x00E3},
|
|
458
|
+
{"adieresis", 0x00E4},
|
|
459
|
+
{"aring", 0x00E5},
|
|
460
|
+
{"ae", 0x00E6},
|
|
461
|
+
{"ccedilla", 0x00E7},
|
|
462
|
+
{"egrave", 0x00E8},
|
|
463
|
+
{"eacute", 0x00E9},
|
|
464
|
+
{"ecircumflex", 0x00EA},
|
|
465
|
+
{"edieresis", 0x00EB},
|
|
466
|
+
{"igrave", 0x00EC},
|
|
467
|
+
{"iacute", 0x00ED},
|
|
468
|
+
{"icircumflex", 0x00EE},
|
|
469
|
+
{"idieresis", 0x00EF},
|
|
470
|
+
{"eth", 0x00F0},
|
|
471
|
+
{"ntilde", 0x00F1},
|
|
472
|
+
{"ograve", 0x00F2},
|
|
473
|
+
{"oacute", 0x00F3},
|
|
474
|
+
{"ocircumflex", 0x00F4},
|
|
475
|
+
{"otilde", 0x00F5},
|
|
476
|
+
{"odieresis", 0x00F6},
|
|
477
|
+
{"divide", 0x00F7},
|
|
478
|
+
{"oslash", 0x00F8},
|
|
479
|
+
{"ugrave", 0x00F9},
|
|
480
|
+
{"uacute", 0x00FA},
|
|
481
|
+
{"ucircumflex", 0x00FB},
|
|
482
|
+
{"udieresis", 0x00FC},
|
|
483
|
+
{"yacute", 0x00FD},
|
|
484
|
+
{"thorn", 0x00FE},
|
|
485
|
+
{"ydieresis", 0x00FF},
|
|
486
|
+
// Latin Extended-A (commonly found in CE font /Differences)
|
|
487
|
+
{"Amacron", 0x0100},
|
|
488
|
+
{"amacron", 0x0101},
|
|
489
|
+
{"Abreve", 0x0102},
|
|
490
|
+
{"abreve", 0x0103},
|
|
491
|
+
{"Aogonek", 0x0104},
|
|
492
|
+
{"aogonek", 0x0105},
|
|
493
|
+
{"Cacute", 0x0106},
|
|
494
|
+
{"cacute", 0x0107},
|
|
495
|
+
{"Ccircumflex", 0x0108},
|
|
496
|
+
{"ccircumflex", 0x0109},
|
|
497
|
+
{"Cdotaccent", 0x010A},
|
|
498
|
+
{"cdotaccent", 0x010B},
|
|
499
|
+
{"Ccaron", 0x010C},
|
|
500
|
+
{"ccaron", 0x010D},
|
|
501
|
+
{"Dcaron", 0x010E},
|
|
502
|
+
{"dcaron", 0x010F},
|
|
503
|
+
{"Dcroat", 0x0110},
|
|
504
|
+
{"dcroat", 0x0111},
|
|
505
|
+
{"Emacron", 0x0112},
|
|
506
|
+
{"emacron", 0x0113},
|
|
507
|
+
{"Ebreve", 0x0114},
|
|
508
|
+
{"ebreve", 0x0115},
|
|
509
|
+
{"Edotaccent", 0x0116},
|
|
510
|
+
{"edotaccent", 0x0117},
|
|
511
|
+
{"Eogonek", 0x0118},
|
|
512
|
+
{"eogonek", 0x0119},
|
|
513
|
+
{"Ecaron", 0x011A},
|
|
514
|
+
{"ecaron", 0x011B},
|
|
515
|
+
{"Gbreve", 0x011E},
|
|
516
|
+
{"gbreve", 0x011F},
|
|
517
|
+
{"Gdotaccent", 0x0120},
|
|
518
|
+
{"gdotaccent", 0x0121},
|
|
519
|
+
{"Gcommaaccent", 0x0122},
|
|
520
|
+
{"gcommaaccent", 0x0123},
|
|
521
|
+
{"Idotaccent", 0x0130},
|
|
522
|
+
{"dotlessi", 0x0131},
|
|
523
|
+
{"IJ", 0x0132},
|
|
524
|
+
{"ij", 0x0133},
|
|
525
|
+
{"Lacute", 0x0139},
|
|
526
|
+
{"lacute", 0x013A},
|
|
527
|
+
{"Lcommaaccent", 0x013B},
|
|
528
|
+
{"lcommaaccent", 0x013C},
|
|
529
|
+
{"Lcaron", 0x013D},
|
|
530
|
+
{"lcaron", 0x013E},
|
|
531
|
+
{"Ldot", 0x013F},
|
|
532
|
+
{"ldot", 0x0140},
|
|
533
|
+
{"Lslash", 0x0141},
|
|
534
|
+
{"lslash", 0x0142},
|
|
535
|
+
{"Nacute", 0x0143},
|
|
536
|
+
{"nacute", 0x0144},
|
|
537
|
+
{"Ncommaaccent", 0x0145},
|
|
538
|
+
{"ncommaaccent", 0x0146},
|
|
539
|
+
{"Ncaron", 0x0147},
|
|
540
|
+
{"ncaron", 0x0148},
|
|
541
|
+
{"Eng", 0x014A},
|
|
542
|
+
{"eng", 0x014B},
|
|
543
|
+
{"Omacron", 0x014C},
|
|
544
|
+
{"omacron", 0x014D},
|
|
545
|
+
{"Obreve", 0x014E},
|
|
546
|
+
{"obreve", 0x014F},
|
|
547
|
+
{"Ohungarumlaut", 0x0150},
|
|
548
|
+
{"ohungarumlaut", 0x0151},
|
|
549
|
+
{"Racute", 0x0154},
|
|
550
|
+
{"racute", 0x0155},
|
|
551
|
+
{"Rcommaaccent", 0x0156},
|
|
552
|
+
{"rcommaaccent", 0x0157},
|
|
553
|
+
{"Rcaron", 0x0158},
|
|
554
|
+
{"rcaron", 0x0159},
|
|
555
|
+
{"Sacute", 0x015A},
|
|
556
|
+
{"sacute", 0x015B},
|
|
557
|
+
{"Scircumflex", 0x015C},
|
|
558
|
+
{"scircumflex", 0x015D},
|
|
559
|
+
{"Scedilla", 0x015E},
|
|
560
|
+
{"scedilla", 0x015F},
|
|
561
|
+
{"Tcaron", 0x0164},
|
|
562
|
+
{"tcaron", 0x0165},
|
|
563
|
+
{"Tbar", 0x0166},
|
|
564
|
+
{"tbar", 0x0167},
|
|
565
|
+
{"Umacron", 0x016A},
|
|
566
|
+
{"umacron", 0x016B},
|
|
567
|
+
{"Ubreve", 0x016C},
|
|
568
|
+
{"ubreve", 0x016D},
|
|
569
|
+
{"Uring", 0x016E},
|
|
570
|
+
{"uring", 0x016F},
|
|
571
|
+
{"Uhungarumlaut", 0x0170},
|
|
572
|
+
{"uhungarumlaut", 0x0171},
|
|
573
|
+
{"Uogonek", 0x0172},
|
|
574
|
+
{"uogonek", 0x0173},
|
|
575
|
+
{"Zacute", 0x0179},
|
|
576
|
+
{"zacute", 0x017A},
|
|
577
|
+
{"Zdotaccent", 0x017B},
|
|
578
|
+
{"zdotaccent", 0x017C},
|
|
579
|
+
// common currency symbols
|
|
580
|
+
{"colonmonetary", 0x20A1},
|
|
581
|
+
{"franc", 0x20A3},
|
|
582
|
+
{"lira", 0x20A4},
|
|
583
|
+
{"peseta", 0x20A7},
|
|
584
|
+
{"won", 0x20A9},
|
|
585
|
+
{"dong", 0x20AB},
|
|
586
|
+
// typographic symbols
|
|
587
|
+
{"fi", 0xFB01},
|
|
588
|
+
{"fl", 0xFB02},
|
|
589
|
+
{"minus", 0x2212},
|
|
590
|
+
{"fraction", 0x2044},
|
|
591
|
+
};
|
|
592
|
+
|
|
593
|
+
auto it = agl.find(base);
|
|
594
|
+
if (it != agl.end())
|
|
595
|
+
return it->second;
|
|
596
|
+
return 0;
|
|
597
|
+
}
|
|
598
|
+
|
|
599
|
+
// parse /Encoding /Differences to build a map of code → glyph name.
|
|
600
|
+
// used to identify which codes are remapped by /Differences.
|
|
601
|
+
static std::map<uint16_t, std::string>
|
|
602
|
+
getDifferencesCodeMap(QPDFObjectHandle fontObj) {
|
|
603
|
+
std::map<uint16_t, std::string> codeToName;
|
|
604
|
+
|
|
605
|
+
auto encoding = fontObj.getKey("/Encoding");
|
|
606
|
+
if (!encoding.isDictionary())
|
|
607
|
+
return codeToName;
|
|
608
|
+
|
|
609
|
+
auto diffs = encoding.getKey("/Differences");
|
|
610
|
+
if (!diffs.isArray())
|
|
611
|
+
return codeToName;
|
|
612
|
+
|
|
613
|
+
int currentCode = 0;
|
|
614
|
+
for (int i = 0; i < diffs.getArrayNItems(); ++i) {
|
|
615
|
+
auto item = diffs.getArrayItem(i);
|
|
616
|
+
if (item.isInteger()) {
|
|
617
|
+
currentCode = static_cast<int>(item.getIntValue());
|
|
618
|
+
} else if (item.isName()) {
|
|
619
|
+
std::string name = item.getName();
|
|
620
|
+
if (!name.empty() && name[0] == '/')
|
|
621
|
+
name = name.substr(1);
|
|
622
|
+
codeToName[static_cast<uint16_t>(currentCode)] = name;
|
|
623
|
+
++currentCode;
|
|
624
|
+
}
|
|
625
|
+
}
|
|
626
|
+
return codeToName;
|
|
627
|
+
}
|
|
628
|
+
|
|
629
|
+
// convert encoding-specific character codes to Unicode for cmap lookup
|
|
630
|
+
static std::set<uint16_t> convertCodesToUnicode(const std::set<uint16_t> &codes,
|
|
631
|
+
QPDFObjectHandle fontObj) {
|
|
632
|
+
auto encoding = fontObj.getKey("/Encoding");
|
|
633
|
+
|
|
634
|
+
bool isWinAnsi = false;
|
|
635
|
+
bool isMacRoman = false;
|
|
636
|
+
if (encoding.isName()) {
|
|
637
|
+
if (encoding.getName() == "/WinAnsiEncoding")
|
|
638
|
+
isWinAnsi = true;
|
|
639
|
+
else if (encoding.getName() == "/MacRomanEncoding")
|
|
640
|
+
isMacRoman = true;
|
|
641
|
+
} else if (encoding.isDictionary()) {
|
|
642
|
+
auto baseEnc = encoding.getKey("/BaseEncoding");
|
|
643
|
+
if (baseEnc.isName()) {
|
|
644
|
+
if (baseEnc.getName() == "/WinAnsiEncoding")
|
|
645
|
+
isWinAnsi = true;
|
|
646
|
+
else if (baseEnc.getName() == "/MacRomanEncoding")
|
|
647
|
+
isMacRoman = true;
|
|
648
|
+
}
|
|
649
|
+
}
|
|
650
|
+
|
|
651
|
+
if (!isWinAnsi && !isMacRoman)
|
|
652
|
+
return codes;
|
|
653
|
+
|
|
654
|
+
std::set<uint16_t> unicodeCodes;
|
|
655
|
+
for (uint16_t code : codes) {
|
|
656
|
+
if (isWinAnsi)
|
|
657
|
+
unicodeCodes.insert(winAnsiToUnicode(static_cast<uint8_t>(code)));
|
|
658
|
+
else
|
|
659
|
+
unicodeCodes.insert(macRomanToUnicode(static_cast<uint8_t>(code)));
|
|
660
|
+
}
|
|
661
|
+
return unicodeCodes;
|
|
662
|
+
}
|
|
663
|
+
|
|
664
|
+
// ---------------------------------------------------------------------------
|
|
665
|
+
// Combined font optimization — remove unused fonts AND subset remaining
|
|
666
|
+
// fonts in a single page walk
|
|
667
|
+
// ---------------------------------------------------------------------------
|
|
668
|
+
|
|
669
|
+
void optimizeFonts(QPDF &qpdf) {
|
|
670
|
+
// single page walk: collect used font names + character codes per font
|
|
671
|
+
std::map<QPDFObjGen, std::set<uint16_t>> fontUsedCodes;
|
|
672
|
+
|
|
673
|
+
for (auto &page : QPDFPageDocumentHelper(qpdf).getAllPages()) {
|
|
674
|
+
auto pageObj = page.getObjectHandle();
|
|
675
|
+
auto resources = pageObj.getKey("/Resources");
|
|
676
|
+
if (!resources.isDictionary())
|
|
677
|
+
continue;
|
|
678
|
+
auto fonts = resources.getKey("/Font");
|
|
679
|
+
if (!fonts.isDictionary())
|
|
680
|
+
continue;
|
|
681
|
+
|
|
682
|
+
// build list of (stream, fontDict) pairs to scan
|
|
683
|
+
struct StreamFonts {
|
|
684
|
+
QPDFObjectHandle stream;
|
|
685
|
+
QPDFObjectHandle fontDict;
|
|
686
|
+
};
|
|
687
|
+
std::vector<StreamFonts> scanTargets;
|
|
688
|
+
|
|
689
|
+
try {
|
|
690
|
+
auto contents = pageObj.getKey("/Contents");
|
|
691
|
+
scanTargets.push_back({contents, fonts});
|
|
692
|
+
} catch (...) {
|
|
693
|
+
continue;
|
|
694
|
+
}
|
|
695
|
+
|
|
696
|
+
auto xobjects = resources.getKey("/XObject");
|
|
697
|
+
if (xobjects.isDictionary()) {
|
|
698
|
+
// recursively scan Form XObjects (queue-based to handle nesting)
|
|
699
|
+
std::vector<QPDFObjectHandle> xobjQueue;
|
|
700
|
+
std::set<QPDFObjGen> visitedXObjects;
|
|
701
|
+
for (auto &xobjKey : xobjects.getKeys())
|
|
702
|
+
xobjQueue.push_back(xobjects.getKey(xobjKey));
|
|
703
|
+
|
|
704
|
+
while (!xobjQueue.empty()) {
|
|
705
|
+
auto xobj = xobjQueue.back();
|
|
706
|
+
xobjQueue.pop_back();
|
|
707
|
+
if (!xobj.isStream())
|
|
708
|
+
continue;
|
|
709
|
+
auto xobjOg = xobj.getObjGen();
|
|
710
|
+
if (visitedXObjects.count(xobjOg))
|
|
711
|
+
continue;
|
|
712
|
+
visitedXObjects.insert(xobjOg);
|
|
713
|
+
|
|
714
|
+
auto xobjDict = xobj.getDict();
|
|
715
|
+
auto xobjSubtype = xobjDict.getKey("/Subtype");
|
|
716
|
+
if (!xobjSubtype.isName() || xobjSubtype.getName() != "/Form")
|
|
717
|
+
continue;
|
|
718
|
+
|
|
719
|
+
auto xobjRes = xobjDict.getKey("/Resources");
|
|
720
|
+
if (xobjRes.isDictionary() && xobjRes.getKey("/Font").isDictionary())
|
|
721
|
+
scanTargets.push_back({xobj, xobjRes.getKey("/Font")});
|
|
722
|
+
else
|
|
723
|
+
scanTargets.push_back({xobj, fonts});
|
|
724
|
+
|
|
725
|
+
// queue nested XObjects for recursive scanning
|
|
726
|
+
if (xobjRes.isDictionary()) {
|
|
727
|
+
auto nestedXObjects = xobjRes.getKey("/XObject");
|
|
728
|
+
if (nestedXObjects.isDictionary()) {
|
|
729
|
+
for (auto &nestedKey : nestedXObjects.getKeys())
|
|
730
|
+
xobjQueue.push_back(nestedXObjects.getKey(nestedKey));
|
|
731
|
+
}
|
|
732
|
+
}
|
|
733
|
+
}
|
|
734
|
+
}
|
|
735
|
+
|
|
736
|
+
// determine which fonts are CID for the collector
|
|
737
|
+
FontUsageCollector collector;
|
|
738
|
+
for (auto &target : scanTargets) {
|
|
739
|
+
for (auto &key : target.fontDict.getKeys()) {
|
|
740
|
+
auto fontObj = target.fontDict.getKey(key);
|
|
741
|
+
if (!fontObj.isDictionary())
|
|
742
|
+
continue;
|
|
743
|
+
auto subtypeKey = fontObj.getKey("/Subtype");
|
|
744
|
+
if (subtypeKey.isName() && subtypeKey.getName() == "/Type0")
|
|
745
|
+
collector.cidFonts.insert(key);
|
|
746
|
+
}
|
|
747
|
+
}
|
|
748
|
+
|
|
749
|
+
// parse all content streams for this page in one pass per stream
|
|
750
|
+
for (auto &target : scanTargets)
|
|
751
|
+
collectFontUsageFromStream(target.stream, collector);
|
|
752
|
+
|
|
753
|
+
// map per-name codes to per-ObjGen codes for subsetting
|
|
754
|
+
for (auto &target : scanTargets) {
|
|
755
|
+
for (auto &key : target.fontDict.getKeys()) {
|
|
756
|
+
auto it = collector.fontCodes.find(key);
|
|
757
|
+
if (it == collector.fontCodes.end())
|
|
758
|
+
continue;
|
|
759
|
+
auto fontObj = target.fontDict.getKey(key);
|
|
760
|
+
if (!fontObj.isDictionary())
|
|
761
|
+
continue;
|
|
762
|
+
auto og = fontObj.getObjGen();
|
|
763
|
+
fontUsedCodes[og].insert(it->second.begin(), it->second.end());
|
|
764
|
+
}
|
|
765
|
+
}
|
|
766
|
+
}
|
|
767
|
+
|
|
768
|
+
// for each font with a /Widths array, zero out widths for unused glyphs
|
|
769
|
+
// and truncate trailing zeros
|
|
770
|
+
for (auto &[og, usedCodes] : fontUsedCodes) {
|
|
771
|
+
auto fontObj = qpdf.getObjectByObjGen(og);
|
|
772
|
+
if (!fontObj.isDictionary())
|
|
773
|
+
continue;
|
|
774
|
+
|
|
775
|
+
auto subtype = fontObj.getKey("/Subtype");
|
|
776
|
+
if (!subtype.isName())
|
|
777
|
+
continue;
|
|
778
|
+
|
|
779
|
+
// only handle simple fonts with /Widths arrays (TrueType, Type1)
|
|
780
|
+
if (subtype.getName() != "/TrueType" && subtype.getName() != "/Type1")
|
|
781
|
+
continue;
|
|
782
|
+
|
|
783
|
+
// skip already-subset fonts — their encodings are custom and our usage
|
|
784
|
+
// collector may miscount character codes
|
|
785
|
+
if (isAlreadySubset(fontObj))
|
|
786
|
+
continue;
|
|
787
|
+
|
|
788
|
+
auto widths = fontObj.getKey("/Widths");
|
|
789
|
+
auto firstCharObj = fontObj.getKey("/FirstChar");
|
|
790
|
+
if (!widths.isArray() || !firstCharObj.isInteger())
|
|
791
|
+
continue;
|
|
792
|
+
|
|
793
|
+
int firstChar = static_cast<int>(firstCharObj.getIntValue());
|
|
794
|
+
int widthCount = widths.getArrayNItems();
|
|
795
|
+
|
|
796
|
+
// zero out widths for unused character codes
|
|
797
|
+
bool modified = false;
|
|
798
|
+
for (int i = 0; i < widthCount; ++i) {
|
|
799
|
+
int charCode = firstChar + i;
|
|
800
|
+
if (usedCodes.find(static_cast<uint16_t>(charCode)) == usedCodes.end()) {
|
|
801
|
+
auto w = widths.getArrayItem(i);
|
|
802
|
+
if (w.isInteger() && w.getIntValue() != 0) {
|
|
803
|
+
widths.setArrayItem(i, QPDFObjectHandle::newInteger(0));
|
|
804
|
+
modified = true;
|
|
805
|
+
} else if (w.isReal() && w.getNumericValue() != 0.0) {
|
|
806
|
+
widths.setArrayItem(i, QPDFObjectHandle::newInteger(0));
|
|
807
|
+
modified = true;
|
|
808
|
+
}
|
|
809
|
+
}
|
|
810
|
+
}
|
|
811
|
+
if (!modified)
|
|
812
|
+
continue;
|
|
813
|
+
|
|
814
|
+
// trim trailing zero-width entries and adjust /LastChar
|
|
815
|
+
int lastUsed = widthCount - 1;
|
|
816
|
+
while (lastUsed >= 0) {
|
|
817
|
+
auto w = widths.getArrayItem(lastUsed);
|
|
818
|
+
if (w.isInteger() && w.getIntValue() == 0)
|
|
819
|
+
--lastUsed;
|
|
820
|
+
else
|
|
821
|
+
break;
|
|
822
|
+
}
|
|
823
|
+
|
|
824
|
+
if (lastUsed < widthCount - 1) {
|
|
825
|
+
// rebuild the widths array with only the needed entries
|
|
826
|
+
auto newWidths = QPDFObjectHandle::newArray();
|
|
827
|
+
for (int i = 0; i <= lastUsed; ++i)
|
|
828
|
+
newWidths.appendItem(widths.getArrayItem(i));
|
|
829
|
+
|
|
830
|
+
fontObj.replaceKey("/Widths", newWidths);
|
|
831
|
+
fontObj.replaceKey("/LastChar",
|
|
832
|
+
QPDFObjectHandle::newInteger(firstChar + lastUsed));
|
|
833
|
+
}
|
|
834
|
+
}
|
|
835
|
+
|
|
836
|
+
// true font subsetting: strip unused glyph outlines from embedded fonts.
|
|
837
|
+
// two-pass approach: first collect glyph IDs per font file (merging across
|
|
838
|
+
// all font objects sharing the same embedded file), then subset each file.
|
|
839
|
+
std::set<QPDFObjGen> processedFonts;
|
|
840
|
+
std::map<QPDFObjGen, std::set<uint16_t>> fontFileGlyphIds;
|
|
841
|
+
std::map<QPDFObjGen, QPDFObjectHandle> fontFileHandles;
|
|
842
|
+
std::map<QPDFObjGen, bool> fontFileCID; // true if CID font
|
|
843
|
+
std::set<QPDFObjGen> fontFileSkip; // skip if any font fails safety check
|
|
844
|
+
|
|
845
|
+
// pass 1: collect glyph IDs from each font object
|
|
846
|
+
for (auto &[og, usedCodes] : fontUsedCodes) {
|
|
847
|
+
if (processedFonts.count(og))
|
|
848
|
+
continue;
|
|
849
|
+
processedFonts.insert(og);
|
|
850
|
+
|
|
851
|
+
auto fontObj = qpdf.getObjectByObjGen(og);
|
|
852
|
+
if (!fontObj.isDictionary())
|
|
853
|
+
continue;
|
|
854
|
+
|
|
855
|
+
auto subtype = fontObj.getKey("/Subtype");
|
|
856
|
+
if (!subtype.isName())
|
|
857
|
+
continue;
|
|
858
|
+
// Type0 composite fonts — subset CIDFontType2 and CIDFontType0
|
|
859
|
+
// descendants
|
|
860
|
+
if (subtype.getName() == "/Type0") {
|
|
861
|
+
// skip already-subset CID fonts
|
|
862
|
+
if (isAlreadySubset(fontObj))
|
|
863
|
+
continue;
|
|
864
|
+
|
|
865
|
+
auto descendants = fontObj.getKey("/DescendantFonts");
|
|
866
|
+
if (!descendants.isArray() || descendants.getArrayNItems() < 1)
|
|
867
|
+
continue;
|
|
868
|
+
|
|
869
|
+
auto cidFont = descendants.getArrayItem(0);
|
|
870
|
+
if (!cidFont.isDictionary())
|
|
871
|
+
continue;
|
|
872
|
+
|
|
873
|
+
if (isAlreadySubset(cidFont))
|
|
874
|
+
continue;
|
|
875
|
+
|
|
876
|
+
auto cidSubtype = cidFont.getKey("/Subtype");
|
|
877
|
+
if (!cidSubtype.isName())
|
|
878
|
+
continue;
|
|
879
|
+
|
|
880
|
+
bool isCIDFontType2 = cidSubtype.getName() == "/CIDFontType2";
|
|
881
|
+
bool isCIDFontType0 = cidSubtype.getName() == "/CIDFontType0";
|
|
882
|
+
if (!isCIDFontType2 && !isCIDFontType0)
|
|
883
|
+
continue;
|
|
884
|
+
|
|
885
|
+
auto cidDescriptor = cidFont.getKey("/FontDescriptor");
|
|
886
|
+
if (!cidDescriptor.isDictionary())
|
|
887
|
+
continue;
|
|
888
|
+
|
|
889
|
+
// determine which font file key to use:
|
|
890
|
+
// CIDFontType2 → /FontFile2 (TrueType)
|
|
891
|
+
// CIDFontType0 → /FontFile3 (CFF)
|
|
892
|
+
std::string fontFileKey;
|
|
893
|
+
if (isCIDFontType2 && cidDescriptor.hasKey("/FontFile2"))
|
|
894
|
+
fontFileKey = "/FontFile2";
|
|
895
|
+
else if (isCIDFontType0 && cidDescriptor.hasKey("/FontFile3"))
|
|
896
|
+
fontFileKey = "/FontFile3";
|
|
897
|
+
else
|
|
898
|
+
continue;
|
|
899
|
+
|
|
900
|
+
auto fontFile = cidDescriptor.getKey(fontFileKey);
|
|
901
|
+
if (!fontFile.isStream())
|
|
902
|
+
continue;
|
|
903
|
+
|
|
904
|
+
auto ffOg = fontFile.getObjGen();
|
|
905
|
+
fontFileHandles[ffOg] = fontFile;
|
|
906
|
+
fontFileCID[ffOg] = true;
|
|
907
|
+
|
|
908
|
+
try {
|
|
909
|
+
std::set<uint16_t> glyphIds;
|
|
910
|
+
|
|
911
|
+
auto cidToGid = cidFont.getKey("/CIDToGIDMap");
|
|
912
|
+
if (cidToGid.isName() && cidToGid.getName() == "/Identity") {
|
|
913
|
+
glyphIds = usedCodes;
|
|
914
|
+
} else if (cidToGid.isStream()) {
|
|
915
|
+
auto mapData = cidToGid.getStreamData(qpdf_dl_all);
|
|
916
|
+
const uint8_t *mapBuf = mapData->getBuffer();
|
|
917
|
+
size_t mapSize = mapData->getSize();
|
|
918
|
+
for (uint16_t cid : usedCodes) {
|
|
919
|
+
size_t offset = static_cast<size_t>(cid) * 2;
|
|
920
|
+
if (offset + 2 <= mapSize) {
|
|
921
|
+
uint16_t gid = static_cast<uint16_t>((mapBuf[offset] << 8) |
|
|
922
|
+
mapBuf[offset + 1]);
|
|
923
|
+
if (gid != 0)
|
|
924
|
+
glyphIds.insert(gid);
|
|
925
|
+
}
|
|
926
|
+
}
|
|
927
|
+
} else if (isCIDFontType0) {
|
|
928
|
+
glyphIds = usedCodes;
|
|
929
|
+
} else {
|
|
930
|
+
continue;
|
|
931
|
+
}
|
|
932
|
+
|
|
933
|
+
glyphIds.insert(0);
|
|
934
|
+
fontFileGlyphIds[ffOg].insert(glyphIds.begin(), glyphIds.end());
|
|
935
|
+
} catch (...) {
|
|
936
|
+
continue;
|
|
937
|
+
}
|
|
938
|
+
continue;
|
|
939
|
+
}
|
|
940
|
+
|
|
941
|
+
// simple TrueType fonts
|
|
942
|
+
if (subtype.getName() != "/TrueType")
|
|
943
|
+
continue;
|
|
944
|
+
|
|
945
|
+
// skip already-subset fonts (BaseFont like "ABCDEF+FontName") — their
|
|
946
|
+
// custom encodings and glyph tables can't be reliably re-subset
|
|
947
|
+
// skip already-subset fonts — mark their font file as skip so other
|
|
948
|
+
// fonts sharing it won't subset and corrupt the already-subset glyph data
|
|
949
|
+
if (isAlreadySubset(fontObj)) {
|
|
950
|
+
auto descriptor = fontObj.getKey("/FontDescriptor");
|
|
951
|
+
if (descriptor.isDictionary() && descriptor.hasKey("/FontFile2")) {
|
|
952
|
+
auto fontFile = descriptor.getKey("/FontFile2");
|
|
953
|
+
if (fontFile.isStream())
|
|
954
|
+
fontFileSkip.insert(fontFile.getObjGen());
|
|
955
|
+
}
|
|
956
|
+
continue;
|
|
957
|
+
}
|
|
958
|
+
|
|
959
|
+
auto descriptor = fontObj.getKey("/FontDescriptor");
|
|
960
|
+
if (!descriptor.isDictionary() || !descriptor.hasKey("/FontFile2"))
|
|
961
|
+
continue;
|
|
962
|
+
|
|
963
|
+
auto fontFile = descriptor.getKey("/FontFile2");
|
|
964
|
+
if (!fontFile.isStream())
|
|
965
|
+
continue;
|
|
966
|
+
|
|
967
|
+
auto ffOg = fontFile.getObjGen();
|
|
968
|
+
fontFileHandles[ffOg] = fontFile;
|
|
969
|
+
if (!fontFileCID.count(ffOg))
|
|
970
|
+
fontFileCID[ffOg] = false;
|
|
971
|
+
|
|
972
|
+
try {
|
|
973
|
+
auto fontData = fontFile.getStreamData(qpdf_dl_all);
|
|
974
|
+
const uint8_t *ttfData = fontData->getBuffer();
|
|
975
|
+
size_t ttfSize = fontData->getSize();
|
|
976
|
+
|
|
977
|
+
std::set<uint16_t> glyphIds;
|
|
978
|
+
|
|
979
|
+
// build /Differences map to identify remapped codes
|
|
980
|
+
auto diffMap = getDifferencesCodeMap(fontObj);
|
|
981
|
+
std::set<uint16_t> diffCodes; // codes that have /Differences entries
|
|
982
|
+
for (auto &[code, name] : diffMap) {
|
|
983
|
+
if (usedCodes.count(code))
|
|
984
|
+
diffCodes.insert(code);
|
|
985
|
+
}
|
|
986
|
+
std::set<uint16_t> baseCodes; // codes using base encoding (no remap)
|
|
987
|
+
for (uint16_t code : usedCodes) {
|
|
988
|
+
if (!diffCodes.count(code))
|
|
989
|
+
baseCodes.insert(code);
|
|
990
|
+
}
|
|
991
|
+
|
|
992
|
+
bool hasToUnicode = false;
|
|
993
|
+
|
|
994
|
+
// strategy 1: /ToUnicode CMap — most reliable Unicode mapping
|
|
995
|
+
auto toUnicode = fontObj.getKey("/ToUnicode");
|
|
996
|
+
if (toUnicode.isStream()) {
|
|
997
|
+
hasToUnicode = true;
|
|
998
|
+
auto tuCodes = parseToUnicode(toUnicode, usedCodes);
|
|
999
|
+
if (!tuCodes.empty()) {
|
|
1000
|
+
auto ids = mapCodesToGlyphIds(ttfData, ttfSize, tuCodes);
|
|
1001
|
+
glyphIds.insert(ids.begin(), ids.end());
|
|
1002
|
+
}
|
|
1003
|
+
}
|
|
1004
|
+
|
|
1005
|
+
// strategy 2: /Encoding /Differences — glyph name lookup via post table
|
|
1006
|
+
auto diffNames = getGlyphNamesFromEncoding(usedCodes, fontObj);
|
|
1007
|
+
if (!diffNames.empty()) {
|
|
1008
|
+
auto nameIds = mapGlyphNamesToGlyphIds(ttfData, ttfSize, diffNames);
|
|
1009
|
+
glyphIds.insert(nameIds.begin(), nameIds.end());
|
|
1010
|
+
}
|
|
1011
|
+
|
|
1012
|
+
// strategy 3: /Differences glyph names → Unicode → cmap lookup.
|
|
1013
|
+
// converts glyph names to Unicode codepoints (via AGL/uniXXXX)
|
|
1014
|
+
// and looks them up in the font's cmap. this catches characters like
|
|
1015
|
+
// ẞ (uni1E9E), € (Euro), • (bullet) that have non-standard byte codes
|
|
1016
|
+
// in /Differences but standard Unicode entries in the cmap.
|
|
1017
|
+
if (!diffCodes.empty()) {
|
|
1018
|
+
std::set<uint16_t> diffUnicodes;
|
|
1019
|
+
for (uint16_t code : diffCodes) {
|
|
1020
|
+
auto it = diffMap.find(code);
|
|
1021
|
+
if (it != diffMap.end()) {
|
|
1022
|
+
uint16_t unicode = glyphNameToUnicode(it->second);
|
|
1023
|
+
if (unicode > 0)
|
|
1024
|
+
diffUnicodes.insert(unicode);
|
|
1025
|
+
}
|
|
1026
|
+
}
|
|
1027
|
+
if (!diffUnicodes.empty()) {
|
|
1028
|
+
auto ids = mapCodesToGlyphIds(ttfData, ttfSize, diffUnicodes);
|
|
1029
|
+
glyphIds.insert(ids.begin(), ids.end());
|
|
1030
|
+
}
|
|
1031
|
+
}
|
|
1032
|
+
|
|
1033
|
+
// strategy 4: base encoding conversion (WinAnsi, MacRoman) → cmap
|
|
1034
|
+
// only for codes NOT remapped by /Differences — those are already
|
|
1035
|
+
// handled by strategies 2 and 3 with correct Unicode
|
|
1036
|
+
if (!baseCodes.empty()) {
|
|
1037
|
+
auto unicodeCodes = convertCodesToUnicode(baseCodes, fontObj);
|
|
1038
|
+
auto encIds = mapCodesToGlyphIds(ttfData, ttfSize, unicodeCodes);
|
|
1039
|
+
glyphIds.insert(encIds.begin(), encIds.end());
|
|
1040
|
+
|
|
1041
|
+
// strategy 5: raw character codes as fallback — only when byte codes
|
|
1042
|
+
// might directly index the cmap (no /ToUnicode remapping)
|
|
1043
|
+
if (!hasToUnicode && unicodeCodes != baseCodes) {
|
|
1044
|
+
auto rawGlyphs = mapCodesToGlyphIds(ttfData, ttfSize, baseCodes);
|
|
1045
|
+
glyphIds.insert(rawGlyphs.begin(), rawGlyphs.end());
|
|
1046
|
+
}
|
|
1047
|
+
|
|
1048
|
+
// strategy 6: "uniXXXX" glyph names via post table — catches fonts
|
|
1049
|
+
// where glyphs have no cmap entry but are accessible by name
|
|
1050
|
+
{
|
|
1051
|
+
std::vector<std::string> uniNames;
|
|
1052
|
+
for (uint16_t u : unicodeCodes) {
|
|
1053
|
+
if (u > 0x7F) {
|
|
1054
|
+
char buf[8];
|
|
1055
|
+
snprintf(buf, sizeof(buf), "uni%04X", u);
|
|
1056
|
+
uniNames.emplace_back(buf);
|
|
1057
|
+
}
|
|
1058
|
+
}
|
|
1059
|
+
if (!uniNames.empty()) {
|
|
1060
|
+
auto nameIds = mapGlyphNamesToGlyphIds(ttfData, ttfSize, uniNames);
|
|
1061
|
+
glyphIds.insert(nameIds.begin(), nameIds.end());
|
|
1062
|
+
}
|
|
1063
|
+
}
|
|
1064
|
+
}
|
|
1065
|
+
|
|
1066
|
+
// safety check: if we found fewer glyph IDs than used character codes,
|
|
1067
|
+
// some characters couldn't be mapped — mark font file as unsafe to
|
|
1068
|
+
// subset. if ANY font sharing this file fails, skip the entire file.
|
|
1069
|
+
glyphIds.insert(0);
|
|
1070
|
+
if (glyphIds.size() - 1 < usedCodes.size()) {
|
|
1071
|
+
fontFileSkip.insert(ffOg);
|
|
1072
|
+
continue;
|
|
1073
|
+
}
|
|
1074
|
+
|
|
1075
|
+
fontFileGlyphIds[ffOg].insert(glyphIds.begin(), glyphIds.end());
|
|
1076
|
+
} catch (...) {
|
|
1077
|
+
continue;
|
|
1078
|
+
}
|
|
1079
|
+
}
|
|
1080
|
+
|
|
1081
|
+
// pass 2: subset each font file with merged glyph IDs from all font objects
|
|
1082
|
+
for (auto &[ffOg, glyphIds] : fontFileGlyphIds) {
|
|
1083
|
+
if (fontFileSkip.count(ffOg))
|
|
1084
|
+
continue;
|
|
1085
|
+
|
|
1086
|
+
auto it = fontFileHandles.find(ffOg);
|
|
1087
|
+
if (it == fontFileHandles.end())
|
|
1088
|
+
continue;
|
|
1089
|
+
|
|
1090
|
+
auto &fontFile = it->second;
|
|
1091
|
+
bool isCID = fontFileCID[ffOg];
|
|
1092
|
+
|
|
1093
|
+
try {
|
|
1094
|
+
auto fontData = fontFile.getStreamData(qpdf_dl_all);
|
|
1095
|
+
const uint8_t *rawData = fontData->getBuffer();
|
|
1096
|
+
size_t rawSize = fontData->getSize();
|
|
1097
|
+
|
|
1098
|
+
std::vector<uint8_t> subsetResult;
|
|
1099
|
+
if (!subsetFont(rawData, rawSize, glyphIds, subsetResult, !isCID))
|
|
1100
|
+
continue;
|
|
1101
|
+
|
|
1102
|
+
if (subsetResult.size() >= rawSize)
|
|
1103
|
+
continue;
|
|
1104
|
+
|
|
1105
|
+
std::string fontStr(reinterpret_cast<char *>(subsetResult.data()),
|
|
1106
|
+
subsetResult.size());
|
|
1107
|
+
fontFile.replaceStreamData(fontStr, QPDFObjectHandle::newNull(),
|
|
1108
|
+
QPDFObjectHandle::newNull());
|
|
1109
|
+
} catch (...) {
|
|
1110
|
+
continue;
|
|
1111
|
+
}
|
|
1112
|
+
}
|
|
1113
|
+
|
|
1114
|
+
// optimize CID font /W arrays — rebuild with only used CID entries
|
|
1115
|
+
for (auto &[og, usedCodes] : fontUsedCodes) {
|
|
1116
|
+
auto fontObj = qpdf.getObjectByObjGen(og);
|
|
1117
|
+
if (!fontObj.isDictionary())
|
|
1118
|
+
continue;
|
|
1119
|
+
|
|
1120
|
+
auto subtype = fontObj.getKey("/Subtype");
|
|
1121
|
+
if (!subtype.isName() || subtype.getName() != "/Type0")
|
|
1122
|
+
continue;
|
|
1123
|
+
|
|
1124
|
+
// skip already-subset Type0 fonts
|
|
1125
|
+
if (isAlreadySubset(fontObj))
|
|
1126
|
+
continue;
|
|
1127
|
+
|
|
1128
|
+
auto descendants = fontObj.getKey("/DescendantFonts");
|
|
1129
|
+
if (!descendants.isArray() || descendants.getArrayNItems() < 1)
|
|
1130
|
+
continue;
|
|
1131
|
+
|
|
1132
|
+
auto cidFont = descendants.getArrayItem(0);
|
|
1133
|
+
if (!cidFont.isDictionary())
|
|
1134
|
+
continue;
|
|
1135
|
+
|
|
1136
|
+
// skip already-subset CID fonts — their width tables may have
|
|
1137
|
+
// custom CID mappings our usage collector doesn't fully capture
|
|
1138
|
+
if (isAlreadySubset(cidFont))
|
|
1139
|
+
continue;
|
|
1140
|
+
|
|
1141
|
+
auto w = cidFont.getKey("/W");
|
|
1142
|
+
if (!w.isArray() || w.getArrayNItems() == 0)
|
|
1143
|
+
continue;
|
|
1144
|
+
|
|
1145
|
+
// parse /W into CID → width value map
|
|
1146
|
+
std::map<int, QPDFObjectHandle> cidWidths;
|
|
1147
|
+
int n = w.getArrayNItems();
|
|
1148
|
+
int i = 0;
|
|
1149
|
+
while (i < n) {
|
|
1150
|
+
auto first = w.getArrayItem(i);
|
|
1151
|
+
if (!first.isInteger()) {
|
|
1152
|
+
++i;
|
|
1153
|
+
continue;
|
|
1154
|
+
}
|
|
1155
|
+
int cidStart = static_cast<int>(first.getIntValue());
|
|
1156
|
+
++i;
|
|
1157
|
+
if (i >= n)
|
|
1158
|
+
break;
|
|
1159
|
+
|
|
1160
|
+
auto second = w.getArrayItem(i);
|
|
1161
|
+
if (second.isArray()) {
|
|
1162
|
+
// format: cidStart [w1 w2 w3 ...]
|
|
1163
|
+
for (int j = 0; j < second.getArrayNItems(); ++j)
|
|
1164
|
+
cidWidths[cidStart + j] = second.getArrayItem(j);
|
|
1165
|
+
++i;
|
|
1166
|
+
} else if (second.isInteger()) {
|
|
1167
|
+
// format: cidStart cidEnd sameWidth
|
|
1168
|
+
int cidEnd = static_cast<int>(second.getIntValue());
|
|
1169
|
+
++i;
|
|
1170
|
+
if (i >= n)
|
|
1171
|
+
break;
|
|
1172
|
+
auto width = w.getArrayItem(i);
|
|
1173
|
+
for (int cid = cidStart; cid <= cidEnd; ++cid)
|
|
1174
|
+
cidWidths[cid] = width;
|
|
1175
|
+
++i;
|
|
1176
|
+
} else {
|
|
1177
|
+
++i;
|
|
1178
|
+
}
|
|
1179
|
+
}
|
|
1180
|
+
|
|
1181
|
+
// rebuild /W with only used CIDs grouped by consecutive runs
|
|
1182
|
+
std::vector<int> sortedUsed;
|
|
1183
|
+
for (uint16_t c : usedCodes) {
|
|
1184
|
+
if (cidWidths.count(static_cast<int>(c)))
|
|
1185
|
+
sortedUsed.push_back(static_cast<int>(c));
|
|
1186
|
+
}
|
|
1187
|
+
std::sort(sortedUsed.begin(), sortedUsed.end());
|
|
1188
|
+
|
|
1189
|
+
auto newW = QPDFObjectHandle::newArray();
|
|
1190
|
+
size_t idx = 0;
|
|
1191
|
+
while (idx < sortedUsed.size()) {
|
|
1192
|
+
int start = sortedUsed[idx];
|
|
1193
|
+
auto widths = QPDFObjectHandle::newArray();
|
|
1194
|
+
widths.appendItem(cidWidths[start]);
|
|
1195
|
+
++idx;
|
|
1196
|
+
while (idx < sortedUsed.size() &&
|
|
1197
|
+
sortedUsed[idx] == sortedUsed[idx - 1] + 1) {
|
|
1198
|
+
widths.appendItem(cidWidths[sortedUsed[idx]]);
|
|
1199
|
+
++idx;
|
|
1200
|
+
}
|
|
1201
|
+
newW.appendItem(QPDFObjectHandle::newInteger(start));
|
|
1202
|
+
newW.appendItem(widths);
|
|
1203
|
+
}
|
|
1204
|
+
|
|
1205
|
+
// only replace if the new /W has fewer entries
|
|
1206
|
+
if (newW.getArrayNItems() < w.getArrayNItems())
|
|
1207
|
+
cidFont.replaceKey("/W", newW);
|
|
1208
|
+
}
|
|
1209
|
+
}
|