qpdf-compress 0.2.0 → 0.4.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +124 -0
- package/README.md +87 -57
- package/binding.gyp +3 -1
- package/dist/index.d.ts +8 -6
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +3 -14
- package/dist/index.js.map +1 -1
- package/dist/types.d.ts +3 -9
- package/dist/types.d.ts.map +1 -1
- package/lib/index.ts +12 -21
- package/lib/types.ts +3 -9
- package/package.json +5 -2
- package/src/font_subset.cc +480 -0
- package/src/font_subset.h +17 -0
- package/src/images.cc +473 -126
- package/src/images.h +6 -6
- package/src/optimize.cc +1048 -0
- package/src/optimize.h +15 -0
- package/src/qpdf_addon.cc +34 -44
package/src/optimize.cc
ADDED
|
@@ -0,0 +1,1048 @@
|
|
|
1
|
+
#include "optimize.h"
|
|
2
|
+
#include "font_subset.h"
|
|
3
|
+
#include "images.h"
|
|
4
|
+
|
|
5
|
+
#include <cctype>
|
|
6
|
+
#include <cstdint>
|
|
7
|
+
#include <cstdlib>
|
|
8
|
+
#include <cstring>
|
|
9
|
+
#include <map>
|
|
10
|
+
#include <set>
|
|
11
|
+
#include <string>
|
|
12
|
+
#include <unordered_map>
|
|
13
|
+
#include <vector>
|
|
14
|
+
|
|
15
|
+
#include <qpdf/Buffer.hh>
|
|
16
|
+
#include <qpdf/QPDFObjectHandle.hh>
|
|
17
|
+
#include <qpdf/QPDFPageDocumentHelper.hh>
|
|
18
|
+
#include <qpdf/QPDFPageObjectHelper.hh>
|
|
19
|
+
|
|
20
|
+
// ---------------------------------------------------------------------------
|
|
21
|
+
// Metadata stripping
|
|
22
|
+
// ---------------------------------------------------------------------------
|
|
23
|
+
|
|
24
|
+
void stripMetadata(QPDF &qpdf) {
|
|
25
|
+
auto root = qpdf.getRoot();
|
|
26
|
+
|
|
27
|
+
// remove XMP metadata stream
|
|
28
|
+
if (root.hasKey("/Metadata"))
|
|
29
|
+
root.removeKey("/Metadata");
|
|
30
|
+
|
|
31
|
+
// remove document info dictionary
|
|
32
|
+
auto trailer = qpdf.getTrailer();
|
|
33
|
+
if (trailer.hasKey("/Info"))
|
|
34
|
+
trailer.removeKey("/Info");
|
|
35
|
+
|
|
36
|
+
// remove page-level metadata and PieceInfo
|
|
37
|
+
for (auto &page : QPDFPageDocumentHelper(qpdf).getAllPages()) {
|
|
38
|
+
auto pageObj = page.getObjectHandle();
|
|
39
|
+
if (pageObj.hasKey("/Metadata"))
|
|
40
|
+
pageObj.removeKey("/Metadata");
|
|
41
|
+
if (pageObj.hasKey("/PieceInfo"))
|
|
42
|
+
pageObj.removeKey("/PieceInfo");
|
|
43
|
+
}
|
|
44
|
+
|
|
45
|
+
// remove embedded thumbnails
|
|
46
|
+
for (auto &page : QPDFPageDocumentHelper(qpdf).getAllPages()) {
|
|
47
|
+
auto pageObj = page.getObjectHandle();
|
|
48
|
+
if (pageObj.hasKey("/Thumb"))
|
|
49
|
+
pageObj.removeKey("/Thumb");
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
// remove MarkInfo and page labels (optional metadata)
|
|
53
|
+
if (root.hasKey("/MarkInfo"))
|
|
54
|
+
root.removeKey("/MarkInfo");
|
|
55
|
+
}
|
|
56
|
+
|
|
57
|
+
// ---------------------------------------------------------------------------
|
|
58
|
+
// Remove unused font resources
|
|
59
|
+
// ---------------------------------------------------------------------------
|
|
60
|
+
|
|
61
|
+
void removeUnusedFonts(QPDF &qpdf) {
|
|
62
|
+
for (auto &page : QPDFPageDocumentHelper(qpdf).getAllPages()) {
|
|
63
|
+
auto pageObj = page.getObjectHandle();
|
|
64
|
+
auto resources = pageObj.getKey("/Resources");
|
|
65
|
+
if (!resources.isDictionary())
|
|
66
|
+
continue;
|
|
67
|
+
auto fonts = resources.getKey("/Font");
|
|
68
|
+
if (!fonts.isDictionary())
|
|
69
|
+
continue;
|
|
70
|
+
|
|
71
|
+
// collect all font names referenced in this page's content stream(s)
|
|
72
|
+
std::set<std::string> usedFonts;
|
|
73
|
+
|
|
74
|
+
try {
|
|
75
|
+
// get unparsed content stream data
|
|
76
|
+
auto contents = pageObj.getKey("/Contents");
|
|
77
|
+
std::string contentStr;
|
|
78
|
+
|
|
79
|
+
if (contents.isStream()) {
|
|
80
|
+
auto buf = contents.getStreamData(qpdf_dl_generalized);
|
|
81
|
+
contentStr.assign(reinterpret_cast<const char *>(buf->getBuffer()),
|
|
82
|
+
buf->getSize());
|
|
83
|
+
} else if (contents.isArray()) {
|
|
84
|
+
for (int i = 0; i < contents.getArrayNItems(); ++i) {
|
|
85
|
+
auto stream = contents.getArrayItem(i);
|
|
86
|
+
if (stream.isStream()) {
|
|
87
|
+
auto buf = stream.getStreamData(qpdf_dl_generalized);
|
|
88
|
+
contentStr.append(reinterpret_cast<const char *>(buf->getBuffer()),
|
|
89
|
+
buf->getSize());
|
|
90
|
+
contentStr += '\n';
|
|
91
|
+
}
|
|
92
|
+
}
|
|
93
|
+
}
|
|
94
|
+
|
|
95
|
+
// scan for /FontName references — Tf operator uses font name
|
|
96
|
+
// pattern: /FontName <size> Tf
|
|
97
|
+
for (auto &fontKey : fonts.getKeys()) {
|
|
98
|
+
// fontKey includes the leading '/', e.g. "/F1"
|
|
99
|
+
if (contentStr.find(fontKey) != std::string::npos)
|
|
100
|
+
usedFonts.insert(fontKey);
|
|
101
|
+
}
|
|
102
|
+
} catch (...) {
|
|
103
|
+
continue; // skip this page if content can't be read
|
|
104
|
+
}
|
|
105
|
+
|
|
106
|
+
// remove fonts that are not referenced in the content stream
|
|
107
|
+
auto allFontKeys = fonts.getKeys();
|
|
108
|
+
for (auto &fontKey : allFontKeys) {
|
|
109
|
+
if (usedFonts.find(fontKey) == usedFonts.end())
|
|
110
|
+
fonts.removeKey(fontKey);
|
|
111
|
+
}
|
|
112
|
+
}
|
|
113
|
+
}
|
|
114
|
+
|
|
115
|
+
// ---------------------------------------------------------------------------
|
|
116
|
+
// Content stream coalescing — merge multiple content streams per page into one
|
|
117
|
+
// ---------------------------------------------------------------------------
|
|
118
|
+
|
|
119
|
+
void coalesceContentStreams(QPDF &qpdf) {
|
|
120
|
+
for (auto &page : QPDFPageDocumentHelper(qpdf).getAllPages()) {
|
|
121
|
+
auto pageObj = page.getObjectHandle();
|
|
122
|
+
auto contents = pageObj.getKey("/Contents");
|
|
123
|
+
|
|
124
|
+
// only coalesce if there are multiple content streams (array)
|
|
125
|
+
if (contents.isArray() && contents.getArrayNItems() > 1) {
|
|
126
|
+
page.coalesceContentStreams();
|
|
127
|
+
}
|
|
128
|
+
}
|
|
129
|
+
}
|
|
130
|
+
|
|
131
|
+
// ---------------------------------------------------------------------------
|
|
132
|
+
// Deduplicate identical non-image streams (fonts, ICC profiles, etc.)
|
|
133
|
+
// ---------------------------------------------------------------------------
|
|
134
|
+
|
|
135
|
+
void deduplicateStreams(QPDF &qpdf) {
|
|
136
|
+
// collect all stream objects and their raw data hashes
|
|
137
|
+
struct StreamEntry {
|
|
138
|
+
QPDFObjGen og;
|
|
139
|
+
size_t dataSize;
|
|
140
|
+
QPDFObjectHandle handle;
|
|
141
|
+
};
|
|
142
|
+
|
|
143
|
+
std::unordered_map<uint64_t, std::vector<StreamEntry>> hashGroups;
|
|
144
|
+
std::set<QPDFObjGen> imageObjGens;
|
|
145
|
+
|
|
146
|
+
// collect image object IDs to skip them (already handled by
|
|
147
|
+
// deduplicateImages)
|
|
148
|
+
forEachImage(qpdf, [&](const std::string &, QPDFObjectHandle xobj,
|
|
149
|
+
QPDFObjectHandle, QPDFPageObjectHelper &) {
|
|
150
|
+
imageObjGens.insert(xobj.getObjGen());
|
|
151
|
+
});
|
|
152
|
+
|
|
153
|
+
for (auto &obj : qpdf.getAllObjects()) {
|
|
154
|
+
if (!obj.isStream())
|
|
155
|
+
continue;
|
|
156
|
+
|
|
157
|
+
auto og = obj.getObjGen();
|
|
158
|
+
|
|
159
|
+
// skip images
|
|
160
|
+
if (imageObjGens.count(og))
|
|
161
|
+
continue;
|
|
162
|
+
|
|
163
|
+
try {
|
|
164
|
+
auto rawData = obj.getRawStreamData();
|
|
165
|
+
size_t size = rawData->getSize();
|
|
166
|
+
if (size == 0)
|
|
167
|
+
continue;
|
|
168
|
+
|
|
169
|
+
// FNV-1a hash
|
|
170
|
+
uint64_t hash = 14695981039346656037ULL;
|
|
171
|
+
auto *p = rawData->getBuffer();
|
|
172
|
+
for (size_t i = 0; i < size; ++i) {
|
|
173
|
+
hash ^= static_cast<uint64_t>(p[i]);
|
|
174
|
+
hash *= 1099511628211ULL;
|
|
175
|
+
}
|
|
176
|
+
|
|
177
|
+
hashGroups[hash].push_back({og, size, obj});
|
|
178
|
+
} catch (...) {
|
|
179
|
+
}
|
|
180
|
+
}
|
|
181
|
+
|
|
182
|
+
// find duplicates via full byte comparison
|
|
183
|
+
std::map<QPDFObjGen, QPDFObjectHandle> replacements;
|
|
184
|
+
|
|
185
|
+
for (auto &[hash, group] : hashGroups) {
|
|
186
|
+
if (group.size() < 2)
|
|
187
|
+
continue;
|
|
188
|
+
|
|
189
|
+
for (size_t i = 0; i < group.size(); ++i) {
|
|
190
|
+
if (replacements.count(group[i].og))
|
|
191
|
+
continue;
|
|
192
|
+
|
|
193
|
+
auto rawI = group[i].handle.getRawStreamData();
|
|
194
|
+
for (size_t j = i + 1; j < group.size(); ++j) {
|
|
195
|
+
if (replacements.count(group[j].og))
|
|
196
|
+
continue;
|
|
197
|
+
|
|
198
|
+
auto rawJ = group[j].handle.getRawStreamData();
|
|
199
|
+
if (rawI->getSize() != rawJ->getSize())
|
|
200
|
+
continue;
|
|
201
|
+
|
|
202
|
+
if (memcmp(rawI->getBuffer(), rawJ->getBuffer(), rawI->getSize()) ==
|
|
203
|
+
0) {
|
|
204
|
+
// verify stream dictionaries are compatible
|
|
205
|
+
auto dictI = group[i].handle.getDict();
|
|
206
|
+
auto dictJ = group[j].handle.getDict();
|
|
207
|
+
|
|
208
|
+
auto filterI = dictI.getKey("/Filter");
|
|
209
|
+
auto filterJ = dictJ.getKey("/Filter");
|
|
210
|
+
bool filtersMatch = (filterI.isName() && filterJ.isName() &&
|
|
211
|
+
filterI.getName() == filterJ.getName()) ||
|
|
212
|
+
(!filterI.isName() && !filterJ.isName());
|
|
213
|
+
|
|
214
|
+
// also verify DecodeParms match — different predictors on
|
|
215
|
+
// identical raw bytes would produce different decoded content
|
|
216
|
+
auto dpI = dictI.getKey("/DecodeParms");
|
|
217
|
+
auto dpJ = dictJ.getKey("/DecodeParms");
|
|
218
|
+
bool paramsMatch = (dpI.isNull() && dpJ.isNull()) ||
|
|
219
|
+
(!dpI.isNull() && !dpJ.isNull() &&
|
|
220
|
+
dpI.unparse() == dpJ.unparse());
|
|
221
|
+
|
|
222
|
+
if (filtersMatch && paramsMatch)
|
|
223
|
+
replacements[group[j].og] = group[i].handle;
|
|
224
|
+
}
|
|
225
|
+
}
|
|
226
|
+
}
|
|
227
|
+
}
|
|
228
|
+
|
|
229
|
+
if (replacements.empty())
|
|
230
|
+
return;
|
|
231
|
+
|
|
232
|
+
// rewrite references: scan all objects for indirect references to duplicates
|
|
233
|
+
for (auto &obj : qpdf.getAllObjects()) {
|
|
234
|
+
if (!obj.isDictionary() && !obj.isStream())
|
|
235
|
+
continue;
|
|
236
|
+
|
|
237
|
+
auto dict = obj.isStream() ? obj.getDict() : obj;
|
|
238
|
+
for (auto &key : dict.getKeys()) {
|
|
239
|
+
auto val = dict.getKey(key);
|
|
240
|
+
if (val.isIndirect()) {
|
|
241
|
+
auto it = replacements.find(val.getObjGen());
|
|
242
|
+
if (it != replacements.end())
|
|
243
|
+
dict.replaceKey(key, it->second);
|
|
244
|
+
}
|
|
245
|
+
}
|
|
246
|
+
}
|
|
247
|
+
}
|
|
248
|
+
|
|
249
|
+
// ---------------------------------------------------------------------------
|
|
250
|
+
// Font subsetting — remove unused glyphs from TrueType/CIDFont fonts
|
|
251
|
+
// ---------------------------------------------------------------------------
|
|
252
|
+
|
|
253
|
+
// collects all Unicode code points used in a page's content stream by
|
|
254
|
+
// parsing text-showing operators (Tj, TJ, ', ")
|
|
255
|
+
static std::set<uint16_t> collectUsedCodes(const std::string &contentStr,
|
|
256
|
+
const std::string &fontKey,
|
|
257
|
+
QPDFObjectHandle fontObj) {
|
|
258
|
+
std::set<uint16_t> usedCodes;
|
|
259
|
+
|
|
260
|
+
// simple scan: find all string operands between font selection (Tf) and
|
|
261
|
+
// text-showing operators. For simplicity, collect all hex/literal string
|
|
262
|
+
// bytes used when this font is active.
|
|
263
|
+
|
|
264
|
+
// check if this font uses 2-byte CID encoding
|
|
265
|
+
auto subtypeKey = fontObj.getKey("/Subtype");
|
|
266
|
+
bool isCIDFont = subtypeKey.isName() && subtypeKey.getName() == "/Type0";
|
|
267
|
+
|
|
268
|
+
// find all ranges where this font is active and collect string bytes
|
|
269
|
+
bool fontActive = false;
|
|
270
|
+
size_t pos = 0;
|
|
271
|
+
|
|
272
|
+
while (pos < contentStr.size()) {
|
|
273
|
+
// skip whitespace
|
|
274
|
+
while (pos < contentStr.size() && std::isspace(contentStr[pos]))
|
|
275
|
+
++pos;
|
|
276
|
+
|
|
277
|
+
if (pos >= contentStr.size())
|
|
278
|
+
break;
|
|
279
|
+
|
|
280
|
+
// check for font selection: /FontName ... Tf
|
|
281
|
+
if (contentStr[pos] == '/') {
|
|
282
|
+
// read name
|
|
283
|
+
size_t nameStart = pos;
|
|
284
|
+
++pos;
|
|
285
|
+
while (pos < contentStr.size() && !std::isspace(contentStr[pos]) &&
|
|
286
|
+
contentStr[pos] != '/' && contentStr[pos] != '<' &&
|
|
287
|
+
contentStr[pos] != '(' && contentStr[pos] != '[')
|
|
288
|
+
++pos;
|
|
289
|
+
std::string name = contentStr.substr(nameStart, pos - nameStart);
|
|
290
|
+
|
|
291
|
+
// look ahead for Tf
|
|
292
|
+
size_t lookAhead = pos;
|
|
293
|
+
while (lookAhead < contentStr.size() &&
|
|
294
|
+
std::isspace(contentStr[lookAhead]))
|
|
295
|
+
++lookAhead;
|
|
296
|
+
// skip number
|
|
297
|
+
while (
|
|
298
|
+
lookAhead < contentStr.size() &&
|
|
299
|
+
(std::isdigit(contentStr[lookAhead]) || contentStr[lookAhead] == '.'))
|
|
300
|
+
++lookAhead;
|
|
301
|
+
while (lookAhead < contentStr.size() &&
|
|
302
|
+
std::isspace(contentStr[lookAhead]))
|
|
303
|
+
++lookAhead;
|
|
304
|
+
if (lookAhead + 1 < contentStr.size() && contentStr[lookAhead] == 'T' &&
|
|
305
|
+
contentStr[lookAhead + 1] == 'f') {
|
|
306
|
+
fontActive = (name == fontKey);
|
|
307
|
+
pos = lookAhead + 2;
|
|
308
|
+
continue;
|
|
309
|
+
}
|
|
310
|
+
}
|
|
311
|
+
|
|
312
|
+
// collect string data when our font is active
|
|
313
|
+
if (fontActive && contentStr[pos] == '(') {
|
|
314
|
+
// literal string
|
|
315
|
+
++pos;
|
|
316
|
+
int depth = 1;
|
|
317
|
+
while (pos < contentStr.size() && depth > 0) {
|
|
318
|
+
if (contentStr[pos] == '\\') {
|
|
319
|
+
++pos; // skip escaped char
|
|
320
|
+
if (pos < contentStr.size())
|
|
321
|
+
++pos;
|
|
322
|
+
continue;
|
|
323
|
+
}
|
|
324
|
+
if (contentStr[pos] == '(')
|
|
325
|
+
++depth;
|
|
326
|
+
else if (contentStr[pos] == ')')
|
|
327
|
+
--depth;
|
|
328
|
+
if (depth > 0) {
|
|
329
|
+
if (isCIDFont && pos + 1 < contentStr.size()) {
|
|
330
|
+
uint16_t code = (static_cast<uint8_t>(contentStr[pos]) << 8) |
|
|
331
|
+
static_cast<uint8_t>(contentStr[pos + 1]);
|
|
332
|
+
usedCodes.insert(code);
|
|
333
|
+
++pos;
|
|
334
|
+
} else {
|
|
335
|
+
usedCodes.insert(static_cast<uint8_t>(contentStr[pos]));
|
|
336
|
+
}
|
|
337
|
+
++pos;
|
|
338
|
+
}
|
|
339
|
+
}
|
|
340
|
+
if (depth == 0)
|
|
341
|
+
++pos; // skip closing ')'
|
|
342
|
+
continue;
|
|
343
|
+
}
|
|
344
|
+
|
|
345
|
+
if (fontActive && contentStr[pos] == '<') {
|
|
346
|
+
// hex string
|
|
347
|
+
++pos;
|
|
348
|
+
std::vector<uint8_t> hexBytes;
|
|
349
|
+
while (pos < contentStr.size() && contentStr[pos] != '>') {
|
|
350
|
+
if (std::isxdigit(contentStr[pos])) {
|
|
351
|
+
char hex[3] = {contentStr[pos], '0', '\0'};
|
|
352
|
+
if (pos + 1 < contentStr.size() &&
|
|
353
|
+
std::isxdigit(contentStr[pos + 1])) {
|
|
354
|
+
hex[1] = contentStr[pos + 1];
|
|
355
|
+
++pos;
|
|
356
|
+
}
|
|
357
|
+
hexBytes.push_back(
|
|
358
|
+
static_cast<uint8_t>(std::strtol(hex, nullptr, 16)));
|
|
359
|
+
}
|
|
360
|
+
++pos;
|
|
361
|
+
}
|
|
362
|
+
if (pos < contentStr.size())
|
|
363
|
+
++pos; // skip '>'
|
|
364
|
+
|
|
365
|
+
if (isCIDFont) {
|
|
366
|
+
for (size_t b = 0; b + 1 < hexBytes.size(); b += 2) {
|
|
367
|
+
uint16_t code = (hexBytes[b] << 8) | hexBytes[b + 1];
|
|
368
|
+
usedCodes.insert(code);
|
|
369
|
+
}
|
|
370
|
+
} else {
|
|
371
|
+
for (auto b : hexBytes)
|
|
372
|
+
usedCodes.insert(b);
|
|
373
|
+
}
|
|
374
|
+
continue;
|
|
375
|
+
}
|
|
376
|
+
|
|
377
|
+
// skip other tokens
|
|
378
|
+
while (pos < contentStr.size() && !std::isspace(contentStr[pos]))
|
|
379
|
+
++pos;
|
|
380
|
+
}
|
|
381
|
+
|
|
382
|
+
return usedCodes;
|
|
383
|
+
}
|
|
384
|
+
|
|
385
|
+
void subsetFonts(QPDF &qpdf) {
|
|
386
|
+
// collect used character codes per font object
|
|
387
|
+
std::map<QPDFObjGen, std::set<uint16_t>> fontUsedCodes;
|
|
388
|
+
|
|
389
|
+
for (auto &page : QPDFPageDocumentHelper(qpdf).getAllPages()) {
|
|
390
|
+
auto pageObj = page.getObjectHandle();
|
|
391
|
+
auto resources = pageObj.getKey("/Resources");
|
|
392
|
+
if (!resources.isDictionary())
|
|
393
|
+
continue;
|
|
394
|
+
auto fonts = resources.getKey("/Font");
|
|
395
|
+
if (!fonts.isDictionary())
|
|
396
|
+
continue;
|
|
397
|
+
|
|
398
|
+
// decode content stream once per page (shared across all fonts)
|
|
399
|
+
auto contents = pageObj.getKey("/Contents");
|
|
400
|
+
std::string contentStr;
|
|
401
|
+
try {
|
|
402
|
+
if (contents.isStream()) {
|
|
403
|
+
auto buf = contents.getStreamData(qpdf_dl_generalized);
|
|
404
|
+
contentStr.assign(reinterpret_cast<const char *>(buf->getBuffer()),
|
|
405
|
+
buf->getSize());
|
|
406
|
+
} else if (contents.isArray()) {
|
|
407
|
+
for (int i = 0; i < contents.getArrayNItems(); ++i) {
|
|
408
|
+
auto stream = contents.getArrayItem(i);
|
|
409
|
+
if (stream.isStream()) {
|
|
410
|
+
auto buf = stream.getStreamData(qpdf_dl_generalized);
|
|
411
|
+
contentStr.append(reinterpret_cast<const char *>(buf->getBuffer()),
|
|
412
|
+
buf->getSize());
|
|
413
|
+
contentStr += '\n';
|
|
414
|
+
}
|
|
415
|
+
}
|
|
416
|
+
}
|
|
417
|
+
} catch (...) {
|
|
418
|
+
continue;
|
|
419
|
+
}
|
|
420
|
+
|
|
421
|
+
for (auto &key : fonts.getKeys()) {
|
|
422
|
+
auto fontObj = fonts.getKey(key);
|
|
423
|
+
if (!fontObj.isDictionary())
|
|
424
|
+
continue;
|
|
425
|
+
|
|
426
|
+
auto og = fontObj.getObjGen();
|
|
427
|
+
auto codes = collectUsedCodes(contentStr, key, fontObj);
|
|
428
|
+
fontUsedCodes[og].insert(codes.begin(), codes.end());
|
|
429
|
+
}
|
|
430
|
+
}
|
|
431
|
+
|
|
432
|
+
// for each font with a /Widths array, zero out widths for unused glyphs
|
|
433
|
+
// and truncate trailing zeros
|
|
434
|
+
for (auto &[og, usedCodes] : fontUsedCodes) {
|
|
435
|
+
auto fontObj = qpdf.getObjectByObjGen(og);
|
|
436
|
+
if (!fontObj.isDictionary())
|
|
437
|
+
continue;
|
|
438
|
+
|
|
439
|
+
auto subtype = fontObj.getKey("/Subtype");
|
|
440
|
+
if (!subtype.isName())
|
|
441
|
+
continue;
|
|
442
|
+
|
|
443
|
+
// only handle simple fonts with /Widths arrays (TrueType, Type1)
|
|
444
|
+
if (subtype.getName() != "/TrueType" && subtype.getName() != "/Type1")
|
|
445
|
+
continue;
|
|
446
|
+
|
|
447
|
+
auto widths = fontObj.getKey("/Widths");
|
|
448
|
+
auto firstCharObj = fontObj.getKey("/FirstChar");
|
|
449
|
+
if (!widths.isArray() || !firstCharObj.isInteger())
|
|
450
|
+
continue;
|
|
451
|
+
|
|
452
|
+
int firstChar = static_cast<int>(firstCharObj.getIntValue());
|
|
453
|
+
int widthCount = widths.getArrayNItems();
|
|
454
|
+
|
|
455
|
+
// zero out widths for unused character codes
|
|
456
|
+
bool modified = false;
|
|
457
|
+
for (int i = 0; i < widthCount; ++i) {
|
|
458
|
+
int charCode = firstChar + i;
|
|
459
|
+
if (usedCodes.find(static_cast<uint16_t>(charCode)) == usedCodes.end()) {
|
|
460
|
+
auto w = widths.getArrayItem(i);
|
|
461
|
+
if (w.isInteger() && w.getIntValue() != 0) {
|
|
462
|
+
widths.setArrayItem(i, QPDFObjectHandle::newInteger(0));
|
|
463
|
+
modified = true;
|
|
464
|
+
} else if (w.isReal() && std::stod(w.getRealValue()) != 0.0) {
|
|
465
|
+
widths.setArrayItem(i, QPDFObjectHandle::newInteger(0));
|
|
466
|
+
modified = true;
|
|
467
|
+
}
|
|
468
|
+
}
|
|
469
|
+
}
|
|
470
|
+
|
|
471
|
+
if (!modified)
|
|
472
|
+
continue;
|
|
473
|
+
|
|
474
|
+
// trim trailing zero-width entries and adjust /LastChar
|
|
475
|
+
int lastUsed = widthCount - 1;
|
|
476
|
+
while (lastUsed >= 0) {
|
|
477
|
+
auto w = widths.getArrayItem(lastUsed);
|
|
478
|
+
if (w.isInteger() && w.getIntValue() == 0)
|
|
479
|
+
--lastUsed;
|
|
480
|
+
else
|
|
481
|
+
break;
|
|
482
|
+
}
|
|
483
|
+
|
|
484
|
+
if (lastUsed < widthCount - 1) {
|
|
485
|
+
// rebuild the widths array with only the needed entries
|
|
486
|
+
auto newWidths = QPDFObjectHandle::newArray();
|
|
487
|
+
for (int i = 0; i <= lastUsed; ++i)
|
|
488
|
+
newWidths.appendItem(widths.getArrayItem(i));
|
|
489
|
+
|
|
490
|
+
fontObj.replaceKey("/Widths", newWidths);
|
|
491
|
+
fontObj.replaceKey("/LastChar",
|
|
492
|
+
QPDFObjectHandle::newInteger(firstChar + lastUsed));
|
|
493
|
+
}
|
|
494
|
+
}
|
|
495
|
+
|
|
496
|
+
// true font subsetting: strip unused glyph outlines from TrueType fonts
|
|
497
|
+
std::set<QPDFObjGen> processedFonts;
|
|
498
|
+
for (auto &[og, usedCodes] : fontUsedCodes) {
|
|
499
|
+
if (processedFonts.count(og))
|
|
500
|
+
continue;
|
|
501
|
+
processedFonts.insert(og);
|
|
502
|
+
|
|
503
|
+
auto fontObj = qpdf.getObjectByObjGen(og);
|
|
504
|
+
if (!fontObj.isDictionary())
|
|
505
|
+
continue;
|
|
506
|
+
|
|
507
|
+
auto subtype = fontObj.getKey("/Subtype");
|
|
508
|
+
if (!subtype.isName() || subtype.getName() != "/TrueType")
|
|
509
|
+
continue;
|
|
510
|
+
|
|
511
|
+
auto descriptor = fontObj.getKey("/FontDescriptor");
|
|
512
|
+
if (!descriptor.isDictionary())
|
|
513
|
+
continue;
|
|
514
|
+
|
|
515
|
+
// TrueType fonts use /FontFile2
|
|
516
|
+
if (!descriptor.hasKey("/FontFile2"))
|
|
517
|
+
continue;
|
|
518
|
+
|
|
519
|
+
auto fontFile = descriptor.getKey("/FontFile2");
|
|
520
|
+
if (!fontFile.isStream())
|
|
521
|
+
continue;
|
|
522
|
+
|
|
523
|
+
try {
|
|
524
|
+
auto fontData = fontFile.getStreamData(qpdf_dl_all);
|
|
525
|
+
const uint8_t *ttfData = fontData->getBuffer();
|
|
526
|
+
size_t ttfSize = fontData->getSize();
|
|
527
|
+
|
|
528
|
+
// map character codes → glyph IDs via cmap
|
|
529
|
+
auto glyphIds = mapCodesToGlyphIds(ttfData, ttfSize, usedCodes);
|
|
530
|
+
|
|
531
|
+
// skip if we'd keep all or nearly all glyphs
|
|
532
|
+
// (subsetting overhead wouldn't be worth it)
|
|
533
|
+
if (glyphIds.size() >= 200)
|
|
534
|
+
continue;
|
|
535
|
+
|
|
536
|
+
std::vector<uint8_t> subsetFont;
|
|
537
|
+
if (!subsetTrueTypeFont(ttfData, ttfSize, glyphIds, subsetFont))
|
|
538
|
+
continue;
|
|
539
|
+
|
|
540
|
+
// only replace if the subset is smaller than the original uncompressed
|
|
541
|
+
// font (both will be Flate-compressed by QPDFWriter, so comparing
|
|
542
|
+
// uncompressed sizes is the fair comparison)
|
|
543
|
+
if (subsetFont.size() >= ttfSize)
|
|
544
|
+
continue;
|
|
545
|
+
|
|
546
|
+
std::string fontStr(reinterpret_cast<char *>(subsetFont.data()),
|
|
547
|
+
subsetFont.size());
|
|
548
|
+
fontFile.replaceStreamData(fontStr, QPDFObjectHandle::newNull(),
|
|
549
|
+
QPDFObjectHandle::newNull());
|
|
550
|
+
} catch (...) {
|
|
551
|
+
continue;
|
|
552
|
+
}
|
|
553
|
+
}
|
|
554
|
+
}
|
|
555
|
+
|
|
556
|
+
// ---------------------------------------------------------------------------
|
|
557
|
+
// ICC profile stripping — replace ICCBased color spaces with Device
|
|
558
|
+
// equivalents
|
|
559
|
+
// ---------------------------------------------------------------------------
|
|
560
|
+
|
|
561
|
+
void stripIccProfiles(QPDF &qpdf) {
|
|
562
|
+
std::set<QPDFObjGen> processed;
|
|
563
|
+
|
|
564
|
+
// strip ICC profiles from images
|
|
565
|
+
forEachImage(qpdf, [&](const std::string &, QPDFObjectHandle xobj,
|
|
566
|
+
QPDFObjectHandle, QPDFPageObjectHelper &) {
|
|
567
|
+
auto og = xobj.getObjGen();
|
|
568
|
+
if (processed.count(og))
|
|
569
|
+
return;
|
|
570
|
+
processed.insert(og);
|
|
571
|
+
|
|
572
|
+
auto dict = xobj.getDict();
|
|
573
|
+
auto cs = dict.getKey("/ColorSpace");
|
|
574
|
+
|
|
575
|
+
if (!cs.isArray() || cs.getArrayNItems() < 2)
|
|
576
|
+
return;
|
|
577
|
+
|
|
578
|
+
auto csName = cs.getArrayItem(0);
|
|
579
|
+
if (!csName.isName() || csName.getName() != "/ICCBased")
|
|
580
|
+
return;
|
|
581
|
+
|
|
582
|
+
auto profile = cs.getArrayItem(1);
|
|
583
|
+
if (!profile.isStream())
|
|
584
|
+
return;
|
|
585
|
+
|
|
586
|
+
auto n = profile.getDict().getKey("/N");
|
|
587
|
+
if (!n.isInteger())
|
|
588
|
+
return;
|
|
589
|
+
|
|
590
|
+
int components = static_cast<int>(n.getIntValue());
|
|
591
|
+
if (components == 3)
|
|
592
|
+
dict.replaceKey("/ColorSpace", QPDFObjectHandle::newName("/DeviceRGB"));
|
|
593
|
+
else if (components == 1)
|
|
594
|
+
dict.replaceKey("/ColorSpace", QPDFObjectHandle::newName("/DeviceGray"));
|
|
595
|
+
else if (components == 4)
|
|
596
|
+
dict.replaceKey("/ColorSpace", QPDFObjectHandle::newName("/DeviceCMYK"));
|
|
597
|
+
});
|
|
598
|
+
|
|
599
|
+
// strip ICC profiles from page-level color space resources
|
|
600
|
+
for (auto &page : QPDFPageDocumentHelper(qpdf).getAllPages()) {
|
|
601
|
+
auto resources = page.getObjectHandle().getKey("/Resources");
|
|
602
|
+
if (!resources.isDictionary())
|
|
603
|
+
continue;
|
|
604
|
+
|
|
605
|
+
auto colorSpaces = resources.getKey("/ColorSpace");
|
|
606
|
+
if (!colorSpaces.isDictionary())
|
|
607
|
+
continue;
|
|
608
|
+
|
|
609
|
+
for (auto &key : colorSpaces.getKeys()) {
|
|
610
|
+
auto cs = colorSpaces.getKey(key);
|
|
611
|
+
if (!cs.isArray() || cs.getArrayNItems() < 2)
|
|
612
|
+
continue;
|
|
613
|
+
|
|
614
|
+
auto csName = cs.getArrayItem(0);
|
|
615
|
+
if (!csName.isName() || csName.getName() != "/ICCBased")
|
|
616
|
+
continue;
|
|
617
|
+
|
|
618
|
+
auto profile = cs.getArrayItem(1);
|
|
619
|
+
if (!profile.isStream())
|
|
620
|
+
continue;
|
|
621
|
+
|
|
622
|
+
auto n = profile.getDict().getKey("/N");
|
|
623
|
+
if (!n.isInteger())
|
|
624
|
+
continue;
|
|
625
|
+
|
|
626
|
+
int components = static_cast<int>(n.getIntValue());
|
|
627
|
+
if (components == 3)
|
|
628
|
+
colorSpaces.replaceKey(key, QPDFObjectHandle::newName("/DeviceRGB"));
|
|
629
|
+
else if (components == 1)
|
|
630
|
+
colorSpaces.replaceKey(key, QPDFObjectHandle::newName("/DeviceGray"));
|
|
631
|
+
else if (components == 4)
|
|
632
|
+
colorSpaces.replaceKey(key, QPDFObjectHandle::newName("/DeviceCMYK"));
|
|
633
|
+
}
|
|
634
|
+
}
|
|
635
|
+
}
|
|
636
|
+
|
|
637
|
+
// ---------------------------------------------------------------------------
|
|
638
|
+
// Embedded file stripping — remove /EmbeddedFiles from the name tree
|
|
639
|
+
// ---------------------------------------------------------------------------
|
|
640
|
+
|
|
641
|
+
void stripEmbeddedFiles(QPDF &qpdf) {
|
|
642
|
+
auto root = qpdf.getRoot();
|
|
643
|
+
if (!root.hasKey("/Names"))
|
|
644
|
+
return;
|
|
645
|
+
|
|
646
|
+
auto names = root.getKey("/Names");
|
|
647
|
+
if (!names.isDictionary())
|
|
648
|
+
return;
|
|
649
|
+
|
|
650
|
+
if (names.hasKey("/EmbeddedFiles"))
|
|
651
|
+
names.removeKey("/EmbeddedFiles");
|
|
652
|
+
|
|
653
|
+
// if /Names is now empty, remove it too
|
|
654
|
+
if (names.getKeys().empty())
|
|
655
|
+
root.removeKey("/Names");
|
|
656
|
+
}
|
|
657
|
+
|
|
658
|
+
// ---------------------------------------------------------------------------
|
|
659
|
+
// JavaScript and action removal — strip JS, open actions, and additional
|
|
660
|
+
// actions from the catalog and all pages
|
|
661
|
+
// ---------------------------------------------------------------------------
|
|
662
|
+
|
|
663
|
+
void stripJavaScript(QPDF &qpdf) {
|
|
664
|
+
auto root = qpdf.getRoot();
|
|
665
|
+
|
|
666
|
+
// remove document-level open action
|
|
667
|
+
if (root.hasKey("/OpenAction"))
|
|
668
|
+
root.removeKey("/OpenAction");
|
|
669
|
+
|
|
670
|
+
// remove document-level additional actions
|
|
671
|
+
if (root.hasKey("/AA"))
|
|
672
|
+
root.removeKey("/AA");
|
|
673
|
+
|
|
674
|
+
// remove /JavaScript name tree
|
|
675
|
+
if (root.hasKey("/Names")) {
|
|
676
|
+
auto names = root.getKey("/Names");
|
|
677
|
+
if (names.isDictionary() && names.hasKey("/JavaScript"))
|
|
678
|
+
names.removeKey("/JavaScript");
|
|
679
|
+
}
|
|
680
|
+
|
|
681
|
+
// remove page-level actions and annotations with JS
|
|
682
|
+
for (auto &page : QPDFPageDocumentHelper(qpdf).getAllPages()) {
|
|
683
|
+
auto pageObj = page.getObjectHandle();
|
|
684
|
+
|
|
685
|
+
if (pageObj.hasKey("/AA"))
|
|
686
|
+
pageObj.removeKey("/AA");
|
|
687
|
+
|
|
688
|
+
// strip JS actions from annotations
|
|
689
|
+
if (!pageObj.hasKey("/Annots"))
|
|
690
|
+
continue;
|
|
691
|
+
|
|
692
|
+
auto annots = pageObj.getKey("/Annots");
|
|
693
|
+
if (!annots.isArray())
|
|
694
|
+
continue;
|
|
695
|
+
|
|
696
|
+
for (int i = 0; i < annots.getArrayNItems(); ++i) {
|
|
697
|
+
auto annot = annots.getArrayItem(i);
|
|
698
|
+
if (!annot.isDictionary())
|
|
699
|
+
continue;
|
|
700
|
+
if (annot.hasKey("/AA"))
|
|
701
|
+
annot.removeKey("/AA");
|
|
702
|
+
if (annot.hasKey("/A")) {
|
|
703
|
+
auto action = annot.getKey("/A");
|
|
704
|
+
if (action.isDictionary()) {
|
|
705
|
+
auto s = action.getKey("/S");
|
|
706
|
+
if (s.isName() && s.getName() == "/JavaScript")
|
|
707
|
+
annot.removeKey("/A");
|
|
708
|
+
}
|
|
709
|
+
}
|
|
710
|
+
}
|
|
711
|
+
}
|
|
712
|
+
}
|
|
713
|
+
|
|
714
|
+
// ---------------------------------------------------------------------------
|
|
715
|
+
// Form flattening — merge interactive form field appearances into page
|
|
716
|
+
// content and remove the /AcroForm dictionary
|
|
717
|
+
// ---------------------------------------------------------------------------
|
|
718
|
+
|
|
719
|
+
void flattenForms(QPDF &qpdf) {
|
|
720
|
+
auto root = qpdf.getRoot();
|
|
721
|
+
if (!root.hasKey("/AcroForm"))
|
|
722
|
+
return;
|
|
723
|
+
|
|
724
|
+
// stamp each widget annotation's appearance into the page content,
|
|
725
|
+
// then remove the annotation
|
|
726
|
+
for (auto &page : QPDFPageDocumentHelper(qpdf).getAllPages()) {
|
|
727
|
+
auto pageObj = page.getObjectHandle();
|
|
728
|
+
if (!pageObj.hasKey("/Annots"))
|
|
729
|
+
continue;
|
|
730
|
+
|
|
731
|
+
auto annots = pageObj.getKey("/Annots");
|
|
732
|
+
if (!annots.isArray())
|
|
733
|
+
continue;
|
|
734
|
+
|
|
735
|
+
std::vector<int> widgetIndices;
|
|
736
|
+
for (int i = 0; i < annots.getArrayNItems(); ++i) {
|
|
737
|
+
auto annot = annots.getArrayItem(i);
|
|
738
|
+
if (!annot.isDictionary())
|
|
739
|
+
continue;
|
|
740
|
+
|
|
741
|
+
auto subtype = annot.getKey("/Subtype");
|
|
742
|
+
if (!subtype.isName() || subtype.getName() != "/Widget")
|
|
743
|
+
continue;
|
|
744
|
+
|
|
745
|
+
// check if there's a normal appearance to flatten
|
|
746
|
+
auto ap = annot.getKey("/AP");
|
|
747
|
+
if (!ap.isDictionary())
|
|
748
|
+
continue;
|
|
749
|
+
auto nAppearance = ap.getKey("/N");
|
|
750
|
+
if (!nAppearance.isStream())
|
|
751
|
+
continue;
|
|
752
|
+
|
|
753
|
+
// get widget rectangle
|
|
754
|
+
auto rect = annot.getKey("/Rect");
|
|
755
|
+
if (!rect.isArray() || rect.getArrayNItems() < 4)
|
|
756
|
+
continue;
|
|
757
|
+
|
|
758
|
+
try {
|
|
759
|
+
double x1 = rect.getArrayItem(0).getNumericValue();
|
|
760
|
+
double y1 = rect.getArrayItem(1).getNumericValue();
|
|
761
|
+
double x2 = rect.getArrayItem(2).getNumericValue();
|
|
762
|
+
double y2 = rect.getArrayItem(3).getNumericValue();
|
|
763
|
+
|
|
764
|
+
double w = x2 - x1;
|
|
765
|
+
double h = y2 - y1;
|
|
766
|
+
if (w <= 0 || h <= 0)
|
|
767
|
+
continue;
|
|
768
|
+
|
|
769
|
+
// get appearance stream bounding box for scaling
|
|
770
|
+
auto apDict = nAppearance.getDict();
|
|
771
|
+
double scaleX = 1.0, scaleY = 1.0;
|
|
772
|
+
if (apDict.hasKey("/BBox")) {
|
|
773
|
+
auto bbox = apDict.getKey("/BBox");
|
|
774
|
+
if (bbox.isArray() && bbox.getArrayNItems() >= 4) {
|
|
775
|
+
double bw = bbox.getArrayItem(2).getNumericValue() -
|
|
776
|
+
bbox.getArrayItem(0).getNumericValue();
|
|
777
|
+
double bh = bbox.getArrayItem(3).getNumericValue() -
|
|
778
|
+
bbox.getArrayItem(1).getNumericValue();
|
|
779
|
+
if (bw > 0)
|
|
780
|
+
scaleX = w / bw;
|
|
781
|
+
if (bh > 0)
|
|
782
|
+
scaleY = h / bh;
|
|
783
|
+
}
|
|
784
|
+
}
|
|
785
|
+
|
|
786
|
+
// register appearance as a form XObject on the page
|
|
787
|
+
auto resources = pageObj.getKey("/Resources");
|
|
788
|
+
if (!resources.isDictionary()) {
|
|
789
|
+
resources = QPDFObjectHandle::newDictionary();
|
|
790
|
+
pageObj.replaceKey("/Resources", resources);
|
|
791
|
+
}
|
|
792
|
+
auto xobjects = resources.getKey("/XObject");
|
|
793
|
+
if (!xobjects.isDictionary()) {
|
|
794
|
+
xobjects = QPDFObjectHandle::newDictionary();
|
|
795
|
+
resources.replaceKey("/XObject", xobjects);
|
|
796
|
+
}
|
|
797
|
+
|
|
798
|
+
std::string xobjName = "/FlatForm" + std::to_string(i);
|
|
799
|
+
xobjects.replaceKey(xobjName, nAppearance);
|
|
800
|
+
|
|
801
|
+
// ensure the appearance stream has /Type /XObject /Subtype /Form
|
|
802
|
+
if (!apDict.hasKey("/Type"))
|
|
803
|
+
apDict.replaceKey("/Type", QPDFObjectHandle::newName("/XObject"));
|
|
804
|
+
if (!apDict.hasKey("/Subtype"))
|
|
805
|
+
apDict.replaceKey("/Subtype", QPDFObjectHandle::newName("/Form"));
|
|
806
|
+
|
|
807
|
+
// build content stream snippet to stamp the appearance
|
|
808
|
+
std::string snippet = "q " + std::to_string(scaleX) + " 0 0 " +
|
|
809
|
+
std::to_string(scaleY) + " " +
|
|
810
|
+
std::to_string(x1) + " " + std::to_string(y1) +
|
|
811
|
+
" cm " + xobjName + " Do Q\n";
|
|
812
|
+
|
|
813
|
+
// append to page content stream
|
|
814
|
+
page.addPageContents(QPDFObjectHandle::newStream(&qpdf, snippet),
|
|
815
|
+
false);
|
|
816
|
+
|
|
817
|
+
widgetIndices.push_back(i);
|
|
818
|
+
} catch (...) {
|
|
819
|
+
continue;
|
|
820
|
+
}
|
|
821
|
+
}
|
|
822
|
+
|
|
823
|
+
// remove widget annotations (reverse order to preserve indices)
|
|
824
|
+
for (auto it = widgetIndices.rbegin(); it != widgetIndices.rend(); ++it)
|
|
825
|
+
annots.eraseItem(*it);
|
|
826
|
+
}
|
|
827
|
+
|
|
828
|
+
// remove the /AcroForm dictionary
|
|
829
|
+
root.removeKey("/AcroForm");
|
|
830
|
+
}
|
|
831
|
+
|
|
832
|
+
// ---------------------------------------------------------------------------
|
|
833
|
+
// Page tree flattening — push inherited attributes to pages so QPDFWriter
|
|
834
|
+
// can generate a flat single-level page tree
|
|
835
|
+
// ---------------------------------------------------------------------------
|
|
836
|
+
|
|
837
|
+
void flattenPageTree(QPDF &qpdf) { qpdf.pushInheritedAttributesToPage(); }
|
|
838
|
+
|
|
839
|
+
// ---------------------------------------------------------------------------
|
|
840
|
+
// Content stream minification — normalize whitespace and number formatting
|
|
841
|
+
// to reduce content stream size before Flate compression
|
|
842
|
+
// ---------------------------------------------------------------------------
|
|
843
|
+
|
|
844
|
+
// trims a numeric string: remove trailing zeros after decimal point,
|
|
845
|
+
// remove the decimal point if it becomes the last char,
|
|
846
|
+
// and strip a leading zero for values between -1 and 1.
|
|
847
|
+
static std::string trimNumber(const std::string &s) {
|
|
848
|
+
// only process strings that look like decimal numbers
|
|
849
|
+
if (s.find('.') == std::string::npos)
|
|
850
|
+
return s;
|
|
851
|
+
|
|
852
|
+
std::string result = s;
|
|
853
|
+
|
|
854
|
+
// strip trailing zeros after decimal point
|
|
855
|
+
size_t dot = result.find('.');
|
|
856
|
+
if (dot != std::string::npos) {
|
|
857
|
+
size_t last = result.size() - 1;
|
|
858
|
+
while (last > dot && result[last] == '0')
|
|
859
|
+
--last;
|
|
860
|
+
if (last == dot)
|
|
861
|
+
result.erase(dot); // remove the dot too (e.g. "1." → "1")
|
|
862
|
+
else
|
|
863
|
+
result.erase(last + 1);
|
|
864
|
+
}
|
|
865
|
+
|
|
866
|
+
// strip leading zero for values like "0.5" → ".5" or "-0.5" → "-.5"
|
|
867
|
+
if (result.size() >= 2 && result[0] == '0' && result[1] == '.')
|
|
868
|
+
result.erase(0, 1);
|
|
869
|
+
else if (result.size() >= 3 && result[0] == '-' && result[1] == '0' &&
|
|
870
|
+
result[2] == '.')
|
|
871
|
+
result.erase(1, 1);
|
|
872
|
+
|
|
873
|
+
return result;
|
|
874
|
+
}
|
|
875
|
+
|
|
876
|
+
void minifyContentStreams(QPDF &qpdf) {
|
|
877
|
+
for (auto &page : QPDFPageDocumentHelper(qpdf).getAllPages()) {
|
|
878
|
+
auto pageObj = page.getObjectHandle();
|
|
879
|
+
auto contents = pageObj.getKey("/Contents");
|
|
880
|
+
|
|
881
|
+
if (!contents.isStream())
|
|
882
|
+
continue;
|
|
883
|
+
|
|
884
|
+
std::string raw;
|
|
885
|
+
try {
|
|
886
|
+
auto buf = contents.getStreamData(qpdf_dl_generalized);
|
|
887
|
+
raw.assign(reinterpret_cast<const char *>(buf->getBuffer()),
|
|
888
|
+
buf->getSize());
|
|
889
|
+
} catch (...) {
|
|
890
|
+
continue;
|
|
891
|
+
}
|
|
892
|
+
|
|
893
|
+
// tokenize preserving string literals and hex strings intact
|
|
894
|
+
std::string minified;
|
|
895
|
+
minified.reserve(raw.size());
|
|
896
|
+
bool needSpace = false;
|
|
897
|
+
size_t pos = 0;
|
|
898
|
+
|
|
899
|
+
while (pos < raw.size()) {
|
|
900
|
+
char ch = raw[pos];
|
|
901
|
+
|
|
902
|
+
// skip whitespace
|
|
903
|
+
if (ch == ' ' || ch == '\t' || ch == '\r' || ch == '\n') {
|
|
904
|
+
if (!minified.empty())
|
|
905
|
+
needSpace = true;
|
|
906
|
+
++pos;
|
|
907
|
+
continue;
|
|
908
|
+
}
|
|
909
|
+
|
|
910
|
+
// comments — skip to end of line
|
|
911
|
+
if (ch == '%') {
|
|
912
|
+
while (pos < raw.size() && raw[pos] != '\n')
|
|
913
|
+
++pos;
|
|
914
|
+
continue;
|
|
915
|
+
}
|
|
916
|
+
|
|
917
|
+
// literal string — copy verbatim
|
|
918
|
+
if (ch == '(') {
|
|
919
|
+
if (needSpace) {
|
|
920
|
+
minified += '\n';
|
|
921
|
+
needSpace = false;
|
|
922
|
+
}
|
|
923
|
+
int depth = 1;
|
|
924
|
+
minified += '(';
|
|
925
|
+
++pos;
|
|
926
|
+
while (pos < raw.size() && depth > 0) {
|
|
927
|
+
if (raw[pos] == '\\') {
|
|
928
|
+
minified += raw[pos++];
|
|
929
|
+
if (pos < raw.size())
|
|
930
|
+
minified += raw[pos++];
|
|
931
|
+
} else {
|
|
932
|
+
if (raw[pos] == '(')
|
|
933
|
+
++depth;
|
|
934
|
+
else if (raw[pos] == ')')
|
|
935
|
+
--depth;
|
|
936
|
+
minified += raw[pos++];
|
|
937
|
+
}
|
|
938
|
+
}
|
|
939
|
+
needSpace = true;
|
|
940
|
+
continue;
|
|
941
|
+
}
|
|
942
|
+
|
|
943
|
+
// hex string — copy verbatim
|
|
944
|
+
if (ch == '<' && pos + 1 < raw.size() && raw[pos + 1] != '<') {
|
|
945
|
+
if (needSpace) {
|
|
946
|
+
minified += '\n';
|
|
947
|
+
needSpace = false;
|
|
948
|
+
}
|
|
949
|
+
minified += '<';
|
|
950
|
+
++pos;
|
|
951
|
+
while (pos < raw.size() && raw[pos] != '>') {
|
|
952
|
+
if (!std::isspace(static_cast<unsigned char>(raw[pos])))
|
|
953
|
+
minified += raw[pos];
|
|
954
|
+
++pos;
|
|
955
|
+
}
|
|
956
|
+
if (pos < raw.size()) {
|
|
957
|
+
minified += '>';
|
|
958
|
+
++pos;
|
|
959
|
+
}
|
|
960
|
+
needSpace = true;
|
|
961
|
+
continue;
|
|
962
|
+
}
|
|
963
|
+
|
|
964
|
+
// dict delimiters << >> — self-delimiting, no space needed around them
|
|
965
|
+
if (ch == '<' && pos + 1 < raw.size() && raw[pos + 1] == '<') {
|
|
966
|
+
if (needSpace) {
|
|
967
|
+
minified += '\n';
|
|
968
|
+
needSpace = false;
|
|
969
|
+
}
|
|
970
|
+
minified += "<<";
|
|
971
|
+
pos += 2;
|
|
972
|
+
continue;
|
|
973
|
+
}
|
|
974
|
+
if (ch == '>' && pos + 1 < raw.size() && raw[pos + 1] == '>') {
|
|
975
|
+
minified += ">>";
|
|
976
|
+
pos += 2;
|
|
977
|
+
needSpace = true;
|
|
978
|
+
continue;
|
|
979
|
+
}
|
|
980
|
+
|
|
981
|
+
// array delimiters — self-delimiting
|
|
982
|
+
if (ch == '[' || ch == ']') {
|
|
983
|
+
if (needSpace && ch == '[') {
|
|
984
|
+
minified += '\n';
|
|
985
|
+
needSpace = false;
|
|
986
|
+
}
|
|
987
|
+
minified += ch;
|
|
988
|
+
++pos;
|
|
989
|
+
if (ch == ']')
|
|
990
|
+
needSpace = true;
|
|
991
|
+
continue;
|
|
992
|
+
}
|
|
993
|
+
|
|
994
|
+
// name — starts with /
|
|
995
|
+
if (ch == '/') {
|
|
996
|
+
if (needSpace) {
|
|
997
|
+
minified += '\n';
|
|
998
|
+
needSpace = false;
|
|
999
|
+
}
|
|
1000
|
+
size_t start = pos;
|
|
1001
|
+
++pos;
|
|
1002
|
+
while (pos < raw.size() &&
|
|
1003
|
+
!std::isspace(static_cast<unsigned char>(raw[pos])) &&
|
|
1004
|
+
raw[pos] != '/' && raw[pos] != '[' && raw[pos] != ']' &&
|
|
1005
|
+
raw[pos] != '<' && raw[pos] != '>' && raw[pos] != '(' &&
|
|
1006
|
+
raw[pos] != ')')
|
|
1007
|
+
++pos;
|
|
1008
|
+
minified.append(raw, start, pos - start);
|
|
1009
|
+
needSpace = true;
|
|
1010
|
+
continue;
|
|
1011
|
+
}
|
|
1012
|
+
|
|
1013
|
+
// regular token (number, operator)
|
|
1014
|
+
{
|
|
1015
|
+
if (needSpace) {
|
|
1016
|
+
minified += '\n';
|
|
1017
|
+
needSpace = false;
|
|
1018
|
+
}
|
|
1019
|
+
size_t start = pos;
|
|
1020
|
+
while (pos < raw.size() &&
|
|
1021
|
+
!std::isspace(static_cast<unsigned char>(raw[pos])) &&
|
|
1022
|
+
raw[pos] != '/' && raw[pos] != '[' && raw[pos] != ']' &&
|
|
1023
|
+
raw[pos] != '<' && raw[pos] != '>' && raw[pos] != '(' &&
|
|
1024
|
+
raw[pos] != ')')
|
|
1025
|
+
++pos;
|
|
1026
|
+
|
|
1027
|
+
std::string token(raw, start, pos - start);
|
|
1028
|
+
|
|
1029
|
+
// trim numeric formatting
|
|
1030
|
+
if (!token.empty() &&
|
|
1031
|
+
(token[0] == '-' || token[0] == '+' || token[0] == '.' ||
|
|
1032
|
+
(token[0] >= '0' && token[0] <= '9'))) {
|
|
1033
|
+
token = trimNumber(token);
|
|
1034
|
+
}
|
|
1035
|
+
|
|
1036
|
+
minified += token;
|
|
1037
|
+
needSpace = true;
|
|
1038
|
+
}
|
|
1039
|
+
}
|
|
1040
|
+
|
|
1041
|
+
// only replace if we actually reduced the size
|
|
1042
|
+
if (minified.size() >= raw.size())
|
|
1043
|
+
continue;
|
|
1044
|
+
|
|
1045
|
+
contents.replaceStreamData(minified, QPDFObjectHandle::newNull(),
|
|
1046
|
+
QPDFObjectHandle::newNull());
|
|
1047
|
+
}
|
|
1048
|
+
}
|