qpdf-compress 0.2.0 → 0.4.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,1048 @@
1
+ #include "optimize.h"
2
+ #include "font_subset.h"
3
+ #include "images.h"
4
+
5
+ #include <cctype>
6
+ #include <cstdint>
7
+ #include <cstdlib>
8
+ #include <cstring>
9
+ #include <map>
10
+ #include <set>
11
+ #include <string>
12
+ #include <unordered_map>
13
+ #include <vector>
14
+
15
+ #include <qpdf/Buffer.hh>
16
+ #include <qpdf/QPDFObjectHandle.hh>
17
+ #include <qpdf/QPDFPageDocumentHelper.hh>
18
+ #include <qpdf/QPDFPageObjectHelper.hh>
19
+
20
+ // ---------------------------------------------------------------------------
21
+ // Metadata stripping
22
+ // ---------------------------------------------------------------------------
23
+
24
+ void stripMetadata(QPDF &qpdf) {
25
+ auto root = qpdf.getRoot();
26
+
27
+ // remove XMP metadata stream
28
+ if (root.hasKey("/Metadata"))
29
+ root.removeKey("/Metadata");
30
+
31
+ // remove document info dictionary
32
+ auto trailer = qpdf.getTrailer();
33
+ if (trailer.hasKey("/Info"))
34
+ trailer.removeKey("/Info");
35
+
36
+ // remove page-level metadata and PieceInfo
37
+ for (auto &page : QPDFPageDocumentHelper(qpdf).getAllPages()) {
38
+ auto pageObj = page.getObjectHandle();
39
+ if (pageObj.hasKey("/Metadata"))
40
+ pageObj.removeKey("/Metadata");
41
+ if (pageObj.hasKey("/PieceInfo"))
42
+ pageObj.removeKey("/PieceInfo");
43
+ }
44
+
45
+ // remove embedded thumbnails
46
+ for (auto &page : QPDFPageDocumentHelper(qpdf).getAllPages()) {
47
+ auto pageObj = page.getObjectHandle();
48
+ if (pageObj.hasKey("/Thumb"))
49
+ pageObj.removeKey("/Thumb");
50
+ }
51
+
52
+ // remove MarkInfo and page labels (optional metadata)
53
+ if (root.hasKey("/MarkInfo"))
54
+ root.removeKey("/MarkInfo");
55
+ }
56
+
57
+ // ---------------------------------------------------------------------------
58
+ // Remove unused font resources
59
+ // ---------------------------------------------------------------------------
60
+
61
+ void removeUnusedFonts(QPDF &qpdf) {
62
+ for (auto &page : QPDFPageDocumentHelper(qpdf).getAllPages()) {
63
+ auto pageObj = page.getObjectHandle();
64
+ auto resources = pageObj.getKey("/Resources");
65
+ if (!resources.isDictionary())
66
+ continue;
67
+ auto fonts = resources.getKey("/Font");
68
+ if (!fonts.isDictionary())
69
+ continue;
70
+
71
+ // collect all font names referenced in this page's content stream(s)
72
+ std::set<std::string> usedFonts;
73
+
74
+ try {
75
+ // get unparsed content stream data
76
+ auto contents = pageObj.getKey("/Contents");
77
+ std::string contentStr;
78
+
79
+ if (contents.isStream()) {
80
+ auto buf = contents.getStreamData(qpdf_dl_generalized);
81
+ contentStr.assign(reinterpret_cast<const char *>(buf->getBuffer()),
82
+ buf->getSize());
83
+ } else if (contents.isArray()) {
84
+ for (int i = 0; i < contents.getArrayNItems(); ++i) {
85
+ auto stream = contents.getArrayItem(i);
86
+ if (stream.isStream()) {
87
+ auto buf = stream.getStreamData(qpdf_dl_generalized);
88
+ contentStr.append(reinterpret_cast<const char *>(buf->getBuffer()),
89
+ buf->getSize());
90
+ contentStr += '\n';
91
+ }
92
+ }
93
+ }
94
+
95
+ // scan for /FontName references — Tf operator uses font name
96
+ // pattern: /FontName <size> Tf
97
+ for (auto &fontKey : fonts.getKeys()) {
98
+ // fontKey includes the leading '/', e.g. "/F1"
99
+ if (contentStr.find(fontKey) != std::string::npos)
100
+ usedFonts.insert(fontKey);
101
+ }
102
+ } catch (...) {
103
+ continue; // skip this page if content can't be read
104
+ }
105
+
106
+ // remove fonts that are not referenced in the content stream
107
+ auto allFontKeys = fonts.getKeys();
108
+ for (auto &fontKey : allFontKeys) {
109
+ if (usedFonts.find(fontKey) == usedFonts.end())
110
+ fonts.removeKey(fontKey);
111
+ }
112
+ }
113
+ }
114
+
115
+ // ---------------------------------------------------------------------------
116
+ // Content stream coalescing — merge multiple content streams per page into one
117
+ // ---------------------------------------------------------------------------
118
+
119
+ void coalesceContentStreams(QPDF &qpdf) {
120
+ for (auto &page : QPDFPageDocumentHelper(qpdf).getAllPages()) {
121
+ auto pageObj = page.getObjectHandle();
122
+ auto contents = pageObj.getKey("/Contents");
123
+
124
+ // only coalesce if there are multiple content streams (array)
125
+ if (contents.isArray() && contents.getArrayNItems() > 1) {
126
+ page.coalesceContentStreams();
127
+ }
128
+ }
129
+ }
130
+
131
+ // ---------------------------------------------------------------------------
132
+ // Deduplicate identical non-image streams (fonts, ICC profiles, etc.)
133
+ // ---------------------------------------------------------------------------
134
+
135
+ void deduplicateStreams(QPDF &qpdf) {
136
+ // collect all stream objects and their raw data hashes
137
+ struct StreamEntry {
138
+ QPDFObjGen og;
139
+ size_t dataSize;
140
+ QPDFObjectHandle handle;
141
+ };
142
+
143
+ std::unordered_map<uint64_t, std::vector<StreamEntry>> hashGroups;
144
+ std::set<QPDFObjGen> imageObjGens;
145
+
146
+ // collect image object IDs to skip them (already handled by
147
+ // deduplicateImages)
148
+ forEachImage(qpdf, [&](const std::string &, QPDFObjectHandle xobj,
149
+ QPDFObjectHandle, QPDFPageObjectHelper &) {
150
+ imageObjGens.insert(xobj.getObjGen());
151
+ });
152
+
153
+ for (auto &obj : qpdf.getAllObjects()) {
154
+ if (!obj.isStream())
155
+ continue;
156
+
157
+ auto og = obj.getObjGen();
158
+
159
+ // skip images
160
+ if (imageObjGens.count(og))
161
+ continue;
162
+
163
+ try {
164
+ auto rawData = obj.getRawStreamData();
165
+ size_t size = rawData->getSize();
166
+ if (size == 0)
167
+ continue;
168
+
169
+ // FNV-1a hash
170
+ uint64_t hash = 14695981039346656037ULL;
171
+ auto *p = rawData->getBuffer();
172
+ for (size_t i = 0; i < size; ++i) {
173
+ hash ^= static_cast<uint64_t>(p[i]);
174
+ hash *= 1099511628211ULL;
175
+ }
176
+
177
+ hashGroups[hash].push_back({og, size, obj});
178
+ } catch (...) {
179
+ }
180
+ }
181
+
182
+ // find duplicates via full byte comparison
183
+ std::map<QPDFObjGen, QPDFObjectHandle> replacements;
184
+
185
+ for (auto &[hash, group] : hashGroups) {
186
+ if (group.size() < 2)
187
+ continue;
188
+
189
+ for (size_t i = 0; i < group.size(); ++i) {
190
+ if (replacements.count(group[i].og))
191
+ continue;
192
+
193
+ auto rawI = group[i].handle.getRawStreamData();
194
+ for (size_t j = i + 1; j < group.size(); ++j) {
195
+ if (replacements.count(group[j].og))
196
+ continue;
197
+
198
+ auto rawJ = group[j].handle.getRawStreamData();
199
+ if (rawI->getSize() != rawJ->getSize())
200
+ continue;
201
+
202
+ if (memcmp(rawI->getBuffer(), rawJ->getBuffer(), rawI->getSize()) ==
203
+ 0) {
204
+ // verify stream dictionaries are compatible
205
+ auto dictI = group[i].handle.getDict();
206
+ auto dictJ = group[j].handle.getDict();
207
+
208
+ auto filterI = dictI.getKey("/Filter");
209
+ auto filterJ = dictJ.getKey("/Filter");
210
+ bool filtersMatch = (filterI.isName() && filterJ.isName() &&
211
+ filterI.getName() == filterJ.getName()) ||
212
+ (!filterI.isName() && !filterJ.isName());
213
+
214
+ // also verify DecodeParms match — different predictors on
215
+ // identical raw bytes would produce different decoded content
216
+ auto dpI = dictI.getKey("/DecodeParms");
217
+ auto dpJ = dictJ.getKey("/DecodeParms");
218
+ bool paramsMatch = (dpI.isNull() && dpJ.isNull()) ||
219
+ (!dpI.isNull() && !dpJ.isNull() &&
220
+ dpI.unparse() == dpJ.unparse());
221
+
222
+ if (filtersMatch && paramsMatch)
223
+ replacements[group[j].og] = group[i].handle;
224
+ }
225
+ }
226
+ }
227
+ }
228
+
229
+ if (replacements.empty())
230
+ return;
231
+
232
+ // rewrite references: scan all objects for indirect references to duplicates
233
+ for (auto &obj : qpdf.getAllObjects()) {
234
+ if (!obj.isDictionary() && !obj.isStream())
235
+ continue;
236
+
237
+ auto dict = obj.isStream() ? obj.getDict() : obj;
238
+ for (auto &key : dict.getKeys()) {
239
+ auto val = dict.getKey(key);
240
+ if (val.isIndirect()) {
241
+ auto it = replacements.find(val.getObjGen());
242
+ if (it != replacements.end())
243
+ dict.replaceKey(key, it->second);
244
+ }
245
+ }
246
+ }
247
+ }
248
+
249
+ // ---------------------------------------------------------------------------
250
+ // Font subsetting — remove unused glyphs from TrueType/CIDFont fonts
251
+ // ---------------------------------------------------------------------------
252
+
253
+ // collects all Unicode code points used in a page's content stream by
254
+ // parsing text-showing operators (Tj, TJ, ', ")
255
+ static std::set<uint16_t> collectUsedCodes(const std::string &contentStr,
256
+ const std::string &fontKey,
257
+ QPDFObjectHandle fontObj) {
258
+ std::set<uint16_t> usedCodes;
259
+
260
+ // simple scan: find all string operands between font selection (Tf) and
261
+ // text-showing operators. For simplicity, collect all hex/literal string
262
+ // bytes used when this font is active.
263
+
264
+ // check if this font uses 2-byte CID encoding
265
+ auto subtypeKey = fontObj.getKey("/Subtype");
266
+ bool isCIDFont = subtypeKey.isName() && subtypeKey.getName() == "/Type0";
267
+
268
+ // find all ranges where this font is active and collect string bytes
269
+ bool fontActive = false;
270
+ size_t pos = 0;
271
+
272
+ while (pos < contentStr.size()) {
273
+ // skip whitespace
274
+ while (pos < contentStr.size() && std::isspace(contentStr[pos]))
275
+ ++pos;
276
+
277
+ if (pos >= contentStr.size())
278
+ break;
279
+
280
+ // check for font selection: /FontName ... Tf
281
+ if (contentStr[pos] == '/') {
282
+ // read name
283
+ size_t nameStart = pos;
284
+ ++pos;
285
+ while (pos < contentStr.size() && !std::isspace(contentStr[pos]) &&
286
+ contentStr[pos] != '/' && contentStr[pos] != '<' &&
287
+ contentStr[pos] != '(' && contentStr[pos] != '[')
288
+ ++pos;
289
+ std::string name = contentStr.substr(nameStart, pos - nameStart);
290
+
291
+ // look ahead for Tf
292
+ size_t lookAhead = pos;
293
+ while (lookAhead < contentStr.size() &&
294
+ std::isspace(contentStr[lookAhead]))
295
+ ++lookAhead;
296
+ // skip number
297
+ while (
298
+ lookAhead < contentStr.size() &&
299
+ (std::isdigit(contentStr[lookAhead]) || contentStr[lookAhead] == '.'))
300
+ ++lookAhead;
301
+ while (lookAhead < contentStr.size() &&
302
+ std::isspace(contentStr[lookAhead]))
303
+ ++lookAhead;
304
+ if (lookAhead + 1 < contentStr.size() && contentStr[lookAhead] == 'T' &&
305
+ contentStr[lookAhead + 1] == 'f') {
306
+ fontActive = (name == fontKey);
307
+ pos = lookAhead + 2;
308
+ continue;
309
+ }
310
+ }
311
+
312
+ // collect string data when our font is active
313
+ if (fontActive && contentStr[pos] == '(') {
314
+ // literal string
315
+ ++pos;
316
+ int depth = 1;
317
+ while (pos < contentStr.size() && depth > 0) {
318
+ if (contentStr[pos] == '\\') {
319
+ ++pos; // skip escaped char
320
+ if (pos < contentStr.size())
321
+ ++pos;
322
+ continue;
323
+ }
324
+ if (contentStr[pos] == '(')
325
+ ++depth;
326
+ else if (contentStr[pos] == ')')
327
+ --depth;
328
+ if (depth > 0) {
329
+ if (isCIDFont && pos + 1 < contentStr.size()) {
330
+ uint16_t code = (static_cast<uint8_t>(contentStr[pos]) << 8) |
331
+ static_cast<uint8_t>(contentStr[pos + 1]);
332
+ usedCodes.insert(code);
333
+ ++pos;
334
+ } else {
335
+ usedCodes.insert(static_cast<uint8_t>(contentStr[pos]));
336
+ }
337
+ ++pos;
338
+ }
339
+ }
340
+ if (depth == 0)
341
+ ++pos; // skip closing ')'
342
+ continue;
343
+ }
344
+
345
+ if (fontActive && contentStr[pos] == '<') {
346
+ // hex string
347
+ ++pos;
348
+ std::vector<uint8_t> hexBytes;
349
+ while (pos < contentStr.size() && contentStr[pos] != '>') {
350
+ if (std::isxdigit(contentStr[pos])) {
351
+ char hex[3] = {contentStr[pos], '0', '\0'};
352
+ if (pos + 1 < contentStr.size() &&
353
+ std::isxdigit(contentStr[pos + 1])) {
354
+ hex[1] = contentStr[pos + 1];
355
+ ++pos;
356
+ }
357
+ hexBytes.push_back(
358
+ static_cast<uint8_t>(std::strtol(hex, nullptr, 16)));
359
+ }
360
+ ++pos;
361
+ }
362
+ if (pos < contentStr.size())
363
+ ++pos; // skip '>'
364
+
365
+ if (isCIDFont) {
366
+ for (size_t b = 0; b + 1 < hexBytes.size(); b += 2) {
367
+ uint16_t code = (hexBytes[b] << 8) | hexBytes[b + 1];
368
+ usedCodes.insert(code);
369
+ }
370
+ } else {
371
+ for (auto b : hexBytes)
372
+ usedCodes.insert(b);
373
+ }
374
+ continue;
375
+ }
376
+
377
+ // skip other tokens
378
+ while (pos < contentStr.size() && !std::isspace(contentStr[pos]))
379
+ ++pos;
380
+ }
381
+
382
+ return usedCodes;
383
+ }
384
+
385
+ void subsetFonts(QPDF &qpdf) {
386
+ // collect used character codes per font object
387
+ std::map<QPDFObjGen, std::set<uint16_t>> fontUsedCodes;
388
+
389
+ for (auto &page : QPDFPageDocumentHelper(qpdf).getAllPages()) {
390
+ auto pageObj = page.getObjectHandle();
391
+ auto resources = pageObj.getKey("/Resources");
392
+ if (!resources.isDictionary())
393
+ continue;
394
+ auto fonts = resources.getKey("/Font");
395
+ if (!fonts.isDictionary())
396
+ continue;
397
+
398
+ // decode content stream once per page (shared across all fonts)
399
+ auto contents = pageObj.getKey("/Contents");
400
+ std::string contentStr;
401
+ try {
402
+ if (contents.isStream()) {
403
+ auto buf = contents.getStreamData(qpdf_dl_generalized);
404
+ contentStr.assign(reinterpret_cast<const char *>(buf->getBuffer()),
405
+ buf->getSize());
406
+ } else if (contents.isArray()) {
407
+ for (int i = 0; i < contents.getArrayNItems(); ++i) {
408
+ auto stream = contents.getArrayItem(i);
409
+ if (stream.isStream()) {
410
+ auto buf = stream.getStreamData(qpdf_dl_generalized);
411
+ contentStr.append(reinterpret_cast<const char *>(buf->getBuffer()),
412
+ buf->getSize());
413
+ contentStr += '\n';
414
+ }
415
+ }
416
+ }
417
+ } catch (...) {
418
+ continue;
419
+ }
420
+
421
+ for (auto &key : fonts.getKeys()) {
422
+ auto fontObj = fonts.getKey(key);
423
+ if (!fontObj.isDictionary())
424
+ continue;
425
+
426
+ auto og = fontObj.getObjGen();
427
+ auto codes = collectUsedCodes(contentStr, key, fontObj);
428
+ fontUsedCodes[og].insert(codes.begin(), codes.end());
429
+ }
430
+ }
431
+
432
+ // for each font with a /Widths array, zero out widths for unused glyphs
433
+ // and truncate trailing zeros
434
+ for (auto &[og, usedCodes] : fontUsedCodes) {
435
+ auto fontObj = qpdf.getObjectByObjGen(og);
436
+ if (!fontObj.isDictionary())
437
+ continue;
438
+
439
+ auto subtype = fontObj.getKey("/Subtype");
440
+ if (!subtype.isName())
441
+ continue;
442
+
443
+ // only handle simple fonts with /Widths arrays (TrueType, Type1)
444
+ if (subtype.getName() != "/TrueType" && subtype.getName() != "/Type1")
445
+ continue;
446
+
447
+ auto widths = fontObj.getKey("/Widths");
448
+ auto firstCharObj = fontObj.getKey("/FirstChar");
449
+ if (!widths.isArray() || !firstCharObj.isInteger())
450
+ continue;
451
+
452
+ int firstChar = static_cast<int>(firstCharObj.getIntValue());
453
+ int widthCount = widths.getArrayNItems();
454
+
455
+ // zero out widths for unused character codes
456
+ bool modified = false;
457
+ for (int i = 0; i < widthCount; ++i) {
458
+ int charCode = firstChar + i;
459
+ if (usedCodes.find(static_cast<uint16_t>(charCode)) == usedCodes.end()) {
460
+ auto w = widths.getArrayItem(i);
461
+ if (w.isInteger() && w.getIntValue() != 0) {
462
+ widths.setArrayItem(i, QPDFObjectHandle::newInteger(0));
463
+ modified = true;
464
+ } else if (w.isReal() && std::stod(w.getRealValue()) != 0.0) {
465
+ widths.setArrayItem(i, QPDFObjectHandle::newInteger(0));
466
+ modified = true;
467
+ }
468
+ }
469
+ }
470
+
471
+ if (!modified)
472
+ continue;
473
+
474
+ // trim trailing zero-width entries and adjust /LastChar
475
+ int lastUsed = widthCount - 1;
476
+ while (lastUsed >= 0) {
477
+ auto w = widths.getArrayItem(lastUsed);
478
+ if (w.isInteger() && w.getIntValue() == 0)
479
+ --lastUsed;
480
+ else
481
+ break;
482
+ }
483
+
484
+ if (lastUsed < widthCount - 1) {
485
+ // rebuild the widths array with only the needed entries
486
+ auto newWidths = QPDFObjectHandle::newArray();
487
+ for (int i = 0; i <= lastUsed; ++i)
488
+ newWidths.appendItem(widths.getArrayItem(i));
489
+
490
+ fontObj.replaceKey("/Widths", newWidths);
491
+ fontObj.replaceKey("/LastChar",
492
+ QPDFObjectHandle::newInteger(firstChar + lastUsed));
493
+ }
494
+ }
495
+
496
+ // true font subsetting: strip unused glyph outlines from TrueType fonts
497
+ std::set<QPDFObjGen> processedFonts;
498
+ for (auto &[og, usedCodes] : fontUsedCodes) {
499
+ if (processedFonts.count(og))
500
+ continue;
501
+ processedFonts.insert(og);
502
+
503
+ auto fontObj = qpdf.getObjectByObjGen(og);
504
+ if (!fontObj.isDictionary())
505
+ continue;
506
+
507
+ auto subtype = fontObj.getKey("/Subtype");
508
+ if (!subtype.isName() || subtype.getName() != "/TrueType")
509
+ continue;
510
+
511
+ auto descriptor = fontObj.getKey("/FontDescriptor");
512
+ if (!descriptor.isDictionary())
513
+ continue;
514
+
515
+ // TrueType fonts use /FontFile2
516
+ if (!descriptor.hasKey("/FontFile2"))
517
+ continue;
518
+
519
+ auto fontFile = descriptor.getKey("/FontFile2");
520
+ if (!fontFile.isStream())
521
+ continue;
522
+
523
+ try {
524
+ auto fontData = fontFile.getStreamData(qpdf_dl_all);
525
+ const uint8_t *ttfData = fontData->getBuffer();
526
+ size_t ttfSize = fontData->getSize();
527
+
528
+ // map character codes → glyph IDs via cmap
529
+ auto glyphIds = mapCodesToGlyphIds(ttfData, ttfSize, usedCodes);
530
+
531
+ // skip if we'd keep all or nearly all glyphs
532
+ // (subsetting overhead wouldn't be worth it)
533
+ if (glyphIds.size() >= 200)
534
+ continue;
535
+
536
+ std::vector<uint8_t> subsetFont;
537
+ if (!subsetTrueTypeFont(ttfData, ttfSize, glyphIds, subsetFont))
538
+ continue;
539
+
540
+ // only replace if the subset is smaller than the original uncompressed
541
+ // font (both will be Flate-compressed by QPDFWriter, so comparing
542
+ // uncompressed sizes is the fair comparison)
543
+ if (subsetFont.size() >= ttfSize)
544
+ continue;
545
+
546
+ std::string fontStr(reinterpret_cast<char *>(subsetFont.data()),
547
+ subsetFont.size());
548
+ fontFile.replaceStreamData(fontStr, QPDFObjectHandle::newNull(),
549
+ QPDFObjectHandle::newNull());
550
+ } catch (...) {
551
+ continue;
552
+ }
553
+ }
554
+ }
555
+
556
+ // ---------------------------------------------------------------------------
557
+ // ICC profile stripping — replace ICCBased color spaces with Device
558
+ // equivalents
559
+ // ---------------------------------------------------------------------------
560
+
561
+ void stripIccProfiles(QPDF &qpdf) {
562
+ std::set<QPDFObjGen> processed;
563
+
564
+ // strip ICC profiles from images
565
+ forEachImage(qpdf, [&](const std::string &, QPDFObjectHandle xobj,
566
+ QPDFObjectHandle, QPDFPageObjectHelper &) {
567
+ auto og = xobj.getObjGen();
568
+ if (processed.count(og))
569
+ return;
570
+ processed.insert(og);
571
+
572
+ auto dict = xobj.getDict();
573
+ auto cs = dict.getKey("/ColorSpace");
574
+
575
+ if (!cs.isArray() || cs.getArrayNItems() < 2)
576
+ return;
577
+
578
+ auto csName = cs.getArrayItem(0);
579
+ if (!csName.isName() || csName.getName() != "/ICCBased")
580
+ return;
581
+
582
+ auto profile = cs.getArrayItem(1);
583
+ if (!profile.isStream())
584
+ return;
585
+
586
+ auto n = profile.getDict().getKey("/N");
587
+ if (!n.isInteger())
588
+ return;
589
+
590
+ int components = static_cast<int>(n.getIntValue());
591
+ if (components == 3)
592
+ dict.replaceKey("/ColorSpace", QPDFObjectHandle::newName("/DeviceRGB"));
593
+ else if (components == 1)
594
+ dict.replaceKey("/ColorSpace", QPDFObjectHandle::newName("/DeviceGray"));
595
+ else if (components == 4)
596
+ dict.replaceKey("/ColorSpace", QPDFObjectHandle::newName("/DeviceCMYK"));
597
+ });
598
+
599
+ // strip ICC profiles from page-level color space resources
600
+ for (auto &page : QPDFPageDocumentHelper(qpdf).getAllPages()) {
601
+ auto resources = page.getObjectHandle().getKey("/Resources");
602
+ if (!resources.isDictionary())
603
+ continue;
604
+
605
+ auto colorSpaces = resources.getKey("/ColorSpace");
606
+ if (!colorSpaces.isDictionary())
607
+ continue;
608
+
609
+ for (auto &key : colorSpaces.getKeys()) {
610
+ auto cs = colorSpaces.getKey(key);
611
+ if (!cs.isArray() || cs.getArrayNItems() < 2)
612
+ continue;
613
+
614
+ auto csName = cs.getArrayItem(0);
615
+ if (!csName.isName() || csName.getName() != "/ICCBased")
616
+ continue;
617
+
618
+ auto profile = cs.getArrayItem(1);
619
+ if (!profile.isStream())
620
+ continue;
621
+
622
+ auto n = profile.getDict().getKey("/N");
623
+ if (!n.isInteger())
624
+ continue;
625
+
626
+ int components = static_cast<int>(n.getIntValue());
627
+ if (components == 3)
628
+ colorSpaces.replaceKey(key, QPDFObjectHandle::newName("/DeviceRGB"));
629
+ else if (components == 1)
630
+ colorSpaces.replaceKey(key, QPDFObjectHandle::newName("/DeviceGray"));
631
+ else if (components == 4)
632
+ colorSpaces.replaceKey(key, QPDFObjectHandle::newName("/DeviceCMYK"));
633
+ }
634
+ }
635
+ }
636
+
637
+ // ---------------------------------------------------------------------------
638
+ // Embedded file stripping — remove /EmbeddedFiles from the name tree
639
+ // ---------------------------------------------------------------------------
640
+
641
+ void stripEmbeddedFiles(QPDF &qpdf) {
642
+ auto root = qpdf.getRoot();
643
+ if (!root.hasKey("/Names"))
644
+ return;
645
+
646
+ auto names = root.getKey("/Names");
647
+ if (!names.isDictionary())
648
+ return;
649
+
650
+ if (names.hasKey("/EmbeddedFiles"))
651
+ names.removeKey("/EmbeddedFiles");
652
+
653
+ // if /Names is now empty, remove it too
654
+ if (names.getKeys().empty())
655
+ root.removeKey("/Names");
656
+ }
657
+
658
+ // ---------------------------------------------------------------------------
659
+ // JavaScript and action removal — strip JS, open actions, and additional
660
+ // actions from the catalog and all pages
661
+ // ---------------------------------------------------------------------------
662
+
663
+ void stripJavaScript(QPDF &qpdf) {
664
+ auto root = qpdf.getRoot();
665
+
666
+ // remove document-level open action
667
+ if (root.hasKey("/OpenAction"))
668
+ root.removeKey("/OpenAction");
669
+
670
+ // remove document-level additional actions
671
+ if (root.hasKey("/AA"))
672
+ root.removeKey("/AA");
673
+
674
+ // remove /JavaScript name tree
675
+ if (root.hasKey("/Names")) {
676
+ auto names = root.getKey("/Names");
677
+ if (names.isDictionary() && names.hasKey("/JavaScript"))
678
+ names.removeKey("/JavaScript");
679
+ }
680
+
681
+ // remove page-level actions and annotations with JS
682
+ for (auto &page : QPDFPageDocumentHelper(qpdf).getAllPages()) {
683
+ auto pageObj = page.getObjectHandle();
684
+
685
+ if (pageObj.hasKey("/AA"))
686
+ pageObj.removeKey("/AA");
687
+
688
+ // strip JS actions from annotations
689
+ if (!pageObj.hasKey("/Annots"))
690
+ continue;
691
+
692
+ auto annots = pageObj.getKey("/Annots");
693
+ if (!annots.isArray())
694
+ continue;
695
+
696
+ for (int i = 0; i < annots.getArrayNItems(); ++i) {
697
+ auto annot = annots.getArrayItem(i);
698
+ if (!annot.isDictionary())
699
+ continue;
700
+ if (annot.hasKey("/AA"))
701
+ annot.removeKey("/AA");
702
+ if (annot.hasKey("/A")) {
703
+ auto action = annot.getKey("/A");
704
+ if (action.isDictionary()) {
705
+ auto s = action.getKey("/S");
706
+ if (s.isName() && s.getName() == "/JavaScript")
707
+ annot.removeKey("/A");
708
+ }
709
+ }
710
+ }
711
+ }
712
+ }
713
+
714
+ // ---------------------------------------------------------------------------
715
+ // Form flattening — merge interactive form field appearances into page
716
+ // content and remove the /AcroForm dictionary
717
+ // ---------------------------------------------------------------------------
718
+
719
+ void flattenForms(QPDF &qpdf) {
720
+ auto root = qpdf.getRoot();
721
+ if (!root.hasKey("/AcroForm"))
722
+ return;
723
+
724
+ // stamp each widget annotation's appearance into the page content,
725
+ // then remove the annotation
726
+ for (auto &page : QPDFPageDocumentHelper(qpdf).getAllPages()) {
727
+ auto pageObj = page.getObjectHandle();
728
+ if (!pageObj.hasKey("/Annots"))
729
+ continue;
730
+
731
+ auto annots = pageObj.getKey("/Annots");
732
+ if (!annots.isArray())
733
+ continue;
734
+
735
+ std::vector<int> widgetIndices;
736
+ for (int i = 0; i < annots.getArrayNItems(); ++i) {
737
+ auto annot = annots.getArrayItem(i);
738
+ if (!annot.isDictionary())
739
+ continue;
740
+
741
+ auto subtype = annot.getKey("/Subtype");
742
+ if (!subtype.isName() || subtype.getName() != "/Widget")
743
+ continue;
744
+
745
+ // check if there's a normal appearance to flatten
746
+ auto ap = annot.getKey("/AP");
747
+ if (!ap.isDictionary())
748
+ continue;
749
+ auto nAppearance = ap.getKey("/N");
750
+ if (!nAppearance.isStream())
751
+ continue;
752
+
753
+ // get widget rectangle
754
+ auto rect = annot.getKey("/Rect");
755
+ if (!rect.isArray() || rect.getArrayNItems() < 4)
756
+ continue;
757
+
758
+ try {
759
+ double x1 = rect.getArrayItem(0).getNumericValue();
760
+ double y1 = rect.getArrayItem(1).getNumericValue();
761
+ double x2 = rect.getArrayItem(2).getNumericValue();
762
+ double y2 = rect.getArrayItem(3).getNumericValue();
763
+
764
+ double w = x2 - x1;
765
+ double h = y2 - y1;
766
+ if (w <= 0 || h <= 0)
767
+ continue;
768
+
769
+ // get appearance stream bounding box for scaling
770
+ auto apDict = nAppearance.getDict();
771
+ double scaleX = 1.0, scaleY = 1.0;
772
+ if (apDict.hasKey("/BBox")) {
773
+ auto bbox = apDict.getKey("/BBox");
774
+ if (bbox.isArray() && bbox.getArrayNItems() >= 4) {
775
+ double bw = bbox.getArrayItem(2).getNumericValue() -
776
+ bbox.getArrayItem(0).getNumericValue();
777
+ double bh = bbox.getArrayItem(3).getNumericValue() -
778
+ bbox.getArrayItem(1).getNumericValue();
779
+ if (bw > 0)
780
+ scaleX = w / bw;
781
+ if (bh > 0)
782
+ scaleY = h / bh;
783
+ }
784
+ }
785
+
786
+ // register appearance as a form XObject on the page
787
+ auto resources = pageObj.getKey("/Resources");
788
+ if (!resources.isDictionary()) {
789
+ resources = QPDFObjectHandle::newDictionary();
790
+ pageObj.replaceKey("/Resources", resources);
791
+ }
792
+ auto xobjects = resources.getKey("/XObject");
793
+ if (!xobjects.isDictionary()) {
794
+ xobjects = QPDFObjectHandle::newDictionary();
795
+ resources.replaceKey("/XObject", xobjects);
796
+ }
797
+
798
+ std::string xobjName = "/FlatForm" + std::to_string(i);
799
+ xobjects.replaceKey(xobjName, nAppearance);
800
+
801
+ // ensure the appearance stream has /Type /XObject /Subtype /Form
802
+ if (!apDict.hasKey("/Type"))
803
+ apDict.replaceKey("/Type", QPDFObjectHandle::newName("/XObject"));
804
+ if (!apDict.hasKey("/Subtype"))
805
+ apDict.replaceKey("/Subtype", QPDFObjectHandle::newName("/Form"));
806
+
807
+ // build content stream snippet to stamp the appearance
808
+ std::string snippet = "q " + std::to_string(scaleX) + " 0 0 " +
809
+ std::to_string(scaleY) + " " +
810
+ std::to_string(x1) + " " + std::to_string(y1) +
811
+ " cm " + xobjName + " Do Q\n";
812
+
813
+ // append to page content stream
814
+ page.addPageContents(QPDFObjectHandle::newStream(&qpdf, snippet),
815
+ false);
816
+
817
+ widgetIndices.push_back(i);
818
+ } catch (...) {
819
+ continue;
820
+ }
821
+ }
822
+
823
+ // remove widget annotations (reverse order to preserve indices)
824
+ for (auto it = widgetIndices.rbegin(); it != widgetIndices.rend(); ++it)
825
+ annots.eraseItem(*it);
826
+ }
827
+
828
+ // remove the /AcroForm dictionary
829
+ root.removeKey("/AcroForm");
830
+ }
831
+
832
+ // ---------------------------------------------------------------------------
833
+ // Page tree flattening — push inherited attributes to pages so QPDFWriter
834
+ // can generate a flat single-level page tree
835
+ // ---------------------------------------------------------------------------
836
+
837
+ void flattenPageTree(QPDF &qpdf) { qpdf.pushInheritedAttributesToPage(); }
838
+
839
+ // ---------------------------------------------------------------------------
840
+ // Content stream minification — normalize whitespace and number formatting
841
+ // to reduce content stream size before Flate compression
842
+ // ---------------------------------------------------------------------------
843
+
844
+ // trims a numeric string: remove trailing zeros after decimal point,
845
+ // remove the decimal point if it becomes the last char,
846
+ // and strip a leading zero for values between -1 and 1.
847
+ static std::string trimNumber(const std::string &s) {
848
+ // only process strings that look like decimal numbers
849
+ if (s.find('.') == std::string::npos)
850
+ return s;
851
+
852
+ std::string result = s;
853
+
854
+ // strip trailing zeros after decimal point
855
+ size_t dot = result.find('.');
856
+ if (dot != std::string::npos) {
857
+ size_t last = result.size() - 1;
858
+ while (last > dot && result[last] == '0')
859
+ --last;
860
+ if (last == dot)
861
+ result.erase(dot); // remove the dot too (e.g. "1." → "1")
862
+ else
863
+ result.erase(last + 1);
864
+ }
865
+
866
+ // strip leading zero for values like "0.5" → ".5" or "-0.5" → "-.5"
867
+ if (result.size() >= 2 && result[0] == '0' && result[1] == '.')
868
+ result.erase(0, 1);
869
+ else if (result.size() >= 3 && result[0] == '-' && result[1] == '0' &&
870
+ result[2] == '.')
871
+ result.erase(1, 1);
872
+
873
+ return result;
874
+ }
875
+
876
+ void minifyContentStreams(QPDF &qpdf) {
877
+ for (auto &page : QPDFPageDocumentHelper(qpdf).getAllPages()) {
878
+ auto pageObj = page.getObjectHandle();
879
+ auto contents = pageObj.getKey("/Contents");
880
+
881
+ if (!contents.isStream())
882
+ continue;
883
+
884
+ std::string raw;
885
+ try {
886
+ auto buf = contents.getStreamData(qpdf_dl_generalized);
887
+ raw.assign(reinterpret_cast<const char *>(buf->getBuffer()),
888
+ buf->getSize());
889
+ } catch (...) {
890
+ continue;
891
+ }
892
+
893
+ // tokenize preserving string literals and hex strings intact
894
+ std::string minified;
895
+ minified.reserve(raw.size());
896
+ bool needSpace = false;
897
+ size_t pos = 0;
898
+
899
+ while (pos < raw.size()) {
900
+ char ch = raw[pos];
901
+
902
+ // skip whitespace
903
+ if (ch == ' ' || ch == '\t' || ch == '\r' || ch == '\n') {
904
+ if (!minified.empty())
905
+ needSpace = true;
906
+ ++pos;
907
+ continue;
908
+ }
909
+
910
+ // comments — skip to end of line
911
+ if (ch == '%') {
912
+ while (pos < raw.size() && raw[pos] != '\n')
913
+ ++pos;
914
+ continue;
915
+ }
916
+
917
+ // literal string — copy verbatim
918
+ if (ch == '(') {
919
+ if (needSpace) {
920
+ minified += '\n';
921
+ needSpace = false;
922
+ }
923
+ int depth = 1;
924
+ minified += '(';
925
+ ++pos;
926
+ while (pos < raw.size() && depth > 0) {
927
+ if (raw[pos] == '\\') {
928
+ minified += raw[pos++];
929
+ if (pos < raw.size())
930
+ minified += raw[pos++];
931
+ } else {
932
+ if (raw[pos] == '(')
933
+ ++depth;
934
+ else if (raw[pos] == ')')
935
+ --depth;
936
+ minified += raw[pos++];
937
+ }
938
+ }
939
+ needSpace = true;
940
+ continue;
941
+ }
942
+
943
+ // hex string — copy verbatim
944
+ if (ch == '<' && pos + 1 < raw.size() && raw[pos + 1] != '<') {
945
+ if (needSpace) {
946
+ minified += '\n';
947
+ needSpace = false;
948
+ }
949
+ minified += '<';
950
+ ++pos;
951
+ while (pos < raw.size() && raw[pos] != '>') {
952
+ if (!std::isspace(static_cast<unsigned char>(raw[pos])))
953
+ minified += raw[pos];
954
+ ++pos;
955
+ }
956
+ if (pos < raw.size()) {
957
+ minified += '>';
958
+ ++pos;
959
+ }
960
+ needSpace = true;
961
+ continue;
962
+ }
963
+
964
+ // dict delimiters << >> — self-delimiting, no space needed around them
965
+ if (ch == '<' && pos + 1 < raw.size() && raw[pos + 1] == '<') {
966
+ if (needSpace) {
967
+ minified += '\n';
968
+ needSpace = false;
969
+ }
970
+ minified += "<<";
971
+ pos += 2;
972
+ continue;
973
+ }
974
+ if (ch == '>' && pos + 1 < raw.size() && raw[pos + 1] == '>') {
975
+ minified += ">>";
976
+ pos += 2;
977
+ needSpace = true;
978
+ continue;
979
+ }
980
+
981
+ // array delimiters — self-delimiting
982
+ if (ch == '[' || ch == ']') {
983
+ if (needSpace && ch == '[') {
984
+ minified += '\n';
985
+ needSpace = false;
986
+ }
987
+ minified += ch;
988
+ ++pos;
989
+ if (ch == ']')
990
+ needSpace = true;
991
+ continue;
992
+ }
993
+
994
+ // name — starts with /
995
+ if (ch == '/') {
996
+ if (needSpace) {
997
+ minified += '\n';
998
+ needSpace = false;
999
+ }
1000
+ size_t start = pos;
1001
+ ++pos;
1002
+ while (pos < raw.size() &&
1003
+ !std::isspace(static_cast<unsigned char>(raw[pos])) &&
1004
+ raw[pos] != '/' && raw[pos] != '[' && raw[pos] != ']' &&
1005
+ raw[pos] != '<' && raw[pos] != '>' && raw[pos] != '(' &&
1006
+ raw[pos] != ')')
1007
+ ++pos;
1008
+ minified.append(raw, start, pos - start);
1009
+ needSpace = true;
1010
+ continue;
1011
+ }
1012
+
1013
+ // regular token (number, operator)
1014
+ {
1015
+ if (needSpace) {
1016
+ minified += '\n';
1017
+ needSpace = false;
1018
+ }
1019
+ size_t start = pos;
1020
+ while (pos < raw.size() &&
1021
+ !std::isspace(static_cast<unsigned char>(raw[pos])) &&
1022
+ raw[pos] != '/' && raw[pos] != '[' && raw[pos] != ']' &&
1023
+ raw[pos] != '<' && raw[pos] != '>' && raw[pos] != '(' &&
1024
+ raw[pos] != ')')
1025
+ ++pos;
1026
+
1027
+ std::string token(raw, start, pos - start);
1028
+
1029
+ // trim numeric formatting
1030
+ if (!token.empty() &&
1031
+ (token[0] == '-' || token[0] == '+' || token[0] == '.' ||
1032
+ (token[0] >= '0' && token[0] <= '9'))) {
1033
+ token = trimNumber(token);
1034
+ }
1035
+
1036
+ minified += token;
1037
+ needSpace = true;
1038
+ }
1039
+ }
1040
+
1041
+ // only replace if we actually reduced the size
1042
+ if (minified.size() >= raw.size())
1043
+ continue;
1044
+
1045
+ contents.replaceStreamData(minified, QPDFObjectHandle::newNull(),
1046
+ QPDFObjectHandle::newNull());
1047
+ }
1048
+ }