qpdf-compress 0.2.0 → 0.4.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/images.cc CHANGED
@@ -2,6 +2,7 @@
2
2
  #include "jpeg.h"
3
3
 
4
4
  #include <algorithm>
5
+ #include <cmath>
5
6
  #include <cstdint>
6
7
  #include <cstring>
7
8
  #include <limits>
@@ -72,6 +73,198 @@ static void cmykToRgb(const unsigned char *cmyk, unsigned char *rgb,
72
73
  }
73
74
  }
74
75
 
76
+ // ---------------------------------------------------------------------------
77
+ // Content stream CTM parser — find rendered image dimensions in points
78
+ // ---------------------------------------------------------------------------
79
+
80
+ // 3x3 affine matrix stored as [a b c d e f] (PDF standard order)
81
+ struct Matrix {
82
+ double a = 1, b = 0, c = 0, d = 1, e = 0, f = 0;
83
+ };
84
+
85
+ static Matrix multiply(const Matrix &m1, const Matrix &m2) {
86
+ return {m1.a * m2.a + m1.b * m2.c, m1.a * m2.b + m1.b * m2.d,
87
+ m1.c * m2.a + m1.d * m2.c, m1.c * m2.b + m1.d * m2.d,
88
+ m1.e * m2.a + m1.f * m2.c + m2.e, m1.e * m2.b + m1.f * m2.d + m2.f};
89
+ }
90
+
91
+ // returns the rendered width and height in points for each image XObject name.
92
+ // falls back to pageW/pageH if parsing fails or the image isn't found.
93
+ static std::map<std::string, std::pair<double, double>>
94
+ getImageRenderedSizes(QPDFPageObjectHelper &page) {
95
+ std::map<std::string, std::pair<double, double>> result;
96
+
97
+ auto pageObj = page.getObjectHandle();
98
+ auto contents = pageObj.getKey("/Contents");
99
+
100
+ std::string contentStr;
101
+ try {
102
+ if (contents.isStream()) {
103
+ auto buf = contents.getStreamData(qpdf_dl_generalized);
104
+ contentStr.assign(reinterpret_cast<const char *>(buf->getBuffer()),
105
+ buf->getSize());
106
+ } else if (contents.isArray()) {
107
+ for (int i = 0; i < contents.getArrayNItems(); ++i) {
108
+ auto stream = contents.getArrayItem(i);
109
+ if (stream.isStream()) {
110
+ auto buf = stream.getStreamData(qpdf_dl_generalized);
111
+ contentStr.append(reinterpret_cast<const char *>(buf->getBuffer()),
112
+ buf->getSize());
113
+ contentStr += '\n';
114
+ }
115
+ }
116
+ }
117
+ } catch (...) {
118
+ return result;
119
+ }
120
+
121
+ // simple tokenizer: split on whitespace, handle names (/Xxx)
122
+ std::vector<std::string> tokens;
123
+ size_t pos = 0;
124
+ while (pos < contentStr.size()) {
125
+ // skip whitespace
126
+ while (pos < contentStr.size() &&
127
+ (contentStr[pos] == ' ' || contentStr[pos] == '\n' ||
128
+ contentStr[pos] == '\r' || contentStr[pos] == '\t'))
129
+ ++pos;
130
+ if (pos >= contentStr.size())
131
+ break;
132
+
133
+ // skip comments
134
+ if (contentStr[pos] == '%') {
135
+ while (pos < contentStr.size() && contentStr[pos] != '\n')
136
+ ++pos;
137
+ continue;
138
+ }
139
+
140
+ // skip strings (we don't need them)
141
+ if (contentStr[pos] == '(') {
142
+ int depth = 1;
143
+ ++pos;
144
+ while (pos < contentStr.size() && depth > 0) {
145
+ if (contentStr[pos] == '\\') {
146
+ ++pos;
147
+ if (pos < contentStr.size())
148
+ ++pos;
149
+ } else {
150
+ if (contentStr[pos] == '(')
151
+ ++depth;
152
+ else if (contentStr[pos] == ')')
153
+ --depth;
154
+ ++pos;
155
+ }
156
+ }
157
+ continue;
158
+ }
159
+ if (contentStr[pos] == '<' && pos + 1 < contentStr.size() &&
160
+ contentStr[pos + 1] != '<') {
161
+ ++pos;
162
+ while (pos < contentStr.size() && contentStr[pos] != '>')
163
+ ++pos;
164
+ if (pos < contentStr.size())
165
+ ++pos;
166
+ continue;
167
+ }
168
+
169
+ // skip inline images (BI ... EI)
170
+ // handled after tokenization
171
+
172
+ // read token
173
+ size_t start = pos;
174
+ if (contentStr[pos] == '/') {
175
+ ++pos;
176
+ while (pos < contentStr.size() && contentStr[pos] != ' ' &&
177
+ contentStr[pos] != '\n' && contentStr[pos] != '\r' &&
178
+ contentStr[pos] != '\t' && contentStr[pos] != '/' &&
179
+ contentStr[pos] != '[' && contentStr[pos] != ']' &&
180
+ contentStr[pos] != '<' && contentStr[pos] != '>' &&
181
+ contentStr[pos] != '(' && contentStr[pos] != ')')
182
+ ++pos;
183
+ } else if (contentStr[pos] == '[' || contentStr[pos] == ']') {
184
+ ++pos;
185
+ } else if (contentStr[pos] == '<' && pos + 1 < contentStr.size() &&
186
+ contentStr[pos + 1] == '<') {
187
+ pos += 2; // <<
188
+ } else if (contentStr[pos] == '>' && pos + 1 < contentStr.size() &&
189
+ contentStr[pos + 1] == '>') {
190
+ pos += 2; // >>
191
+ } else {
192
+ while (pos < contentStr.size() && contentStr[pos] != ' ' &&
193
+ contentStr[pos] != '\n' && contentStr[pos] != '\r' &&
194
+ contentStr[pos] != '\t' && contentStr[pos] != '/' &&
195
+ contentStr[pos] != '[' && contentStr[pos] != ']' &&
196
+ contentStr[pos] != '<' && contentStr[pos] != '>' &&
197
+ contentStr[pos] != '(' && contentStr[pos] != ')')
198
+ ++pos;
199
+ }
200
+
201
+ if (pos > start)
202
+ tokens.emplace_back(contentStr.substr(start, pos - start));
203
+ }
204
+
205
+ // walk tokens tracking CTM
206
+ Matrix ctm;
207
+ std::vector<Matrix> stack;
208
+ std::vector<std::string> operandStack;
209
+
210
+ for (size_t i = 0; i < tokens.size(); ++i) {
211
+ auto &tok = tokens[i];
212
+
213
+ if (tok == "q") {
214
+ stack.push_back(ctm);
215
+ operandStack.clear();
216
+ } else if (tok == "Q") {
217
+ if (!stack.empty()) {
218
+ ctm = stack.back();
219
+ stack.pop_back();
220
+ }
221
+ operandStack.clear();
222
+ } else if (tok == "cm" && operandStack.size() >= 6) {
223
+ // cm operator: a b c d e f cm
224
+ try {
225
+ Matrix m;
226
+ m.a = std::stod(operandStack[operandStack.size() - 6]);
227
+ m.b = std::stod(operandStack[operandStack.size() - 5]);
228
+ m.c = std::stod(operandStack[operandStack.size() - 4]);
229
+ m.d = std::stod(operandStack[operandStack.size() - 3]);
230
+ m.e = std::stod(operandStack[operandStack.size() - 2]);
231
+ m.f = std::stod(operandStack[operandStack.size() - 1]);
232
+ ctm = multiply(m, ctm);
233
+ } catch (...) {
234
+ }
235
+ operandStack.clear();
236
+ } else if (tok == "Do" && !operandStack.empty()) {
237
+ // Do operator draws an XObject
238
+ auto &name = operandStack.back();
239
+ if (!name.empty() && name[0] == '/') {
240
+ // rendered width = sqrt(a^2 + c^2), height = sqrt(b^2 + d^2)
241
+ // (these are the lengths of the column vectors of the CTM)
242
+ double rw = std::sqrt(ctm.a * ctm.a + ctm.c * ctm.c);
243
+ double rh = std::sqrt(ctm.b * ctm.b + ctm.d * ctm.d);
244
+ // keep the largest rendered size if an image is drawn multiple times
245
+ auto it = result.find(name);
246
+ if (it == result.end() ||
247
+ rw * rh > it->second.first * it->second.second)
248
+ result[name] = {rw, rh};
249
+ }
250
+ operandStack.clear();
251
+ } else {
252
+ // it's an operand — check if it's an operator we don't track
253
+ // (PDF operators are always non-numeric alpha)
254
+ bool isOperator = !tok.empty() && tok[0] != '/' && tok[0] != '-' &&
255
+ tok[0] != '+' && tok[0] != '.' &&
256
+ !(tok[0] >= '0' && tok[0] <= '9');
257
+ if (isOperator) {
258
+ operandStack.clear();
259
+ } else {
260
+ operandStack.push_back(tok);
261
+ }
262
+ }
263
+ }
264
+
265
+ return result;
266
+ }
267
+
75
268
  // ---------------------------------------------------------------------------
76
269
  // Bilinear downscaling
77
270
  // ---------------------------------------------------------------------------
@@ -118,15 +311,11 @@ static std::vector<uint8_t> bilinearDownscale(const unsigned char *src,
118
311
  // Image recompression for lossy mode
119
312
  // ---------------------------------------------------------------------------
120
313
 
121
- // auto-mode thresholds — only re-encode existing JPEGs whose estimated
122
- // quality exceeds kAutoSkipThreshold (avoids pointless re-encoding where
123
- // generation loss outweighs size savings). Non-JPEG images and high-quality
124
- // JPEGs are (re-)encoded at kAutoTargetQuality.
125
- static constexpr int kAutoSkipThreshold = 90;
126
- static constexpr int kAutoTargetQuality = 85;
314
+ // Image recompression thresholds are passed via CompressOptions:
315
+ // - lossless mode: skipThreshold=90, targetQuality=85 (conservative)
316
+ // - lossy mode: skipThreshold=65, targetQuality=75 (aggressive)
127
317
 
128
318
  void optimizeImages(QPDF &qpdf, const CompressOptions &opts) {
129
- const bool autoQuality = (opts.quality == 0);
130
319
  forEachImage(qpdf, [&](const std::string &, QPDFObjectHandle xobj,
131
320
  QPDFObjectHandle, QPDFPageObjectHelper &) {
132
321
  auto dict = xobj.getDict();
@@ -192,39 +381,26 @@ void optimizeImages(QPDF &qpdf, const CompressOptions &opts) {
192
381
  bool isCurrentlyJpeg =
193
382
  currentFilter.isName() && currentFilter.getName() == "/DCTDecode";
194
383
 
195
- // determine per-image target quality
196
- int targetQuality = autoQuality ? kAutoTargetQuality : opts.quality;
197
-
198
- // in auto mode, skip existing JPEGs unless their quality is very high
199
- // (> 90) — re-encoding a q86 JPEG at q85 saves almost nothing but adds
200
- // artifacts. Only high-quality originals (92, 95, 100…) benefit from
201
- // re-encoding down to 85.
384
+ // skip existing JPEGs that are already at or below the threshold —
385
+ // re-encoding them adds generation loss for negligible savings
202
386
  if (isCurrentlyJpeg && !isCMYK) {
203
387
  auto rawData = xobj.getRawStreamData();
204
388
  int existingQ =
205
389
  estimateJpegQuality(rawData->getBuffer(), rawData->getSize());
206
- if (autoQuality) {
207
- if (existingQ > 0 && existingQ <= kAutoSkipThreshold)
208
- return;
209
- } else {
210
- // explicit quality: use existing ceiling logic
211
- if (existingQ > 0 && existingQ <= targetQuality)
212
- return;
213
- }
390
+ if (existingQ > 0 && existingQ <= opts.skipThreshold)
391
+ return;
214
392
  }
215
393
 
216
394
  // encode as JPEG via libjpeg-turbo
217
395
  std::vector<uint8_t> jpegData;
218
- if (!encodeJpeg(pixels, width, height, encodeComponents, targetQuality,
396
+ if (!encodeJpeg(pixels, width, height, encodeComponents, opts.targetQuality,
219
397
  jpegData))
220
398
  return;
221
399
 
222
- // only replace if we actually reduced size (for non-CMYK images)
223
- if (isCurrentlyJpeg && !isCMYK) {
224
- auto rawData = xobj.getRawStreamData();
225
- if (jpegData.size() >= rawData->getSize())
226
- return;
227
- }
400
+ // only replace if we actually reduced size
401
+ auto rawData = xobj.getRawStreamData();
402
+ if (jpegData.size() >= rawData->getSize())
403
+ return;
228
404
 
229
405
  // replace stream data with JPEG
230
406
  std::string jpegStr(reinterpret_cast<char *>(jpegData.data()),
@@ -373,13 +549,16 @@ void optimizeExistingJpegs(QPDF &qpdf) {
373
549
  // DPI-based image downscaling
374
550
  // ---------------------------------------------------------------------------
375
551
 
376
- void downscaleImages(QPDF &qpdf, int maxDpi) {
552
+ void downscaleImages(QPDF &qpdf, int maxDpi, int quality) {
377
553
  if (maxDpi <= 0)
378
554
  return;
379
555
 
380
556
  std::set<QPDFObjGen> processed;
557
+ // cache rendered sizes per page object (keyed by page objgen)
558
+ std::map<QPDFObjGen, std::map<std::string, std::pair<double, double>>>
559
+ pageSizesCache;
381
560
 
382
- forEachImage(qpdf, [&](const std::string & /*key*/, QPDFObjectHandle xobj,
561
+ forEachImage(qpdf, [&](const std::string &key, QPDFObjectHandle xobj,
383
562
  QPDFObjectHandle /*xobjects*/,
384
563
  QPDFPageObjectHelper &page) {
385
564
  auto og = xobj.getObjGen();
@@ -407,27 +586,42 @@ void downscaleImages(QPDF &qpdf, int maxDpi) {
407
586
  if (components == 0)
408
587
  return;
409
588
 
410
- // get page dimensions from MediaBox (in points, 72 per inch)
411
- auto mediaBox = page.getAttribute("/MediaBox", false);
412
- if (!mediaBox.isArray() || mediaBox.getArrayNItems() < 4)
413
- return;
589
+ // get rendered size from CTM parsing, with MediaBox fallback
590
+ auto pageOg = page.getObjectHandle().getObjGen();
591
+ auto cacheIt = pageSizesCache.find(pageOg);
592
+ if (cacheIt == pageSizesCache.end()) {
593
+ pageSizesCache[pageOg] = getImageRenderedSizes(page);
594
+ cacheIt = pageSizesCache.find(pageOg);
595
+ }
414
596
 
415
- double pageW = 0, pageH = 0;
416
- try {
417
- pageW = mediaBox.getArrayItem(2).getNumericValue() -
418
- mediaBox.getArrayItem(0).getNumericValue();
419
- pageH = mediaBox.getArrayItem(3).getNumericValue() -
420
- mediaBox.getArrayItem(1).getNumericValue();
421
- } catch (...) {
422
- return;
597
+ double renderedW = 0, renderedH = 0;
598
+ auto sizeIt = cacheIt->second.find(key);
599
+ if (sizeIt != cacheIt->second.end() && sizeIt->second.first > 0 &&
600
+ sizeIt->second.second > 0) {
601
+ // CTM gives rendered size in points
602
+ renderedW = sizeIt->second.first;
603
+ renderedH = sizeIt->second.second;
604
+ } else {
605
+ // fallback: assume image fills page
606
+ auto mediaBox = page.getAttribute("/MediaBox", false);
607
+ if (!mediaBox.isArray() || mediaBox.getArrayNItems() < 4)
608
+ return;
609
+ try {
610
+ renderedW = mediaBox.getArrayItem(2).getNumericValue() -
611
+ mediaBox.getArrayItem(0).getNumericValue();
612
+ renderedH = mediaBox.getArrayItem(3).getNumericValue() -
613
+ mediaBox.getArrayItem(1).getNumericValue();
614
+ } catch (...) {
615
+ return;
616
+ }
423
617
  }
424
618
 
425
- if (pageW <= 0 || pageH <= 0)
619
+ if (renderedW <= 0 || renderedH <= 0)
426
620
  return;
427
621
 
428
- // estimate effective DPI (assumes image fills page — conservative)
429
- double dpiX = imgW / (pageW / 72.0);
430
- double dpiY = imgH / (pageH / 72.0);
622
+ // calculate effective DPI from rendered size in points (72 points/inch)
623
+ double dpiX = imgW / (renderedW / 72.0);
624
+ double dpiY = imgH / (renderedH / 72.0);
431
625
  double effectiveDpi = std::max(dpiX, dpiY);
432
626
 
433
627
  if (effectiveDpi <= maxDpi)
@@ -476,13 +670,20 @@ void downscaleImages(QPDF &qpdf, int maxDpi) {
476
670
  auto scaled =
477
671
  bilinearDownscale(pixels, imgW, imgH, downscaleComponents, newW, newH);
478
672
 
479
- // re-encode as Flate-compressed raw pixels
480
- auto newSize = static_cast<size_t>(newW) * newH * downscaleComponents;
481
- if (scaled.size() != newSize)
673
+ // re-encode downscaled pixels as JPEG
674
+ std::vector<uint8_t> jpegData;
675
+ if (!encodeJpeg(scaled.data(), newW, newH, downscaleComponents, quality,
676
+ jpegData))
482
677
  return;
483
678
 
484
- std::string rawStr(reinterpret_cast<char *>(scaled.data()), scaled.size());
485
- xobj.replaceStreamData(rawStr, QPDFObjectHandle::newName("/FlateDecode"),
679
+ // only replace if the JPEG is actually smaller than the original stream
680
+ auto rawData = xobj.getRawStreamData();
681
+ if (jpegData.size() >= rawData->getSize())
682
+ return;
683
+
684
+ std::string jpegStr(reinterpret_cast<char *>(jpegData.data()),
685
+ jpegData.size());
686
+ xobj.replaceStreamData(jpegStr, QPDFObjectHandle::newName("/DCTDecode"),
486
687
  QPDFObjectHandle::newNull());
487
688
 
488
689
  dict.replaceKey("/Width", QPDFObjectHandle::newInteger(newW));
@@ -501,96 +702,242 @@ void downscaleImages(QPDF &qpdf, int maxDpi) {
501
702
  }
502
703
 
503
704
  // ---------------------------------------------------------------------------
504
- // Metadata stripping
705
+ // Grayscale detection — convert RGB images that are actually grayscale to
706
+ // DeviceGray (1/3 the raw data size)
505
707
  // ---------------------------------------------------------------------------
506
708
 
507
- void stripMetadata(QPDF &qpdf) {
508
- auto root = qpdf.getRoot();
509
-
510
- // remove XMP metadata stream
511
- if (root.hasKey("/Metadata"))
512
- root.removeKey("/Metadata");
513
-
514
- // remove document info dictionary
515
- auto trailer = qpdf.getTrailer();
516
- if (trailer.hasKey("/Info"))
517
- trailer.removeKey("/Info");
518
-
519
- // remove page-level metadata and PieceInfo
520
- for (auto &page : QPDFPageDocumentHelper(qpdf).getAllPages()) {
521
- auto pageObj = page.getObjectHandle();
522
- if (pageObj.hasKey("/Metadata"))
523
- pageObj.removeKey("/Metadata");
524
- if (pageObj.hasKey("/PieceInfo"))
525
- pageObj.removeKey("/PieceInfo");
526
- }
709
+ void convertGrayscaleImages(QPDF &qpdf) {
710
+ std::set<QPDFObjGen> processed;
527
711
 
528
- // remove embedded thumbnails
529
- for (auto &page : QPDFPageDocumentHelper(qpdf).getAllPages()) {
530
- auto pageObj = page.getObjectHandle();
531
- if (pageObj.hasKey("/Thumb"))
532
- pageObj.removeKey("/Thumb");
533
- }
712
+ forEachImage(qpdf, [&](const std::string & /*key*/, QPDFObjectHandle xobj,
713
+ QPDFObjectHandle /*xobjects*/,
714
+ QPDFPageObjectHelper & /*page*/) {
715
+ auto og = xobj.getObjGen();
716
+ if (processed.count(og))
717
+ return;
718
+ processed.insert(og);
534
719
 
535
- // remove MarkInfo and page labels (optional metadata)
536
- if (root.hasKey("/MarkInfo"))
537
- root.removeKey("/MarkInfo");
720
+ auto dict = xobj.getDict();
721
+
722
+ // only handle 8-bit RGB images
723
+ if (!dict.getKey("/BitsPerComponent").isInteger() ||
724
+ dict.getKey("/BitsPerComponent").getIntValue() != 8)
725
+ return;
726
+
727
+ auto cs = dict.getKey("/ColorSpace");
728
+ if (!cs.isName() || cs.getName() != "/DeviceRGB")
729
+ return;
730
+
731
+ // skip JPEG-compressed images — converting to raw gray + Flate would be
732
+ // larger than the original JPEG. in lossy mode, optimizeImages downstream
733
+ // re-encodes, but in lossless mode nothing would, causing size inflation.
734
+ auto filter = dict.getKey("/Filter");
735
+ if (filter.isName() && filter.getName() == "/DCTDecode")
736
+ return;
737
+
738
+ int width = 0, height = 0;
739
+ if (dict.getKey("/Width").isInteger())
740
+ width = static_cast<int>(dict.getKey("/Width").getIntValue());
741
+ if (dict.getKey("/Height").isInteger())
742
+ height = static_cast<int>(dict.getKey("/Height").getIntValue());
743
+ if (width <= 0 || height <= 0)
744
+ return;
745
+
746
+ // decode pixels
747
+ std::shared_ptr<Buffer> streamData;
748
+ try {
749
+ streamData = xobj.getStreamData(qpdf_dl_all);
750
+ } catch (...) {
751
+ return;
752
+ }
753
+
754
+ auto w = static_cast<size_t>(width);
755
+ auto h = static_cast<size_t>(height);
756
+ size_t expectedSize = w * h * 3;
757
+ if (streamData->getSize() != expectedSize)
758
+ return;
759
+
760
+ const auto *pixels = streamData->getBuffer();
761
+ size_t pixelCount = w * h;
762
+
763
+ // check if all RGB triples have equal channels (R == G == B)
764
+ bool isGray = true;
765
+ for (size_t i = 0; i < pixelCount; ++i) {
766
+ auto r = pixels[i * 3 + 0];
767
+ auto g = pixels[i * 3 + 1];
768
+ auto b = pixels[i * 3 + 2];
769
+ if (r != g || g != b) {
770
+ isGray = false;
771
+ break;
772
+ }
773
+ }
774
+
775
+ if (!isGray)
776
+ return;
777
+
778
+ // build grayscale pixel buffer
779
+ std::vector<uint8_t> grayPixels(pixelCount);
780
+ for (size_t i = 0; i < pixelCount; ++i)
781
+ grayPixels[i] = pixels[i * 3];
782
+
783
+ // replace stream with raw grayscale data (Flate-compressed by QPDFWriter)
784
+ std::string grayStr(reinterpret_cast<char *>(grayPixels.data()),
785
+ grayPixels.size());
786
+ xobj.replaceStreamData(grayStr, QPDFObjectHandle::newNull(),
787
+ QPDFObjectHandle::newNull());
788
+ dict.replaceKey("/ColorSpace", QPDFObjectHandle::newName("/DeviceGray"));
789
+
790
+ // remove JPEG-specific params
791
+ if (dict.hasKey("/DecodeParms"))
792
+ dict.removeKey("/DecodeParms");
793
+ if (dict.hasKey("/Predictor"))
794
+ dict.removeKey("/Predictor");
795
+ });
538
796
  }
539
797
 
540
798
  // ---------------------------------------------------------------------------
541
- // Remove unused font resources
799
+ // Bitonal conversion — convert 8-bit grayscale images that are effectively
800
+ // black & white (all pixels near 0 or 255) to 1-bit, saving ~8x raw data
542
801
  // ---------------------------------------------------------------------------
543
802
 
544
- void removeUnusedFonts(QPDF &qpdf) {
545
- for (auto &page : QPDFPageDocumentHelper(qpdf).getAllPages()) {
546
- auto pageObj = page.getObjectHandle();
547
- auto resources = pageObj.getKey("/Resources");
548
- if (!resources.isDictionary())
549
- continue;
550
- auto fonts = resources.getKey("/Font");
551
- if (!fonts.isDictionary())
552
- continue;
803
+ void convertBitonalImages(QPDF &qpdf) {
804
+ std::set<QPDFObjGen> processed;
553
805
 
554
- // collect all font names referenced in this page's content stream(s)
555
- std::set<std::string> usedFonts;
806
+ forEachImage(qpdf, [&](const std::string & /*key*/, QPDFObjectHandle xobj,
807
+ QPDFObjectHandle /*xobjects*/,
808
+ QPDFPageObjectHelper & /*page*/) {
809
+ auto og = xobj.getObjGen();
810
+ if (processed.count(og))
811
+ return;
812
+ processed.insert(og);
813
+
814
+ auto dict = xobj.getDict();
556
815
 
816
+ // only handle 8-bit grayscale images
817
+ if (!dict.getKey("/BitsPerComponent").isInteger() ||
818
+ dict.getKey("/BitsPerComponent").getIntValue() != 8)
819
+ return;
820
+
821
+ auto cs = dict.getKey("/ColorSpace");
822
+ if (!cs.isName() || cs.getName() != "/DeviceGray")
823
+ return;
824
+
825
+ // skip images with masks (bitonal conversion may not be safe)
826
+ if (dict.hasKey("/SMask") || dict.hasKey("/Mask"))
827
+ return;
828
+
829
+ int width = 0, height = 0;
830
+ if (dict.getKey("/Width").isInteger())
831
+ width = static_cast<int>(dict.getKey("/Width").getIntValue());
832
+ if (dict.getKey("/Height").isInteger())
833
+ height = static_cast<int>(dict.getKey("/Height").getIntValue());
834
+ if (width <= 0 || height <= 0)
835
+ return;
836
+
837
+ // decode pixels
838
+ std::shared_ptr<Buffer> streamData;
557
839
  try {
558
- // get unparsed content stream data
559
- auto contents = pageObj.getKey("/Contents");
560
- std::string contentStr;
561
-
562
- if (contents.isStream()) {
563
- auto buf = contents.getStreamData(qpdf_dl_generalized);
564
- contentStr.assign(reinterpret_cast<const char *>(buf->getBuffer()),
565
- buf->getSize());
566
- } else if (contents.isArray()) {
567
- for (int i = 0; i < contents.getArrayNItems(); ++i) {
568
- auto stream = contents.getArrayItem(i);
569
- if (stream.isStream()) {
570
- auto buf = stream.getStreamData(qpdf_dl_generalized);
571
- contentStr.append(reinterpret_cast<const char *>(buf->getBuffer()),
572
- buf->getSize());
573
- contentStr += '\n';
574
- }
575
- }
840
+ streamData = xobj.getStreamData(qpdf_dl_all);
841
+ } catch (...) {
842
+ return;
843
+ }
844
+
845
+ auto w = static_cast<size_t>(width);
846
+ auto h = static_cast<size_t>(height);
847
+ if (streamData->getSize() != w * h)
848
+ return;
849
+
850
+ const auto *pixels = streamData->getBuffer();
851
+ size_t pixelCount = w * h;
852
+
853
+ // check if all pixels are near black (<=32) or near white (>=224)
854
+ bool isBitonal = true;
855
+ for (size_t i = 0; i < pixelCount; ++i) {
856
+ if (pixels[i] > 32 && pixels[i] < 224) {
857
+ isBitonal = false;
858
+ break;
576
859
  }
860
+ }
861
+
862
+ if (!isBitonal)
863
+ return;
577
864
 
578
- // scan for /FontName references — Tf operator uses font name
579
- // pattern: /FontName <size> Tf
580
- for (auto &fontKey : fonts.getKeys()) {
581
- // fontKey includes the leading '/', e.g. "/F1"
582
- if (contentStr.find(fontKey) != std::string::npos)
583
- usedFonts.insert(fontKey);
865
+ // pack into 1-bit: 0 = black, 1 = white
866
+ // each row padded to byte boundary
867
+ size_t rowBytes = (w + 7) / 8;
868
+ std::vector<uint8_t> bitonalData(rowBytes * h, 0);
869
+
870
+ for (size_t y = 0; y < h; ++y) {
871
+ for (size_t x = 0; x < w; ++x) {
872
+ bool white = pixels[y * w + x] >= 224;
873
+ if (white)
874
+ bitonalData[y * rowBytes + x / 8] |=
875
+ static_cast<uint8_t>(0x80 >> (x % 8));
584
876
  }
585
- } catch (...) {
586
- continue; // skip this page if content can't be read
587
877
  }
588
878
 
589
- // remove fonts that are not referenced in the content stream
590
- auto allFontKeys = fonts.getKeys();
591
- for (auto &fontKey : allFontKeys) {
592
- if (usedFonts.find(fontKey) == usedFonts.end())
593
- fonts.removeKey(fontKey);
879
+ // only replace if 1-bit data is smaller than original raw stream
880
+ auto rawData = xobj.getRawStreamData();
881
+ if (bitonalData.size() >= rawData->getSize())
882
+ return;
883
+
884
+ std::string bitStr(reinterpret_cast<char *>(bitonalData.data()),
885
+ bitonalData.size());
886
+ xobj.replaceStreamData(bitStr, QPDFObjectHandle::newNull(),
887
+ QPDFObjectHandle::newNull());
888
+ dict.replaceKey("/BitsPerComponent", QPDFObjectHandle::newInteger(1));
889
+
890
+ if (dict.hasKey("/DecodeParms"))
891
+ dict.removeKey("/DecodeParms");
892
+ if (dict.hasKey("/Predictor"))
893
+ dict.removeKey("/Predictor");
894
+ });
895
+ }
896
+
897
+ // ---------------------------------------------------------------------------
898
+ // Soft mask optimization — losslessly optimize /SMask JPEG streams
899
+ // ---------------------------------------------------------------------------
900
+
901
+ void optimizeSoftMasks(QPDF &qpdf) {
902
+ std::set<QPDFObjGen> processed;
903
+
904
+ forEachImage(qpdf, [&](const std::string & /*key*/, QPDFObjectHandle xobj,
905
+ QPDFObjectHandle /*xobjects*/,
906
+ QPDFPageObjectHelper & /*page*/) {
907
+ auto dict = xobj.getDict();
908
+ if (!dict.hasKey("/SMask"))
909
+ return;
910
+
911
+ auto smask = dict.getKey("/SMask");
912
+ if (!smask.isStream())
913
+ return;
914
+
915
+ auto og = smask.getObjGen();
916
+ if (processed.count(og))
917
+ return;
918
+ processed.insert(og);
919
+
920
+ auto smaskDict = smask.getDict();
921
+ auto filter = smaskDict.getKey("/Filter");
922
+ if (!filter.isName() || filter.getName() != "/DCTDecode")
923
+ return;
924
+
925
+ try {
926
+ auto rawData = smask.getRawStreamData();
927
+
928
+ std::vector<uint8_t> optimized;
929
+ if (!losslessJpegOptimize(rawData->getBuffer(), rawData->getSize(),
930
+ optimized))
931
+ return;
932
+
933
+ if (optimized.size() >= rawData->getSize())
934
+ return;
935
+
936
+ std::string jpegStr(reinterpret_cast<char *>(optimized.data()),
937
+ optimized.size());
938
+ smask.replaceStreamData(jpegStr, QPDFObjectHandle::newName("/DCTDecode"),
939
+ QPDFObjectHandle::newNull());
940
+ } catch (...) {
594
941
  }
595
- }
942
+ });
596
943
  }