qpdf-compress 0.6.0 → 0.6.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/fonts.cc ADDED
@@ -0,0 +1,1209 @@
1
+ #include "fonts.h"
2
+ #include "encoding_tables.h"
3
+ #include "font_subset.h"
4
+
5
+ #include <algorithm>
6
+ #include <cstdint>
7
+ #include <map>
8
+ #include <set>
9
+ #include <string>
10
+ #include <utility>
11
+ #include <vector>
12
+
13
+ #include <qpdf/QPDFObjectHandle.hh>
14
+ #include <qpdf/QPDFPageDocumentHelper.hh>
15
+ #include <qpdf/QPDFPageObjectHelper.hh>
16
+
17
+ namespace {
18
+
19
+ /// Checks if a font BaseFont name has a subset prefix (e.g., "ABCDEF+ArialMT").
20
+ bool isAlreadySubset(QPDFObjectHandle fontObj) {
21
+ auto baseFont = fontObj.getKey("/BaseFont");
22
+ if (!baseFont.isName())
23
+ return false;
24
+ auto name = baseFont.getName();
25
+ if (!name.empty() && name[0] == '/')
26
+ name = name.substr(1);
27
+ if (name.size() <= 7 || name[6] != '+')
28
+ return false;
29
+ for (int i = 0; i < 6; ++i) {
30
+ if (name[i] < 'A' || name[i] > 'Z')
31
+ return false;
32
+ }
33
+ return true;
34
+ }
35
+
36
+ } // namespace
37
+
38
+ // ---------------------------------------------------------------------------
39
+ // Combined parser callback: collects font names referenced by Tf operators
40
+ // AND character codes used by each font in a single pass
41
+ // ---------------------------------------------------------------------------
42
+
43
+ class FontUsageCollector : public QPDFObjectHandle::ParserCallbacks {
44
+ public:
45
+ std::map<std::string, std::set<uint16_t>> fontCodes;
46
+ std::set<std::string> cidFonts; // set before parsing
47
+
48
+ void handleObject(QPDFObjectHandle obj) override {
49
+ if (obj.isOperator()) {
50
+ std::string op = obj.getOperatorValue();
51
+ if (op == "Tf" && operands.size() >= 2) {
52
+ auto nameObj = operands[operands.size() - 2];
53
+ if (nameObj.isName()) {
54
+ currentFont = nameObj.getName();
55
+ }
56
+ }
57
+ if (!currentFont.empty()) {
58
+ bool isCID = cidFonts.count(currentFont) > 0;
59
+ if (op == "Tj" || op == "'" || op == "\"") {
60
+ if (!operands.empty() && operands.back().isString())
61
+ collectFromString(operands.back(), isCID);
62
+ } else if (op == "TJ") {
63
+ if (!operands.empty() && operands.back().isArray()) {
64
+ auto arr = operands.back();
65
+ for (int i = 0; i < arr.getArrayNItems(); ++i) {
66
+ auto item = arr.getArrayItem(i);
67
+ if (item.isString())
68
+ collectFromString(item, isCID);
69
+ }
70
+ }
71
+ }
72
+ }
73
+ operands.clear();
74
+ } else {
75
+ operands.push_back(obj);
76
+ }
77
+ }
78
+ void handleEOF() override {}
79
+
80
+ private:
81
+ std::string currentFont;
82
+ std::vector<QPDFObjectHandle> operands;
83
+
84
+ void collectFromString(QPDFObjectHandle strObj, bool isCID) {
85
+ std::string raw = strObj.getStringValue();
86
+ if (isCID) {
87
+ for (size_t i = 0; i + 1 < raw.size(); i += 2) {
88
+ uint16_t code = (static_cast<uint8_t>(raw[i]) << 8) |
89
+ static_cast<uint8_t>(raw[i + 1]);
90
+ fontCodes[currentFont].insert(code);
91
+ }
92
+ } else {
93
+ for (unsigned char c : raw)
94
+ fontCodes[currentFont].insert(c);
95
+ }
96
+ }
97
+ };
98
+
99
+ // helper: run FontUsageCollector on a stream, swallowing parse errors
100
+ static void collectFontUsageFromStream(QPDFObjectHandle stream,
101
+ FontUsageCollector &collector) {
102
+ try {
103
+ QPDFObjectHandle::parseContentStream(stream, &collector);
104
+ } catch (...) {
105
+ }
106
+ }
107
+
108
+ // ---------------------------------------------------------------------------
109
+ // /ToUnicode CMap parser — converts character codes to Unicode using the
110
+ // font's /ToUnicode stream. handles beginbfchar, beginbfrange (scalar and
111
+ // array forms). this is the most reliable way to map codes to Unicode.
112
+ // ---------------------------------------------------------------------------
113
+
114
+ static std::set<uint16_t> parseToUnicode(QPDFObjectHandle toUnicodeStream,
115
+ const std::set<uint16_t> &charCodes) {
116
+ std::set<uint16_t> unicodeCodes;
117
+
118
+ try {
119
+ auto data = toUnicodeStream.getStreamData(qpdf_dl_all);
120
+ std::string cmap(reinterpret_cast<const char *>(data->getBuffer()),
121
+ data->getSize());
122
+
123
+ // helper: find <hex> value at/after pos, return parsed value and position
124
+ // after closing >
125
+ auto findHexValue = [&](size_t pos,
126
+ size_t end) -> std::pair<uint16_t, size_t> {
127
+ size_t start = cmap.find('<', pos);
128
+ if (start == std::string::npos || start >= end)
129
+ return {0, std::string::npos};
130
+ size_t stop = cmap.find('>', start);
131
+ if (stop == std::string::npos || stop >= end)
132
+ return {0, std::string::npos};
133
+ std::string hex = cmap.substr(start + 1, stop - start - 1);
134
+ if (hex.empty())
135
+ return {0, stop + 1};
136
+ return {static_cast<uint16_t>(std::stoul(hex, nullptr, 16)), stop + 1};
137
+ };
138
+
139
+ // process beginbfchar sections: <srcCode> <dstUnicode>
140
+ size_t pos = 0;
141
+ while (true) {
142
+ size_t start = cmap.find("beginbfchar", pos);
143
+ if (start == std::string::npos)
144
+ break;
145
+ start += 11;
146
+ size_t end = cmap.find("endbfchar", start);
147
+ if (end == std::string::npos)
148
+ break;
149
+
150
+ size_t p = start;
151
+ while (p < end) {
152
+ auto [src, after1] = findHexValue(p, end);
153
+ if (after1 == std::string::npos)
154
+ break;
155
+ auto [dst, after2] = findHexValue(after1, end);
156
+ if (after2 == std::string::npos)
157
+ break;
158
+ if (charCodes.count(src))
159
+ unicodeCodes.insert(dst);
160
+ p = after2;
161
+ }
162
+ pos = end + 9;
163
+ }
164
+
165
+ // process beginbfrange sections: <lo> <hi> <dstStart> or <lo> <hi> [...]
166
+ pos = 0;
167
+ while (true) {
168
+ size_t start = cmap.find("beginbfrange", pos);
169
+ if (start == std::string::npos)
170
+ break;
171
+ start += 12;
172
+ size_t end = cmap.find("endbfrange", start);
173
+ if (end == std::string::npos)
174
+ break;
175
+
176
+ size_t p = start;
177
+ while (p < end) {
178
+ auto [lo, after1] = findHexValue(p, end);
179
+ if (after1 == std::string::npos)
180
+ break;
181
+ auto [hi, after2] = findHexValue(after1, end);
182
+ if (after2 == std::string::npos)
183
+ break;
184
+
185
+ // check for array form vs scalar form
186
+ size_t next = after2;
187
+ while (next < end &&
188
+ std::isspace(static_cast<unsigned char>(cmap[next])))
189
+ ++next;
190
+
191
+ if (next < end && cmap[next] == '[') {
192
+ // array form: <lo> <hi> [<v1> <v2> ...]
193
+ size_t arrEnd = cmap.find(']', next);
194
+ if (arrEnd == std::string::npos || arrEnd >= end)
195
+ break;
196
+ uint16_t code = lo;
197
+ size_t ap = next + 1;
198
+ while (ap < arrEnd && code <= hi) {
199
+ auto [val, afterVal] = findHexValue(ap, arrEnd);
200
+ if (afterVal == std::string::npos)
201
+ break;
202
+ if (charCodes.count(code))
203
+ unicodeCodes.insert(val);
204
+ ++code;
205
+ ap = afterVal;
206
+ }
207
+ p = arrEnd + 1;
208
+ } else {
209
+ // scalar form: <lo> <hi> <dstStart>
210
+ auto [dstStart, after3] = findHexValue(after2, end);
211
+ if (after3 == std::string::npos)
212
+ break;
213
+ for (uint16_t c = lo; c <= hi; ++c) {
214
+ if (charCodes.count(c))
215
+ unicodeCodes.insert(static_cast<uint16_t>(dstStart + (c - lo)));
216
+ }
217
+ p = after3;
218
+ }
219
+ }
220
+ pos = end + 10;
221
+ }
222
+ } catch (...) {
223
+ // parsing failed — return what we have so far
224
+ }
225
+
226
+ return unicodeCodes;
227
+ }
228
+
229
+ // ---------------------------------------------------------------------------
230
+ // /Encoding /Differences parser — extract glyph names for character codes
231
+ // that have been remapped via /Differences entries.
232
+ // format: [code1 /name1 /name2 ... code2 /name3 ...]
233
+ // integers set current position; names are assigned sequentially.
234
+ // ---------------------------------------------------------------------------
235
+
236
+ static std::vector<std::string>
237
+ getGlyphNamesFromEncoding(const std::set<uint16_t> &codes,
238
+ QPDFObjectHandle fontObj) {
239
+ std::vector<std::string> names;
240
+
241
+ auto encoding = fontObj.getKey("/Encoding");
242
+ if (!encoding.isDictionary())
243
+ return names;
244
+
245
+ auto diffs = encoding.getKey("/Differences");
246
+ if (!diffs.isArray())
247
+ return names;
248
+
249
+ // parse /Differences: integers set position, names are glyph names
250
+ std::map<uint16_t, std::string> codeToName;
251
+ int currentCode = 0;
252
+ for (int i = 0; i < diffs.getArrayNItems(); ++i) {
253
+ auto item = diffs.getArrayItem(i);
254
+ if (item.isInteger()) {
255
+ currentCode = static_cast<int>(item.getIntValue());
256
+ } else if (item.isName()) {
257
+ std::string name = item.getName();
258
+ // strip leading / from PDF name
259
+ if (!name.empty() && name[0] == '/')
260
+ name = name.substr(1);
261
+ codeToName[static_cast<uint16_t>(currentCode)] = name;
262
+ ++currentCode;
263
+ }
264
+ }
265
+
266
+ // return glyph names for used codes that have /Differences entries
267
+ for (uint16_t code : codes) {
268
+ auto it = codeToName.find(code);
269
+ if (it != codeToName.end())
270
+ names.push_back(it->second);
271
+ }
272
+
273
+ return names;
274
+ }
275
+
276
+ // ---------------------------------------------------------------------------
277
+ // Encoding helpers — convert encoding-specific byte codes to Unicode
278
+ // for correct cmap lookups during font subsetting
279
+ // ---------------------------------------------------------------------------
280
+
281
+ // convert Adobe Glyph List (AGL) name to Unicode codepoint. handles:
282
+ // - "uniXXXX" convention (4 hex digits after "uni")
283
+ // - "uXXXX"/"uXXXXX" convention (4-6 hex digits after "u")
284
+ // - standard AGL names for common Latin/currency/typographic characters
285
+ static uint16_t glyphNameToUnicode(const std::string &name) {
286
+ // strip any variant suffix (e.g., "Euro.oldstyle" → "Euro")
287
+ std::string base = name;
288
+ auto dot = base.find('.');
289
+ if (dot != std::string::npos)
290
+ base = base.substr(0, dot);
291
+
292
+ // "uniXXXX" — standard Unicode naming convention
293
+ if (base.size() == 7 && base[0] == 'u' && base[1] == 'n' && base[2] == 'i') {
294
+ char *end = nullptr;
295
+ unsigned long val = strtoul(base.c_str() + 3, &end, 16);
296
+ if (end == base.c_str() + 7 && val > 0 && val <= 0xFFFF)
297
+ return static_cast<uint16_t>(val);
298
+ }
299
+
300
+ // "uXXXX" or "uXXXXX" — alternate Unicode naming convention
301
+ if (base.size() >= 5 && base.size() <= 7 && base[0] == 'u' &&
302
+ base[1] != 'n') {
303
+ char *end = nullptr;
304
+ unsigned long val = strtoul(base.c_str() + 1, &end, 16);
305
+ if (end == base.c_str() + static_cast<ptrdiff_t>(base.size()) && val > 0 &&
306
+ val <= 0xFFFF)
307
+ return static_cast<uint16_t>(val);
308
+ }
309
+
310
+ // single-character names map to their ASCII value
311
+ if (base.size() == 1 && base[0] >= 0x20)
312
+ return static_cast<uint16_t>(static_cast<unsigned char>(base[0]));
313
+
314
+ // standard AGL names — covers WinAnsi, Latin Extended, and common symbols
315
+ // sourced from the Adobe Glyph List (agl-aglfn) specification
316
+ static const std::map<std::string, uint16_t> agl = {
317
+ // basic ASCII names
318
+ {"space", 0x0020},
319
+ {"exclam", 0x0021},
320
+ {"quotedbl", 0x0022},
321
+ {"numbersign", 0x0023},
322
+ {"dollar", 0x0024},
323
+ {"percent", 0x0025},
324
+ {"ampersand", 0x0026},
325
+ {"quotesingle", 0x0027},
326
+ {"parenleft", 0x0028},
327
+ {"parenright", 0x0029},
328
+ {"asterisk", 0x002A},
329
+ {"plus", 0x002B},
330
+ {"comma", 0x002C},
331
+ {"hyphen", 0x002D},
332
+ {"period", 0x002E},
333
+ {"slash", 0x002F},
334
+ {"zero", 0x0030},
335
+ {"one", 0x0031},
336
+ {"two", 0x0032},
337
+ {"three", 0x0033},
338
+ {"four", 0x0034},
339
+ {"five", 0x0035},
340
+ {"six", 0x0036},
341
+ {"seven", 0x0037},
342
+ {"eight", 0x0038},
343
+ {"nine", 0x0039},
344
+ {"colon", 0x003A},
345
+ {"semicolon", 0x003B},
346
+ {"less", 0x003C},
347
+ {"equal", 0x003D},
348
+ {"greater", 0x003E},
349
+ {"question", 0x003F},
350
+ {"at", 0x0040},
351
+ {"bracketleft", 0x005B},
352
+ {"backslash", 0x005C},
353
+ {"bracketright", 0x005D},
354
+ {"asciicircum", 0x005E},
355
+ {"underscore", 0x005F},
356
+ {"grave", 0x0060},
357
+ {"braceleft", 0x007B},
358
+ {"bar", 0x007C},
359
+ {"braceright", 0x007D},
360
+ {"asciitilde", 0x007E},
361
+ // WinAnsi 0x80-0x9F range
362
+ {"Euro", 0x20AC},
363
+ {"quotesinglbase", 0x201A},
364
+ {"florin", 0x0192},
365
+ {"quotedblbase", 0x201E},
366
+ {"ellipsis", 0x2026},
367
+ {"dagger", 0x2020},
368
+ {"daggerdbl", 0x2021},
369
+ {"circumflex", 0x02C6},
370
+ {"perthousand", 0x2030},
371
+ {"Scaron", 0x0160},
372
+ {"guilsinglleft", 0x2039},
373
+ {"OE", 0x0152},
374
+ {"Zcaron", 0x017D},
375
+ {"quoteleft", 0x2018},
376
+ {"quoteright", 0x2019},
377
+ {"quotedblleft", 0x201C},
378
+ {"quotedblright", 0x201D},
379
+ {"bullet", 0x2022},
380
+ {"endash", 0x2013},
381
+ {"emdash", 0x2014},
382
+ {"tilde", 0x02DC},
383
+ {"trademark", 0x2122},
384
+ {"scaron", 0x0161},
385
+ {"guilsinglright", 0x203A},
386
+ {"oe", 0x0153},
387
+ {"zcaron", 0x017E},
388
+ {"Ydieresis", 0x0178},
389
+ // Latin-1 Supplement (0xA0-0xFF)
390
+ {"nbspace", 0x00A0},
391
+ {"exclamdown", 0x00A1},
392
+ {"cent", 0x00A2},
393
+ {"sterling", 0x00A3},
394
+ {"currency", 0x00A4},
395
+ {"yen", 0x00A5},
396
+ {"brokenbar", 0x00A6},
397
+ {"section", 0x00A7},
398
+ {"dieresis", 0x00A8},
399
+ {"copyright", 0x00A9},
400
+ {"ordfeminine", 0x00AA},
401
+ {"guillemotleft", 0x00AB},
402
+ {"logicalnot", 0x00AC},
403
+ {"softhyphen", 0x00AD},
404
+ {"registered", 0x00AE},
405
+ {"macron", 0x00AF},
406
+ {"degree", 0x00B0},
407
+ {"plusminus", 0x00B1},
408
+ {"twosuperior", 0x00B2},
409
+ {"threesuperior", 0x00B3},
410
+ {"acute", 0x00B4},
411
+ {"mu", 0x00B5},
412
+ {"paragraph", 0x00B6},
413
+ {"periodcentered", 0x00B7},
414
+ {"cedilla", 0x00B8},
415
+ {"onesuperior", 0x00B9},
416
+ {"ordmasculine", 0x00BA},
417
+ {"guillemotright", 0x00BB},
418
+ {"onequarter", 0x00BC},
419
+ {"onehalf", 0x00BD},
420
+ {"threequarters", 0x00BE},
421
+ {"questiondown", 0x00BF},
422
+ {"Agrave", 0x00C0},
423
+ {"Aacute", 0x00C1},
424
+ {"Acircumflex", 0x00C2},
425
+ {"Atilde", 0x00C3},
426
+ {"Adieresis", 0x00C4},
427
+ {"Aring", 0x00C5},
428
+ {"AE", 0x00C6},
429
+ {"Ccedilla", 0x00C7},
430
+ {"Egrave", 0x00C8},
431
+ {"Eacute", 0x00C9},
432
+ {"Ecircumflex", 0x00CA},
433
+ {"Edieresis", 0x00CB},
434
+ {"Igrave", 0x00CC},
435
+ {"Iacute", 0x00CD},
436
+ {"Icircumflex", 0x00CE},
437
+ {"Idieresis", 0x00CF},
438
+ {"Eth", 0x00D0},
439
+ {"Ntilde", 0x00D1},
440
+ {"Ograve", 0x00D2},
441
+ {"Oacute", 0x00D3},
442
+ {"Ocircumflex", 0x00D4},
443
+ {"Otilde", 0x00D5},
444
+ {"Odieresis", 0x00D6},
445
+ {"multiply", 0x00D7},
446
+ {"Oslash", 0x00D8},
447
+ {"Ugrave", 0x00D9},
448
+ {"Uacute", 0x00DA},
449
+ {"Ucircumflex", 0x00DB},
450
+ {"Udieresis", 0x00DC},
451
+ {"Yacute", 0x00DD},
452
+ {"Thorn", 0x00DE},
453
+ {"germandbls", 0x00DF},
454
+ {"agrave", 0x00E0},
455
+ {"aacute", 0x00E1},
456
+ {"acircumflex", 0x00E2},
457
+ {"atilde", 0x00E3},
458
+ {"adieresis", 0x00E4},
459
+ {"aring", 0x00E5},
460
+ {"ae", 0x00E6},
461
+ {"ccedilla", 0x00E7},
462
+ {"egrave", 0x00E8},
463
+ {"eacute", 0x00E9},
464
+ {"ecircumflex", 0x00EA},
465
+ {"edieresis", 0x00EB},
466
+ {"igrave", 0x00EC},
467
+ {"iacute", 0x00ED},
468
+ {"icircumflex", 0x00EE},
469
+ {"idieresis", 0x00EF},
470
+ {"eth", 0x00F0},
471
+ {"ntilde", 0x00F1},
472
+ {"ograve", 0x00F2},
473
+ {"oacute", 0x00F3},
474
+ {"ocircumflex", 0x00F4},
475
+ {"otilde", 0x00F5},
476
+ {"odieresis", 0x00F6},
477
+ {"divide", 0x00F7},
478
+ {"oslash", 0x00F8},
479
+ {"ugrave", 0x00F9},
480
+ {"uacute", 0x00FA},
481
+ {"ucircumflex", 0x00FB},
482
+ {"udieresis", 0x00FC},
483
+ {"yacute", 0x00FD},
484
+ {"thorn", 0x00FE},
485
+ {"ydieresis", 0x00FF},
486
+ // Latin Extended-A (commonly found in CE font /Differences)
487
+ {"Amacron", 0x0100},
488
+ {"amacron", 0x0101},
489
+ {"Abreve", 0x0102},
490
+ {"abreve", 0x0103},
491
+ {"Aogonek", 0x0104},
492
+ {"aogonek", 0x0105},
493
+ {"Cacute", 0x0106},
494
+ {"cacute", 0x0107},
495
+ {"Ccircumflex", 0x0108},
496
+ {"ccircumflex", 0x0109},
497
+ {"Cdotaccent", 0x010A},
498
+ {"cdotaccent", 0x010B},
499
+ {"Ccaron", 0x010C},
500
+ {"ccaron", 0x010D},
501
+ {"Dcaron", 0x010E},
502
+ {"dcaron", 0x010F},
503
+ {"Dcroat", 0x0110},
504
+ {"dcroat", 0x0111},
505
+ {"Emacron", 0x0112},
506
+ {"emacron", 0x0113},
507
+ {"Ebreve", 0x0114},
508
+ {"ebreve", 0x0115},
509
+ {"Edotaccent", 0x0116},
510
+ {"edotaccent", 0x0117},
511
+ {"Eogonek", 0x0118},
512
+ {"eogonek", 0x0119},
513
+ {"Ecaron", 0x011A},
514
+ {"ecaron", 0x011B},
515
+ {"Gbreve", 0x011E},
516
+ {"gbreve", 0x011F},
517
+ {"Gdotaccent", 0x0120},
518
+ {"gdotaccent", 0x0121},
519
+ {"Gcommaaccent", 0x0122},
520
+ {"gcommaaccent", 0x0123},
521
+ {"Idotaccent", 0x0130},
522
+ {"dotlessi", 0x0131},
523
+ {"IJ", 0x0132},
524
+ {"ij", 0x0133},
525
+ {"Lacute", 0x0139},
526
+ {"lacute", 0x013A},
527
+ {"Lcommaaccent", 0x013B},
528
+ {"lcommaaccent", 0x013C},
529
+ {"Lcaron", 0x013D},
530
+ {"lcaron", 0x013E},
531
+ {"Ldot", 0x013F},
532
+ {"ldot", 0x0140},
533
+ {"Lslash", 0x0141},
534
+ {"lslash", 0x0142},
535
+ {"Nacute", 0x0143},
536
+ {"nacute", 0x0144},
537
+ {"Ncommaaccent", 0x0145},
538
+ {"ncommaaccent", 0x0146},
539
+ {"Ncaron", 0x0147},
540
+ {"ncaron", 0x0148},
541
+ {"Eng", 0x014A},
542
+ {"eng", 0x014B},
543
+ {"Omacron", 0x014C},
544
+ {"omacron", 0x014D},
545
+ {"Obreve", 0x014E},
546
+ {"obreve", 0x014F},
547
+ {"Ohungarumlaut", 0x0150},
548
+ {"ohungarumlaut", 0x0151},
549
+ {"Racute", 0x0154},
550
+ {"racute", 0x0155},
551
+ {"Rcommaaccent", 0x0156},
552
+ {"rcommaaccent", 0x0157},
553
+ {"Rcaron", 0x0158},
554
+ {"rcaron", 0x0159},
555
+ {"Sacute", 0x015A},
556
+ {"sacute", 0x015B},
557
+ {"Scircumflex", 0x015C},
558
+ {"scircumflex", 0x015D},
559
+ {"Scedilla", 0x015E},
560
+ {"scedilla", 0x015F},
561
+ {"Tcaron", 0x0164},
562
+ {"tcaron", 0x0165},
563
+ {"Tbar", 0x0166},
564
+ {"tbar", 0x0167},
565
+ {"Umacron", 0x016A},
566
+ {"umacron", 0x016B},
567
+ {"Ubreve", 0x016C},
568
+ {"ubreve", 0x016D},
569
+ {"Uring", 0x016E},
570
+ {"uring", 0x016F},
571
+ {"Uhungarumlaut", 0x0170},
572
+ {"uhungarumlaut", 0x0171},
573
+ {"Uogonek", 0x0172},
574
+ {"uogonek", 0x0173},
575
+ {"Zacute", 0x0179},
576
+ {"zacute", 0x017A},
577
+ {"Zdotaccent", 0x017B},
578
+ {"zdotaccent", 0x017C},
579
+ // common currency symbols
580
+ {"colonmonetary", 0x20A1},
581
+ {"franc", 0x20A3},
582
+ {"lira", 0x20A4},
583
+ {"peseta", 0x20A7},
584
+ {"won", 0x20A9},
585
+ {"dong", 0x20AB},
586
+ // typographic symbols
587
+ {"fi", 0xFB01},
588
+ {"fl", 0xFB02},
589
+ {"minus", 0x2212},
590
+ {"fraction", 0x2044},
591
+ };
592
+
593
+ auto it = agl.find(base);
594
+ if (it != agl.end())
595
+ return it->second;
596
+ return 0;
597
+ }
598
+
599
+ // parse /Encoding /Differences to build a map of code → glyph name.
600
+ // used to identify which codes are remapped by /Differences.
601
+ static std::map<uint16_t, std::string>
602
+ getDifferencesCodeMap(QPDFObjectHandle fontObj) {
603
+ std::map<uint16_t, std::string> codeToName;
604
+
605
+ auto encoding = fontObj.getKey("/Encoding");
606
+ if (!encoding.isDictionary())
607
+ return codeToName;
608
+
609
+ auto diffs = encoding.getKey("/Differences");
610
+ if (!diffs.isArray())
611
+ return codeToName;
612
+
613
+ int currentCode = 0;
614
+ for (int i = 0; i < diffs.getArrayNItems(); ++i) {
615
+ auto item = diffs.getArrayItem(i);
616
+ if (item.isInteger()) {
617
+ currentCode = static_cast<int>(item.getIntValue());
618
+ } else if (item.isName()) {
619
+ std::string name = item.getName();
620
+ if (!name.empty() && name[0] == '/')
621
+ name = name.substr(1);
622
+ codeToName[static_cast<uint16_t>(currentCode)] = name;
623
+ ++currentCode;
624
+ }
625
+ }
626
+ return codeToName;
627
+ }
628
+
629
+ // convert encoding-specific character codes to Unicode for cmap lookup
630
+ static std::set<uint16_t> convertCodesToUnicode(const std::set<uint16_t> &codes,
631
+ QPDFObjectHandle fontObj) {
632
+ auto encoding = fontObj.getKey("/Encoding");
633
+
634
+ bool isWinAnsi = false;
635
+ bool isMacRoman = false;
636
+ if (encoding.isName()) {
637
+ if (encoding.getName() == "/WinAnsiEncoding")
638
+ isWinAnsi = true;
639
+ else if (encoding.getName() == "/MacRomanEncoding")
640
+ isMacRoman = true;
641
+ } else if (encoding.isDictionary()) {
642
+ auto baseEnc = encoding.getKey("/BaseEncoding");
643
+ if (baseEnc.isName()) {
644
+ if (baseEnc.getName() == "/WinAnsiEncoding")
645
+ isWinAnsi = true;
646
+ else if (baseEnc.getName() == "/MacRomanEncoding")
647
+ isMacRoman = true;
648
+ }
649
+ }
650
+
651
+ if (!isWinAnsi && !isMacRoman)
652
+ return codes;
653
+
654
+ std::set<uint16_t> unicodeCodes;
655
+ for (uint16_t code : codes) {
656
+ if (isWinAnsi)
657
+ unicodeCodes.insert(winAnsiToUnicode(static_cast<uint8_t>(code)));
658
+ else
659
+ unicodeCodes.insert(macRomanToUnicode(static_cast<uint8_t>(code)));
660
+ }
661
+ return unicodeCodes;
662
+ }
663
+
664
+ // ---------------------------------------------------------------------------
665
+ // Combined font optimization — remove unused fonts AND subset remaining
666
+ // fonts in a single page walk
667
+ // ---------------------------------------------------------------------------
668
+
669
+ void optimizeFonts(QPDF &qpdf) {
670
+ // single page walk: collect used font names + character codes per font
671
+ std::map<QPDFObjGen, std::set<uint16_t>> fontUsedCodes;
672
+
673
+ for (auto &page : QPDFPageDocumentHelper(qpdf).getAllPages()) {
674
+ auto pageObj = page.getObjectHandle();
675
+ auto resources = pageObj.getKey("/Resources");
676
+ if (!resources.isDictionary())
677
+ continue;
678
+ auto fonts = resources.getKey("/Font");
679
+ if (!fonts.isDictionary())
680
+ continue;
681
+
682
+ // build list of (stream, fontDict) pairs to scan
683
+ struct StreamFonts {
684
+ QPDFObjectHandle stream;
685
+ QPDFObjectHandle fontDict;
686
+ };
687
+ std::vector<StreamFonts> scanTargets;
688
+
689
+ try {
690
+ auto contents = pageObj.getKey("/Contents");
691
+ scanTargets.push_back({contents, fonts});
692
+ } catch (...) {
693
+ continue;
694
+ }
695
+
696
+ auto xobjects = resources.getKey("/XObject");
697
+ if (xobjects.isDictionary()) {
698
+ // recursively scan Form XObjects (queue-based to handle nesting)
699
+ std::vector<QPDFObjectHandle> xobjQueue;
700
+ std::set<QPDFObjGen> visitedXObjects;
701
+ for (auto &xobjKey : xobjects.getKeys())
702
+ xobjQueue.push_back(xobjects.getKey(xobjKey));
703
+
704
+ while (!xobjQueue.empty()) {
705
+ auto xobj = xobjQueue.back();
706
+ xobjQueue.pop_back();
707
+ if (!xobj.isStream())
708
+ continue;
709
+ auto xobjOg = xobj.getObjGen();
710
+ if (visitedXObjects.count(xobjOg))
711
+ continue;
712
+ visitedXObjects.insert(xobjOg);
713
+
714
+ auto xobjDict = xobj.getDict();
715
+ auto xobjSubtype = xobjDict.getKey("/Subtype");
716
+ if (!xobjSubtype.isName() || xobjSubtype.getName() != "/Form")
717
+ continue;
718
+
719
+ auto xobjRes = xobjDict.getKey("/Resources");
720
+ if (xobjRes.isDictionary() && xobjRes.getKey("/Font").isDictionary())
721
+ scanTargets.push_back({xobj, xobjRes.getKey("/Font")});
722
+ else
723
+ scanTargets.push_back({xobj, fonts});
724
+
725
+ // queue nested XObjects for recursive scanning
726
+ if (xobjRes.isDictionary()) {
727
+ auto nestedXObjects = xobjRes.getKey("/XObject");
728
+ if (nestedXObjects.isDictionary()) {
729
+ for (auto &nestedKey : nestedXObjects.getKeys())
730
+ xobjQueue.push_back(nestedXObjects.getKey(nestedKey));
731
+ }
732
+ }
733
+ }
734
+ }
735
+
736
+ // determine which fonts are CID for the collector
737
+ FontUsageCollector collector;
738
+ for (auto &target : scanTargets) {
739
+ for (auto &key : target.fontDict.getKeys()) {
740
+ auto fontObj = target.fontDict.getKey(key);
741
+ if (!fontObj.isDictionary())
742
+ continue;
743
+ auto subtypeKey = fontObj.getKey("/Subtype");
744
+ if (subtypeKey.isName() && subtypeKey.getName() == "/Type0")
745
+ collector.cidFonts.insert(key);
746
+ }
747
+ }
748
+
749
+ // parse all content streams for this page in one pass per stream
750
+ for (auto &target : scanTargets)
751
+ collectFontUsageFromStream(target.stream, collector);
752
+
753
+ // map per-name codes to per-ObjGen codes for subsetting
754
+ for (auto &target : scanTargets) {
755
+ for (auto &key : target.fontDict.getKeys()) {
756
+ auto it = collector.fontCodes.find(key);
757
+ if (it == collector.fontCodes.end())
758
+ continue;
759
+ auto fontObj = target.fontDict.getKey(key);
760
+ if (!fontObj.isDictionary())
761
+ continue;
762
+ auto og = fontObj.getObjGen();
763
+ fontUsedCodes[og].insert(it->second.begin(), it->second.end());
764
+ }
765
+ }
766
+ }
767
+
768
+ // for each font with a /Widths array, zero out widths for unused glyphs
769
+ // and truncate trailing zeros
770
+ for (auto &[og, usedCodes] : fontUsedCodes) {
771
+ auto fontObj = qpdf.getObjectByObjGen(og);
772
+ if (!fontObj.isDictionary())
773
+ continue;
774
+
775
+ auto subtype = fontObj.getKey("/Subtype");
776
+ if (!subtype.isName())
777
+ continue;
778
+
779
+ // only handle simple fonts with /Widths arrays (TrueType, Type1)
780
+ if (subtype.getName() != "/TrueType" && subtype.getName() != "/Type1")
781
+ continue;
782
+
783
+ // skip already-subset fonts — their encodings are custom and our usage
784
+ // collector may miscount character codes
785
+ if (isAlreadySubset(fontObj))
786
+ continue;
787
+
788
+ auto widths = fontObj.getKey("/Widths");
789
+ auto firstCharObj = fontObj.getKey("/FirstChar");
790
+ if (!widths.isArray() || !firstCharObj.isInteger())
791
+ continue;
792
+
793
+ int firstChar = static_cast<int>(firstCharObj.getIntValue());
794
+ int widthCount = widths.getArrayNItems();
795
+
796
+ // zero out widths for unused character codes
797
+ bool modified = false;
798
+ for (int i = 0; i < widthCount; ++i) {
799
+ int charCode = firstChar + i;
800
+ if (usedCodes.find(static_cast<uint16_t>(charCode)) == usedCodes.end()) {
801
+ auto w = widths.getArrayItem(i);
802
+ if (w.isInteger() && w.getIntValue() != 0) {
803
+ widths.setArrayItem(i, QPDFObjectHandle::newInteger(0));
804
+ modified = true;
805
+ } else if (w.isReal() && w.getNumericValue() != 0.0) {
806
+ widths.setArrayItem(i, QPDFObjectHandle::newInteger(0));
807
+ modified = true;
808
+ }
809
+ }
810
+ }
811
+ if (!modified)
812
+ continue;
813
+
814
+ // trim trailing zero-width entries and adjust /LastChar
815
+ int lastUsed = widthCount - 1;
816
+ while (lastUsed >= 0) {
817
+ auto w = widths.getArrayItem(lastUsed);
818
+ if (w.isInteger() && w.getIntValue() == 0)
819
+ --lastUsed;
820
+ else
821
+ break;
822
+ }
823
+
824
+ if (lastUsed < widthCount - 1) {
825
+ // rebuild the widths array with only the needed entries
826
+ auto newWidths = QPDFObjectHandle::newArray();
827
+ for (int i = 0; i <= lastUsed; ++i)
828
+ newWidths.appendItem(widths.getArrayItem(i));
829
+
830
+ fontObj.replaceKey("/Widths", newWidths);
831
+ fontObj.replaceKey("/LastChar",
832
+ QPDFObjectHandle::newInteger(firstChar + lastUsed));
833
+ }
834
+ }
835
+
836
+ // true font subsetting: strip unused glyph outlines from embedded fonts.
837
+ // two-pass approach: first collect glyph IDs per font file (merging across
838
+ // all font objects sharing the same embedded file), then subset each file.
839
+ std::set<QPDFObjGen> processedFonts;
840
+ std::map<QPDFObjGen, std::set<uint16_t>> fontFileGlyphIds;
841
+ std::map<QPDFObjGen, QPDFObjectHandle> fontFileHandles;
842
+ std::map<QPDFObjGen, bool> fontFileCID; // true if CID font
843
+ std::set<QPDFObjGen> fontFileSkip; // skip if any font fails safety check
844
+
845
+ // pass 1: collect glyph IDs from each font object
846
+ for (auto &[og, usedCodes] : fontUsedCodes) {
847
+ if (processedFonts.count(og))
848
+ continue;
849
+ processedFonts.insert(og);
850
+
851
+ auto fontObj = qpdf.getObjectByObjGen(og);
852
+ if (!fontObj.isDictionary())
853
+ continue;
854
+
855
+ auto subtype = fontObj.getKey("/Subtype");
856
+ if (!subtype.isName())
857
+ continue;
858
+ // Type0 composite fonts — subset CIDFontType2 and CIDFontType0
859
+ // descendants
860
+ if (subtype.getName() == "/Type0") {
861
+ // skip already-subset CID fonts
862
+ if (isAlreadySubset(fontObj))
863
+ continue;
864
+
865
+ auto descendants = fontObj.getKey("/DescendantFonts");
866
+ if (!descendants.isArray() || descendants.getArrayNItems() < 1)
867
+ continue;
868
+
869
+ auto cidFont = descendants.getArrayItem(0);
870
+ if (!cidFont.isDictionary())
871
+ continue;
872
+
873
+ if (isAlreadySubset(cidFont))
874
+ continue;
875
+
876
+ auto cidSubtype = cidFont.getKey("/Subtype");
877
+ if (!cidSubtype.isName())
878
+ continue;
879
+
880
+ bool isCIDFontType2 = cidSubtype.getName() == "/CIDFontType2";
881
+ bool isCIDFontType0 = cidSubtype.getName() == "/CIDFontType0";
882
+ if (!isCIDFontType2 && !isCIDFontType0)
883
+ continue;
884
+
885
+ auto cidDescriptor = cidFont.getKey("/FontDescriptor");
886
+ if (!cidDescriptor.isDictionary())
887
+ continue;
888
+
889
+ // determine which font file key to use:
890
+ // CIDFontType2 → /FontFile2 (TrueType)
891
+ // CIDFontType0 → /FontFile3 (CFF)
892
+ std::string fontFileKey;
893
+ if (isCIDFontType2 && cidDescriptor.hasKey("/FontFile2"))
894
+ fontFileKey = "/FontFile2";
895
+ else if (isCIDFontType0 && cidDescriptor.hasKey("/FontFile3"))
896
+ fontFileKey = "/FontFile3";
897
+ else
898
+ continue;
899
+
900
+ auto fontFile = cidDescriptor.getKey(fontFileKey);
901
+ if (!fontFile.isStream())
902
+ continue;
903
+
904
+ auto ffOg = fontFile.getObjGen();
905
+ fontFileHandles[ffOg] = fontFile;
906
+ fontFileCID[ffOg] = true;
907
+
908
+ try {
909
+ std::set<uint16_t> glyphIds;
910
+
911
+ auto cidToGid = cidFont.getKey("/CIDToGIDMap");
912
+ if (cidToGid.isName() && cidToGid.getName() == "/Identity") {
913
+ glyphIds = usedCodes;
914
+ } else if (cidToGid.isStream()) {
915
+ auto mapData = cidToGid.getStreamData(qpdf_dl_all);
916
+ const uint8_t *mapBuf = mapData->getBuffer();
917
+ size_t mapSize = mapData->getSize();
918
+ for (uint16_t cid : usedCodes) {
919
+ size_t offset = static_cast<size_t>(cid) * 2;
920
+ if (offset + 2 <= mapSize) {
921
+ uint16_t gid = static_cast<uint16_t>((mapBuf[offset] << 8) |
922
+ mapBuf[offset + 1]);
923
+ if (gid != 0)
924
+ glyphIds.insert(gid);
925
+ }
926
+ }
927
+ } else if (isCIDFontType0) {
928
+ glyphIds = usedCodes;
929
+ } else {
930
+ continue;
931
+ }
932
+
933
+ glyphIds.insert(0);
934
+ fontFileGlyphIds[ffOg].insert(glyphIds.begin(), glyphIds.end());
935
+ } catch (...) {
936
+ continue;
937
+ }
938
+ continue;
939
+ }
940
+
941
+ // simple TrueType fonts
942
+ if (subtype.getName() != "/TrueType")
943
+ continue;
944
+
945
+ // skip already-subset fonts (BaseFont like "ABCDEF+FontName") — their
946
+ // custom encodings and glyph tables can't be reliably re-subset
947
+ // skip already-subset fonts — mark their font file as skip so other
948
+ // fonts sharing it won't subset and corrupt the already-subset glyph data
949
+ if (isAlreadySubset(fontObj)) {
950
+ auto descriptor = fontObj.getKey("/FontDescriptor");
951
+ if (descriptor.isDictionary() && descriptor.hasKey("/FontFile2")) {
952
+ auto fontFile = descriptor.getKey("/FontFile2");
953
+ if (fontFile.isStream())
954
+ fontFileSkip.insert(fontFile.getObjGen());
955
+ }
956
+ continue;
957
+ }
958
+
959
+ auto descriptor = fontObj.getKey("/FontDescriptor");
960
+ if (!descriptor.isDictionary() || !descriptor.hasKey("/FontFile2"))
961
+ continue;
962
+
963
+ auto fontFile = descriptor.getKey("/FontFile2");
964
+ if (!fontFile.isStream())
965
+ continue;
966
+
967
+ auto ffOg = fontFile.getObjGen();
968
+ fontFileHandles[ffOg] = fontFile;
969
+ if (!fontFileCID.count(ffOg))
970
+ fontFileCID[ffOg] = false;
971
+
972
+ try {
973
+ auto fontData = fontFile.getStreamData(qpdf_dl_all);
974
+ const uint8_t *ttfData = fontData->getBuffer();
975
+ size_t ttfSize = fontData->getSize();
976
+
977
+ std::set<uint16_t> glyphIds;
978
+
979
+ // build /Differences map to identify remapped codes
980
+ auto diffMap = getDifferencesCodeMap(fontObj);
981
+ std::set<uint16_t> diffCodes; // codes that have /Differences entries
982
+ for (auto &[code, name] : diffMap) {
983
+ if (usedCodes.count(code))
984
+ diffCodes.insert(code);
985
+ }
986
+ std::set<uint16_t> baseCodes; // codes using base encoding (no remap)
987
+ for (uint16_t code : usedCodes) {
988
+ if (!diffCodes.count(code))
989
+ baseCodes.insert(code);
990
+ }
991
+
992
+ bool hasToUnicode = false;
993
+
994
+ // strategy 1: /ToUnicode CMap — most reliable Unicode mapping
995
+ auto toUnicode = fontObj.getKey("/ToUnicode");
996
+ if (toUnicode.isStream()) {
997
+ hasToUnicode = true;
998
+ auto tuCodes = parseToUnicode(toUnicode, usedCodes);
999
+ if (!tuCodes.empty()) {
1000
+ auto ids = mapCodesToGlyphIds(ttfData, ttfSize, tuCodes);
1001
+ glyphIds.insert(ids.begin(), ids.end());
1002
+ }
1003
+ }
1004
+
1005
+ // strategy 2: /Encoding /Differences — glyph name lookup via post table
1006
+ auto diffNames = getGlyphNamesFromEncoding(usedCodes, fontObj);
1007
+ if (!diffNames.empty()) {
1008
+ auto nameIds = mapGlyphNamesToGlyphIds(ttfData, ttfSize, diffNames);
1009
+ glyphIds.insert(nameIds.begin(), nameIds.end());
1010
+ }
1011
+
1012
+ // strategy 3: /Differences glyph names → Unicode → cmap lookup.
1013
+ // converts glyph names to Unicode codepoints (via AGL/uniXXXX)
1014
+ // and looks them up in the font's cmap. this catches characters like
1015
+ // ẞ (uni1E9E), € (Euro), • (bullet) that have non-standard byte codes
1016
+ // in /Differences but standard Unicode entries in the cmap.
1017
+ if (!diffCodes.empty()) {
1018
+ std::set<uint16_t> diffUnicodes;
1019
+ for (uint16_t code : diffCodes) {
1020
+ auto it = diffMap.find(code);
1021
+ if (it != diffMap.end()) {
1022
+ uint16_t unicode = glyphNameToUnicode(it->second);
1023
+ if (unicode > 0)
1024
+ diffUnicodes.insert(unicode);
1025
+ }
1026
+ }
1027
+ if (!diffUnicodes.empty()) {
1028
+ auto ids = mapCodesToGlyphIds(ttfData, ttfSize, diffUnicodes);
1029
+ glyphIds.insert(ids.begin(), ids.end());
1030
+ }
1031
+ }
1032
+
1033
+ // strategy 4: base encoding conversion (WinAnsi, MacRoman) → cmap
1034
+ // only for codes NOT remapped by /Differences — those are already
1035
+ // handled by strategies 2 and 3 with correct Unicode
1036
+ if (!baseCodes.empty()) {
1037
+ auto unicodeCodes = convertCodesToUnicode(baseCodes, fontObj);
1038
+ auto encIds = mapCodesToGlyphIds(ttfData, ttfSize, unicodeCodes);
1039
+ glyphIds.insert(encIds.begin(), encIds.end());
1040
+
1041
+ // strategy 5: raw character codes as fallback — only when byte codes
1042
+ // might directly index the cmap (no /ToUnicode remapping)
1043
+ if (!hasToUnicode && unicodeCodes != baseCodes) {
1044
+ auto rawGlyphs = mapCodesToGlyphIds(ttfData, ttfSize, baseCodes);
1045
+ glyphIds.insert(rawGlyphs.begin(), rawGlyphs.end());
1046
+ }
1047
+
1048
+ // strategy 6: "uniXXXX" glyph names via post table — catches fonts
1049
+ // where glyphs have no cmap entry but are accessible by name
1050
+ {
1051
+ std::vector<std::string> uniNames;
1052
+ for (uint16_t u : unicodeCodes) {
1053
+ if (u > 0x7F) {
1054
+ char buf[8];
1055
+ snprintf(buf, sizeof(buf), "uni%04X", u);
1056
+ uniNames.emplace_back(buf);
1057
+ }
1058
+ }
1059
+ if (!uniNames.empty()) {
1060
+ auto nameIds = mapGlyphNamesToGlyphIds(ttfData, ttfSize, uniNames);
1061
+ glyphIds.insert(nameIds.begin(), nameIds.end());
1062
+ }
1063
+ }
1064
+ }
1065
+
1066
+ // safety check: if we found fewer glyph IDs than used character codes,
1067
+ // some characters couldn't be mapped — mark font file as unsafe to
1068
+ // subset. if ANY font sharing this file fails, skip the entire file.
1069
+ glyphIds.insert(0);
1070
+ if (glyphIds.size() - 1 < usedCodes.size()) {
1071
+ fontFileSkip.insert(ffOg);
1072
+ continue;
1073
+ }
1074
+
1075
+ fontFileGlyphIds[ffOg].insert(glyphIds.begin(), glyphIds.end());
1076
+ } catch (...) {
1077
+ continue;
1078
+ }
1079
+ }
1080
+
1081
+ // pass 2: subset each font file with merged glyph IDs from all font objects
1082
+ for (auto &[ffOg, glyphIds] : fontFileGlyphIds) {
1083
+ if (fontFileSkip.count(ffOg))
1084
+ continue;
1085
+
1086
+ auto it = fontFileHandles.find(ffOg);
1087
+ if (it == fontFileHandles.end())
1088
+ continue;
1089
+
1090
+ auto &fontFile = it->second;
1091
+ bool isCID = fontFileCID[ffOg];
1092
+
1093
+ try {
1094
+ auto fontData = fontFile.getStreamData(qpdf_dl_all);
1095
+ const uint8_t *rawData = fontData->getBuffer();
1096
+ size_t rawSize = fontData->getSize();
1097
+
1098
+ std::vector<uint8_t> subsetResult;
1099
+ if (!subsetFont(rawData, rawSize, glyphIds, subsetResult, !isCID))
1100
+ continue;
1101
+
1102
+ if (subsetResult.size() >= rawSize)
1103
+ continue;
1104
+
1105
+ std::string fontStr(reinterpret_cast<char *>(subsetResult.data()),
1106
+ subsetResult.size());
1107
+ fontFile.replaceStreamData(fontStr, QPDFObjectHandle::newNull(),
1108
+ QPDFObjectHandle::newNull());
1109
+ } catch (...) {
1110
+ continue;
1111
+ }
1112
+ }
1113
+
1114
+ // optimize CID font /W arrays — rebuild with only used CID entries
1115
+ for (auto &[og, usedCodes] : fontUsedCodes) {
1116
+ auto fontObj = qpdf.getObjectByObjGen(og);
1117
+ if (!fontObj.isDictionary())
1118
+ continue;
1119
+
1120
+ auto subtype = fontObj.getKey("/Subtype");
1121
+ if (!subtype.isName() || subtype.getName() != "/Type0")
1122
+ continue;
1123
+
1124
+ // skip already-subset Type0 fonts
1125
+ if (isAlreadySubset(fontObj))
1126
+ continue;
1127
+
1128
+ auto descendants = fontObj.getKey("/DescendantFonts");
1129
+ if (!descendants.isArray() || descendants.getArrayNItems() < 1)
1130
+ continue;
1131
+
1132
+ auto cidFont = descendants.getArrayItem(0);
1133
+ if (!cidFont.isDictionary())
1134
+ continue;
1135
+
1136
+ // skip already-subset CID fonts — their width tables may have
1137
+ // custom CID mappings our usage collector doesn't fully capture
1138
+ if (isAlreadySubset(cidFont))
1139
+ continue;
1140
+
1141
+ auto w = cidFont.getKey("/W");
1142
+ if (!w.isArray() || w.getArrayNItems() == 0)
1143
+ continue;
1144
+
1145
+ // parse /W into CID → width value map
1146
+ std::map<int, QPDFObjectHandle> cidWidths;
1147
+ int n = w.getArrayNItems();
1148
+ int i = 0;
1149
+ while (i < n) {
1150
+ auto first = w.getArrayItem(i);
1151
+ if (!first.isInteger()) {
1152
+ ++i;
1153
+ continue;
1154
+ }
1155
+ int cidStart = static_cast<int>(first.getIntValue());
1156
+ ++i;
1157
+ if (i >= n)
1158
+ break;
1159
+
1160
+ auto second = w.getArrayItem(i);
1161
+ if (second.isArray()) {
1162
+ // format: cidStart [w1 w2 w3 ...]
1163
+ for (int j = 0; j < second.getArrayNItems(); ++j)
1164
+ cidWidths[cidStart + j] = second.getArrayItem(j);
1165
+ ++i;
1166
+ } else if (second.isInteger()) {
1167
+ // format: cidStart cidEnd sameWidth
1168
+ int cidEnd = static_cast<int>(second.getIntValue());
1169
+ ++i;
1170
+ if (i >= n)
1171
+ break;
1172
+ auto width = w.getArrayItem(i);
1173
+ for (int cid = cidStart; cid <= cidEnd; ++cid)
1174
+ cidWidths[cid] = width;
1175
+ ++i;
1176
+ } else {
1177
+ ++i;
1178
+ }
1179
+ }
1180
+
1181
+ // rebuild /W with only used CIDs grouped by consecutive runs
1182
+ std::vector<int> sortedUsed;
1183
+ for (uint16_t c : usedCodes) {
1184
+ if (cidWidths.count(static_cast<int>(c)))
1185
+ sortedUsed.push_back(static_cast<int>(c));
1186
+ }
1187
+ std::sort(sortedUsed.begin(), sortedUsed.end());
1188
+
1189
+ auto newW = QPDFObjectHandle::newArray();
1190
+ size_t idx = 0;
1191
+ while (idx < sortedUsed.size()) {
1192
+ int start = sortedUsed[idx];
1193
+ auto widths = QPDFObjectHandle::newArray();
1194
+ widths.appendItem(cidWidths[start]);
1195
+ ++idx;
1196
+ while (idx < sortedUsed.size() &&
1197
+ sortedUsed[idx] == sortedUsed[idx - 1] + 1) {
1198
+ widths.appendItem(cidWidths[sortedUsed[idx]]);
1199
+ ++idx;
1200
+ }
1201
+ newW.appendItem(QPDFObjectHandle::newInteger(start));
1202
+ newW.appendItem(widths);
1203
+ }
1204
+
1205
+ // only replace if the new /W has fewer entries
1206
+ if (newW.getArrayNItems() < w.getArrayNItems())
1207
+ cidFont.replaceKey("/W", newW);
1208
+ }
1209
+ }