reamkit 1.26.0 → 1.28.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +14 -8
- package/dist/esm/core/converter/ream.d.ts +0 -9
- package/dist/esm/core/converter/ream.js +0 -1
- package/dist/esm/core/document-model/types.d.ts +5 -3
- package/dist/esm/core/font/index.d.ts +1 -0
- package/dist/esm/core/font/ligatures.d.ts +17 -0
- package/dist/esm/core/font/ligatures.js +49 -0
- package/dist/esm/core/font/ttf-parser.d.ts +2 -1
- package/dist/esm/core/font/ttf-parser.js +11 -2
- package/dist/esm/core/fonts/remote-fonts.d.ts +1 -1
- package/dist/esm/core/fonts/remote-fonts.js +47 -2
- package/dist/esm/core/fonts/scripts.js +10 -5
- package/dist/esm/excel/header-footer.js +53 -8
- package/dist/esm/layout/styled-layout.js +53 -3
- package/dist/esm/pdf/cid-font.js +35 -8
- package/dist/esm/pdf-reader/annot-draw.d.ts +65 -0
- package/dist/esm/pdf-reader/annot-draw.js +487 -0
- package/dist/esm/pdf-reader/annots.d.ts +0 -12
- package/dist/esm/pdf-reader/annots.js +30 -4
- package/dist/esm/pdf-reader/ccitt.d.ts +20 -3
- package/dist/esm/pdf-reader/ccitt.js +102 -6
- package/dist/esm/pdf-reader/cff-outline.d.ts +36 -0
- package/dist/esm/pdf-reader/cff-outline.js +1122 -0
- package/dist/esm/pdf-reader/cie-color.d.ts +33 -0
- package/dist/esm/pdf-reader/cie-color.js +112 -0
- package/dist/esm/pdf-reader/cmap.js +5 -2
- package/dist/esm/pdf-reader/content.d.ts +123 -3
- package/dist/esm/pdf-reader/content.js +232 -54
- package/dist/esm/pdf-reader/dingbats.d.ts +11 -0
- package/dist/esm/pdf-reader/dingbats.js +1033 -0
- package/dist/esm/pdf-reader/display.d.ts +1 -1
- package/dist/esm/pdf-reader/display.js +61 -6
- package/dist/esm/pdf-reader/document.d.ts +6 -0
- package/dist/esm/pdf-reader/document.js +55 -10
- package/dist/esm/pdf-reader/embedded-fonts.d.ts +23 -3
- package/dist/esm/pdf-reader/embedded-fonts.js +36 -4
- package/dist/esm/pdf-reader/encodings.d.ts +25 -0
- package/dist/esm/pdf-reader/encodings.js +110 -0
- package/dist/esm/pdf-reader/flow-build.d.ts +19 -5
- package/dist/esm/pdf-reader/flow-build.js +111 -11
- package/dist/esm/pdf-reader/font.js +403 -21
- package/dist/esm/pdf-reader/function.d.ts +16 -0
- package/dist/esm/pdf-reader/function.js +414 -0
- package/dist/esm/pdf-reader/glyf-outline.d.ts +43 -0
- package/dist/esm/pdf-reader/glyf-outline.js +351 -0
- package/dist/esm/pdf-reader/glyph-names.js +20 -0
- package/dist/esm/pdf-reader/icc.d.ts +10 -0
- package/dist/esm/pdf-reader/icc.js +210 -0
- package/dist/esm/pdf-reader/image-decode.d.ts +9 -4
- package/dist/esm/pdf-reader/image-decode.js +274 -72
- package/dist/esm/pdf-reader/images.d.ts +18 -0
- package/dist/esm/pdf-reader/images.js +154 -11
- package/dist/esm/pdf-reader/jbig2.d.ts +23 -0
- package/dist/esm/pdf-reader/jbig2.js +126 -32
- package/dist/esm/pdf-reader/layout.d.ts +32 -0
- package/dist/esm/pdf-reader/layout.js +1316 -64
- package/dist/esm/pdf-reader/lexer.d.ts +2 -0
- package/dist/esm/pdf-reader/lexer.js +4 -0
- package/dist/esm/pdf-reader/math-rows.d.ts +23 -0
- package/dist/esm/pdf-reader/math-rows.js +198 -0
- package/dist/esm/pdf-reader/optional-content.d.ts +36 -0
- package/dist/esm/pdf-reader/optional-content.js +93 -0
- package/dist/esm/pdf-reader/predefined-cmap.d.ts +21 -0
- package/dist/esm/pdf-reader/predefined-cmap.js +102 -0
- package/dist/esm/pdf-reader/reader.d.ts +6 -6
- package/dist/esm/pdf-reader/reader.js +102 -16
- package/dist/esm/pdf-reader/shading.d.ts +139 -8
- package/dist/esm/pdf-reader/shading.js +309 -39
- package/dist/esm/pdf-reader/standard-metrics.d.ts +8 -0
- package/dist/esm/pdf-reader/standard-metrics.js +18 -0
- package/dist/esm/pdf-reader/standard-widths.d.ts +20 -0
- package/dist/esm/pdf-reader/standard-widths.js +62 -0
- package/dist/esm/pdf-reader/stream-filters.d.ts +6 -0
- package/dist/esm/pdf-reader/stream-filters.js +67 -0
- package/dist/esm/pdf-reader/tagged.js +204 -32
- package/dist/esm/pdf-reader/text-rules.d.ts +16 -0
- package/dist/esm/pdf-reader/text-rules.js +185 -0
- package/dist/esm/pdf-reader/text.js +157 -4
- package/dist/esm/pdf-reader/type1-outline.d.ts +21 -0
- package/dist/esm/pdf-reader/type1-outline.js +576 -0
- package/dist/esm/pdf-reader/vector.d.ts +5 -0
- package/dist/esm/pdf-reader/vector.js +79 -27
- package/dist/esm/word/document-parser.js +4 -1
- package/dist/esm/word/docx-writer.js +74 -11
- package/package.json +1 -1
|
@@ -1,9 +1,19 @@
|
|
|
1
|
+
import { resolveFamilyStyle } from "../core/fonts/remote-fonts.js";
|
|
1
2
|
import { parseTtf } from "../core/font/ttf-parser.js";
|
|
2
3
|
import { PDF_NULL, PdfName, PdfStream } from "../pdf/objects.js";
|
|
3
|
-
import { embeddedFontName } from "./embedded-fonts.js";
|
|
4
4
|
import { parseToUnicodeCMap } from "./cmap.js";
|
|
5
|
+
import { decodePredefined, predefinedCMap, splitPredefined } from "./predefined-cmap.js";
|
|
6
|
+
import { isZapfDingbats, zapfDingbatsChar } from "./dingbats.js";
|
|
5
7
|
import { textForGlyphName } from "./glyph-names.js";
|
|
8
|
+
import { cffCidToGid, cffNameToGid, cffOutlineSource, openTypeCff } from "./cff-outline.js";
|
|
9
|
+
import { type1Font } from "./type1-outline.js";
|
|
10
|
+
import { baseEncodingTable, isStandardLatinFace, standardEncodingTable } from "./encodings.js";
|
|
11
|
+
import { outlineSource, postGlyphNames } from "./glyf-outline.js";
|
|
12
|
+
import { standardFace, standardWidth } from "./standard-widths.js";
|
|
13
|
+
import { embeddedFontName, hasLiftableProgram } from "./embedded-fonts.js";
|
|
6
14
|
//#region src/pdf-reader/font.ts
|
|
15
|
+
/** §9.10.2 — what a composite code nothing can answer for comes to. */
|
|
16
|
+
var UNANSWERABLE = "�";
|
|
7
17
|
/**
|
|
8
18
|
* Build a {@link ContentFont} (the interpreter's decode + advance hooks) from a
|
|
9
19
|
* `/Font` dictionary (E-PDF EP2). Unicode comes from the `/ToUnicode` CMap;
|
|
@@ -20,30 +30,90 @@ function buildContentFont(file, fontDict) {
|
|
|
20
30
|
const isType0 = asName(file.resolve(fontDict.get("Subtype") ?? PDF_NULL)) === "Type0";
|
|
21
31
|
let toUnicode = /* @__PURE__ */ new Map();
|
|
22
32
|
let codeBytes = isType0 ? 2 : 1;
|
|
33
|
+
const encodingName = asName(file.resolve(fontDict.get("Encoding") ?? PDF_NULL));
|
|
34
|
+
const encodingNamed = isType0 && encodingName.length > 0;
|
|
35
|
+
const named = isType0 ? predefinedCMap(encodingName) : void 0;
|
|
23
36
|
const tu = file.resolve(fontDict.get("ToUnicode") ?? PDF_NULL);
|
|
24
37
|
if (tu instanceof PdfStream) {
|
|
25
38
|
const parsed = parseToUnicodeCMap(file.streamData(tu));
|
|
26
39
|
toUnicode = parsed.map;
|
|
27
|
-
if (isType0) codeBytes = parsed.codeBytes;
|
|
40
|
+
if (isType0 && !encodingNamed) codeBytes = parsed.codeBytes;
|
|
28
41
|
}
|
|
29
42
|
const fromProgram = isType0 && toUnicode.size === 0 ? embeddedCmap(file, fontDict) : void 0;
|
|
30
|
-
|
|
43
|
+
let programFallback = null;
|
|
44
|
+
const fromProgramFor = (code) => {
|
|
45
|
+
if (!isType0 || toUnicode.size === 0) return void 0;
|
|
46
|
+
programFallback ??= embeddedCmap(file, fontDict);
|
|
47
|
+
return programFallback?.get(code);
|
|
48
|
+
};
|
|
49
|
+
const glyphs = isType0 ? void 0 : simpleGlyphs(file, fontDict);
|
|
50
|
+
const fromNames = !isType0 ? namedGlyphs(file, fontDict, toUnicode.size > 0, {
|
|
51
|
+
draws: (name) => glyphs?.byName(name) !== void 0,
|
|
52
|
+
blank: (name) => glyphs?.blank(name) === true
|
|
53
|
+
}) : void 0;
|
|
54
|
+
const glyphNames = isType0 ? /* @__PURE__ */ new Map() : differences(file, fontDict);
|
|
31
55
|
const unicode = fromProgram ?? (toUnicode.size > 0 ? toUnicode : fromNames ?? toUnicode);
|
|
32
56
|
const bytesPerCode = codeBytes;
|
|
57
|
+
const indices = !isType0 && glyphs?.indexed === true && fromNames === void 0 && toUnicode.size === 0 && glyphNames.size === 0;
|
|
58
|
+
const dingbats = !isType0 && isZapfDingbats(asName(file.resolve(fontDict.get("BaseFont") ?? PDF_NULL)));
|
|
59
|
+
const baseNames = isType0 ? void 0 : baseEncoding(file, fontDict);
|
|
60
|
+
const fromBase = /* @__PURE__ */ new Map();
|
|
61
|
+
for (const [code, glyph] of baseNames ?? []) {
|
|
62
|
+
const text = textForGlyphName(glyph);
|
|
63
|
+
if (text !== void 0) fromBase.set(code, text);
|
|
64
|
+
}
|
|
65
|
+
const namesOf = new Map([...baseNames ?? [], ...glyphNames]);
|
|
33
66
|
const style = faceStyle(file, fontDict, isType0);
|
|
34
|
-
const name =
|
|
67
|
+
const name = runFontName(file, fontDict, isType0);
|
|
35
68
|
const type3 = asName(file.resolve(fontDict.get("Subtype") ?? PDF_NULL)) === "Type3" ? type3Face(file, fontDict) : void 0;
|
|
36
|
-
const
|
|
69
|
+
const decodeOne = (code) => (indices ? UNANSWERABLE : void 0) ?? unicode.get(code) ?? fromProgramFor(code) ?? (named ? decodePredefined(named, code) : void 0) ?? fromNames?.get(code) ?? (dingbats ? zapfDingbatsChar(code) : void 0) ?? fromBase.get(code) ?? (bytesPerCode === 1 ? latin1(code) : UNANSWERABLE);
|
|
70
|
+
const simple = simpleWidths(file, fontDict, decodeOne);
|
|
37
71
|
const width = isType0 ? cidWidths(file, fontDict) : type3 ? (code) => simple(code) * type3.matrix[0] * 1e3 : simple;
|
|
38
72
|
return {
|
|
39
73
|
bytesPerCode,
|
|
74
|
+
...named ? { splitCodes: (b) => splitPredefined(named, b) } : {},
|
|
75
|
+
...isType0 ? outlineOf(file, fontDict, decodeOne) : simpleOutlineOf(decodeOne, namesOf, glyphs),
|
|
76
|
+
...named?.vertical ? { verticalAdvance: cidVerticalAdvance(file, fontDict) } : {},
|
|
40
77
|
...type3 ? { type3 } : {},
|
|
41
78
|
...name !== void 0 ? { name } : {},
|
|
42
|
-
decode: (codes) => codes.map((c) =>
|
|
79
|
+
decode: (codes) => codes.map((c) => readable(decodeOne(c))).join(""),
|
|
43
80
|
width,
|
|
44
81
|
...style
|
|
45
82
|
};
|
|
46
83
|
}
|
|
84
|
+
/**
|
|
85
|
+
* The character a code stands for when nothing in the font says.
|
|
86
|
+
*
|
|
87
|
+
* Latin-1 is the only guess there is, and it is a good one for a text font —
|
|
88
|
+
* but a C0 control is not a glyph. §9.4.3 shows GLYPHS, and a code falling back
|
|
89
|
+
* to one has landed there by accident: issue11549_reduced.pdf's one line came
|
|
90
|
+
* back as U+0007 through U+0011 and was drawn as six empty boxes over a page
|
|
91
|
+
* that shows nothing at all. A `/ToUnicode` that STATES a control is stating
|
|
92
|
+
* something and is left alone.
|
|
93
|
+
*/
|
|
94
|
+
function latin1(code) {
|
|
95
|
+
return code < 32 || code === 127 ? "�" : String.fromCharCode(code);
|
|
96
|
+
}
|
|
97
|
+
/**
|
|
98
|
+
* What a `/ToUnicode` gives, less what Unicode says is not a character.
|
|
99
|
+
*
|
|
100
|
+
* `U+FFFE` and `U+FFFF` are noncharacters and the `U+FDD0`–`U+FDEF` block with
|
|
101
|
+
* them; a lone surrogate is half of a pair that never came. A producer that
|
|
102
|
+
* maps its glyphs to any of these has said "no text here" in the only way the
|
|
103
|
+
* format lets it, and carrying them on writes bytes no reader can show —
|
|
104
|
+
* arial_unicode_ab_cidfont.pdf maps its four Arabic letters to `U+FFFF` and the
|
|
105
|
+
* page came back holding four of them. They become `U+FFFD`, which the
|
|
106
|
+
* reconstruction counts and reports rather than passing along.
|
|
107
|
+
*/
|
|
108
|
+
function readable(text) {
|
|
109
|
+
let out = "";
|
|
110
|
+
for (const ch of text) {
|
|
111
|
+
const cp = ch.codePointAt(0) ?? 0;
|
|
112
|
+
const noncharacter = (cp & 65534) === 65534 || cp >= 64976 && cp <= 65007 || cp === 0 || cp >= SURROGATE_FIRST && cp <= SURROGATE_LAST;
|
|
113
|
+
out += noncharacter ? "�" : ch;
|
|
114
|
+
}
|
|
115
|
+
return out;
|
|
116
|
+
}
|
|
47
117
|
/** The last Unicode code point in the BMP, and the surrogate block inside it. */
|
|
48
118
|
var BMP_END = 65535;
|
|
49
119
|
var SURROGATE_FIRST = 55296;
|
|
@@ -94,6 +164,192 @@ function embeddedCmap(file, fontDict) {
|
|
|
94
164
|
});
|
|
95
165
|
return out.size > 0 ? out : void 0;
|
|
96
166
|
}
|
|
167
|
+
/**
|
|
168
|
+
* §9.6.6 — the outline a code draws, for a code that stands for no character.
|
|
169
|
+
*
|
|
170
|
+
* The glyph is there even when the character is not: `/Encoding /Identity-H`
|
|
171
|
+
* makes the code a CID, `/CIDToGIDMap` turns that into a glyph index, and the
|
|
172
|
+
* embedded program holds the contours. complex_ttf_font.pdf is eight lines of
|
|
173
|
+
* Arabic in a subset with no `cmap` and no `/ToUnicode`, and every one of them
|
|
174
|
+
* was dropped.
|
|
175
|
+
*
|
|
176
|
+
* Deliberately NOT a fallback for text: a code the font CAN answer for is set
|
|
177
|
+
* as type, and only the unanswerable ones are traced.
|
|
178
|
+
*
|
|
179
|
+
* @param file The owning file.
|
|
180
|
+
* @param fontDict The Type 0 font dictionary.
|
|
181
|
+
* @param decodeOne What one code comes to, to tell the two cases apart.
|
|
182
|
+
* @returns The `outline` field of a {@link ContentFont}, or nothing where the
|
|
183
|
+
* program carries no outlines this reads.
|
|
184
|
+
*/
|
|
185
|
+
function outlineOf(file, fontDict, decodeOne) {
|
|
186
|
+
const cidFont = descendantFont(file, fontDict);
|
|
187
|
+
const descriptor = file.resolve(cidFont.get("FontDescriptor") ?? PDF_NULL);
|
|
188
|
+
if (!(descriptor instanceof Map)) return {};
|
|
189
|
+
const truetype = file.resolve(descriptor.get("FontFile2") ?? PDF_NULL);
|
|
190
|
+
const compact = file.resolve(descriptor.get("FontFile3") ?? PDF_NULL);
|
|
191
|
+
const program = truetype instanceof PdfStream ? truetype : compact;
|
|
192
|
+
if (!(program instanceof PdfStream)) return {};
|
|
193
|
+
let source;
|
|
194
|
+
let charsetCids;
|
|
195
|
+
try {
|
|
196
|
+
const bytes = file.streamData(program);
|
|
197
|
+
source = outlineSource(bytes);
|
|
198
|
+
if (!source) {
|
|
199
|
+
const cff = openTypeCff(bytes) ?? bytes;
|
|
200
|
+
source = cffOutlineSource(cff);
|
|
201
|
+
charsetCids = source ? cffCidToGid(cff) : void 0;
|
|
202
|
+
}
|
|
203
|
+
} catch {
|
|
204
|
+
return {};
|
|
205
|
+
}
|
|
206
|
+
if (!source) return {};
|
|
207
|
+
const cidToGid = readCidToGid(file, cidFont);
|
|
208
|
+
return { outline: {
|
|
209
|
+
matrix: [
|
|
210
|
+
1,
|
|
211
|
+
0,
|
|
212
|
+
0,
|
|
213
|
+
1,
|
|
214
|
+
0,
|
|
215
|
+
0
|
|
216
|
+
],
|
|
217
|
+
path: (code) => {
|
|
218
|
+
if (readable(decodeOne(code)) !== UNANSWERABLE) return void 0;
|
|
219
|
+
const mapped = cidToGid ? cidToGid[code] ?? 0 : code;
|
|
220
|
+
const gid = charsetCids ? charsetCids.get(mapped) ?? mapped : mapped;
|
|
221
|
+
return source.path(gid);
|
|
222
|
+
}
|
|
223
|
+
} };
|
|
224
|
+
}
|
|
225
|
+
/**
|
|
226
|
+
* §9.6.6 — the outline a SIMPLE font's code draws, for a code that stands for
|
|
227
|
+
* no character.
|
|
228
|
+
*
|
|
229
|
+
* A simple font addresses its glyphs by NAME: `/Differences` (or the program's
|
|
230
|
+
* own `/Encoding`) says code 2 is `g18`, and the program says what `g18` looks
|
|
231
|
+
* like. Where that name is not a character there is nothing to write and there
|
|
232
|
+
* is still something to draw — TAMReview.pdf sets most of its body in a Cambria
|
|
233
|
+
* subset whose glyphs are named `g18`, `g152`, `g135`, and seven thousand of
|
|
234
|
+
* its nine thousand characters were dropped.
|
|
235
|
+
*
|
|
236
|
+
* @param file The owning file.
|
|
237
|
+
* @param fontDict The simple font's dictionary.
|
|
238
|
+
* @param decodeOne What one code comes to, to tell the two cases apart.
|
|
239
|
+
* @param nameOf The glyph name a code selects, where the font states one.
|
|
240
|
+
* @returns The `outline` field of a {@link ContentFont}, or nothing.
|
|
241
|
+
*/
|
|
242
|
+
function simpleOutlineOf(decodeOne, nameOf, source) {
|
|
243
|
+
if (!source) return {};
|
|
244
|
+
return { outline: {
|
|
245
|
+
matrix: [
|
|
246
|
+
1,
|
|
247
|
+
0,
|
|
248
|
+
0,
|
|
249
|
+
1,
|
|
250
|
+
0,
|
|
251
|
+
0
|
|
252
|
+
],
|
|
253
|
+
path: (code) => {
|
|
254
|
+
if (readable(decodeOne(code)) !== UNANSWERABLE) return void 0;
|
|
255
|
+
const name = nameOf.get(code) ?? source.builtIn?.get(code);
|
|
256
|
+
if (name === void 0) return source.byIndex?.(code);
|
|
257
|
+
return source.byName(name);
|
|
258
|
+
}
|
|
259
|
+
} };
|
|
260
|
+
}
|
|
261
|
+
/**
|
|
262
|
+
* §9.6.6 — the program a SIMPLE font embeds, ready to draw a glyph BY NAME.
|
|
263
|
+
*
|
|
264
|
+
* All three formats appear here: a Type 1 program keys its charstrings by name
|
|
265
|
+
* outright, a CFF says which name each glyph has in its charset, and a TrueType
|
|
266
|
+
* says nothing at all — but a subsetter that renames glyphs `g24` has written
|
|
267
|
+
* the glyph's index into the name, which is the only handle such a program
|
|
268
|
+
* gives.
|
|
269
|
+
*
|
|
270
|
+
* @param file The owning file.
|
|
271
|
+
* @param fontDict The simple font's dictionary.
|
|
272
|
+
* @returns How to draw one of its glyphs, or `undefined` where nothing here
|
|
273
|
+
* can read the program.
|
|
274
|
+
*/
|
|
275
|
+
function simpleGlyphs(file, fontDict) {
|
|
276
|
+
const descriptor = file.resolve(fontDict.get("FontDescriptor") ?? PDF_NULL);
|
|
277
|
+
if (!(descriptor instanceof Map)) return void 0;
|
|
278
|
+
const typeOne = file.resolve(descriptor.get("FontFile") ?? PDF_NULL);
|
|
279
|
+
const truetype = file.resolve(descriptor.get("FontFile2") ?? PDF_NULL);
|
|
280
|
+
const compact = file.resolve(descriptor.get("FontFile3") ?? PDF_NULL);
|
|
281
|
+
try {
|
|
282
|
+
if (typeOne instanceof PdfStream) {
|
|
283
|
+
const face = type1Font(file.streamData(typeOne));
|
|
284
|
+
if (!face) return void 0;
|
|
285
|
+
return {
|
|
286
|
+
byName: face.path,
|
|
287
|
+
blank: (name) => face.has(name) && face.path(name) === void 0,
|
|
288
|
+
...face.encoding ? { builtIn: face.encoding } : {}
|
|
289
|
+
};
|
|
290
|
+
}
|
|
291
|
+
const stream = compact instanceof PdfStream ? compact : truetype;
|
|
292
|
+
if (!(stream instanceof PdfStream)) return void 0;
|
|
293
|
+
const bytes = file.streamData(stream);
|
|
294
|
+
const glyf = outlineSource(bytes);
|
|
295
|
+
if (glyf) {
|
|
296
|
+
const named = postGlyphNames(bytes);
|
|
297
|
+
const gidOf = (name) => {
|
|
298
|
+
const gid = named?.get(name) ?? numberedGlyph(name);
|
|
299
|
+
return gid !== void 0 && gid < glyf.count ? gid : void 0;
|
|
300
|
+
};
|
|
301
|
+
return {
|
|
302
|
+
byName: (name) => {
|
|
303
|
+
const gid = gidOf(name);
|
|
304
|
+
return gid === void 0 ? void 0 : glyf.path(gid);
|
|
305
|
+
},
|
|
306
|
+
blank: (name) => {
|
|
307
|
+
const gid = gidOf(name);
|
|
308
|
+
return gid !== void 0 && glyf.path(gid) === void 0;
|
|
309
|
+
},
|
|
310
|
+
byIndex: (gid) => gid < glyf.count ? glyf.path(gid) : void 0,
|
|
311
|
+
indexed: !glyf.cmap
|
|
312
|
+
};
|
|
313
|
+
}
|
|
314
|
+
const cff = openTypeCff(bytes) ?? bytes;
|
|
315
|
+
const outlines = cffOutlineSource(cff);
|
|
316
|
+
if (!outlines) return void 0;
|
|
317
|
+
const names = cffNameToGid(cff);
|
|
318
|
+
const gidOf = (name) => {
|
|
319
|
+
const gid = names?.get(name) ?? numberedGlyph(name);
|
|
320
|
+
return gid !== void 0 && gid < outlines.count ? gid : void 0;
|
|
321
|
+
};
|
|
322
|
+
return {
|
|
323
|
+
byName: (name) => {
|
|
324
|
+
const gid = gidOf(name);
|
|
325
|
+
return gid === void 0 ? void 0 : outlines.path(gid);
|
|
326
|
+
},
|
|
327
|
+
blank: (name) => {
|
|
328
|
+
const gid = gidOf(name);
|
|
329
|
+
return gid !== void 0 && outlines.path(gid) === void 0;
|
|
330
|
+
}
|
|
331
|
+
};
|
|
332
|
+
} catch {
|
|
333
|
+
return;
|
|
334
|
+
}
|
|
335
|
+
}
|
|
336
|
+
/**
|
|
337
|
+
* The glyph index a name carries, for the names that carry one.
|
|
338
|
+
*
|
|
339
|
+
* A subsetter that drops a font's `cmap` renames its glyphs after their
|
|
340
|
+
* INDEX — `g24`, `glyph24`, `index24`, `cid24` — and that name is then the only
|
|
341
|
+
* way back to the outline. bug1151216.pdf names them `g24`, `g381`, `g3`, and
|
|
342
|
+
* its three lines of prices are drawn from nothing else.
|
|
343
|
+
*/
|
|
344
|
+
function numberedGlyph(name) {
|
|
345
|
+
const hex = /^g([0-9a-f]{4})$/u.exec(name);
|
|
346
|
+
if (hex) return Number.parseInt(hex[1], 16);
|
|
347
|
+
const m = /^(?:g|glyph|index|cid|G)(\d+)$/u.exec(name);
|
|
348
|
+
const n = m ? Number(m[1]) : NaN;
|
|
349
|
+
return Number.isFinite(n) && n >= 0 && n < MAX_GLYPH_INDEX ? n : void 0;
|
|
350
|
+
}
|
|
351
|
+
/** No font holds more glyphs than this; a bigger number is not an index. */
|
|
352
|
+
var MAX_GLYPH_INDEX = 65536;
|
|
97
353
|
/** §9.7.4.2 `/CIDToGIDMap` — a stream of two-byte glyph indices, CID by CID. */
|
|
98
354
|
function readCidToGid(file, cidFont) {
|
|
99
355
|
const map = file.resolve(cidFont.get("CIDToGIDMap") ?? PDF_NULL);
|
|
@@ -158,15 +414,84 @@ function fontMatrix(file, fontDict) {
|
|
|
158
414
|
* Returns `undefined` where the font names nothing, so the caller keeps its
|
|
159
415
|
* Latin-1 reading rather than replacing it with an empty map.
|
|
160
416
|
*/
|
|
161
|
-
function namedGlyphs(file, fontDict) {
|
|
417
|
+
function namedGlyphs(file, fontDict, statesUnicode, glyph) {
|
|
162
418
|
const names = differences(file, fontDict);
|
|
163
419
|
if (names.size === 0) return void 0;
|
|
164
420
|
const out = /* @__PURE__ */ new Map();
|
|
165
421
|
for (const [code, name] of names) {
|
|
166
422
|
const text = textForGlyphName(name);
|
|
167
|
-
if (text !== void 0)
|
|
423
|
+
if (text !== void 0) {
|
|
424
|
+
out.set(code, text);
|
|
425
|
+
continue;
|
|
426
|
+
}
|
|
427
|
+
if (glyph.blank(name)) {
|
|
428
|
+
out.set(code, " ");
|
|
429
|
+
continue;
|
|
430
|
+
}
|
|
431
|
+
if (!statesUnicode && glyph.draws(name)) out.set(code, UNANSWERABLE);
|
|
168
432
|
}
|
|
169
|
-
|
|
433
|
+
if (out.size > 0) return out;
|
|
434
|
+
if (asName(file.resolve(fontDict.get("Subtype") ?? PDF_NULL)) !== "Type3") return void 0;
|
|
435
|
+
const unreadable = /* @__PURE__ */ new Map();
|
|
436
|
+
for (const code of names.keys()) unreadable.set(code, "�");
|
|
437
|
+
return unreadable;
|
|
438
|
+
}
|
|
439
|
+
/**
|
|
440
|
+
* §9.7.4.3 `/DW2` and `/W2` — how far the pen drops for one glyph, in 1000-unit
|
|
441
|
+
* text space.
|
|
442
|
+
*
|
|
443
|
+
* `/DW2`'s default is `[880 -1000]`: the vertical origin sits 880 above the
|
|
444
|
+
* horizontal one and the displacement is a full em DOWN. Only the displacement
|
|
445
|
+
* is wanted here — the origin shifts where the glyph is drawn, which a reader
|
|
446
|
+
* re-setting the words in another face does not reproduce anyway.
|
|
447
|
+
*/
|
|
448
|
+
function cidVerticalAdvance(file, fontDict) {
|
|
449
|
+
const cid = descendantFont(file, fontDict);
|
|
450
|
+
const dw2 = file.resolve(cid.get("DW2") ?? PDF_NULL);
|
|
451
|
+
const stated = Array.isArray(dw2) && typeof file.resolve(dw2[1] ?? PDF_NULL) === "number" ? file.resolve(dw2[1]) : DEFAULT_DW2_DISPLACEMENT;
|
|
452
|
+
const perCode = /* @__PURE__ */ new Map();
|
|
453
|
+
const w2 = file.resolve(cid.get("W2") ?? PDF_NULL);
|
|
454
|
+
if (Array.isArray(w2)) for (let i = 0; i < w2.length;) {
|
|
455
|
+
const first = file.resolve(w2[i] ?? PDF_NULL);
|
|
456
|
+
const next = file.resolve(w2[i + 1] ?? PDF_NULL);
|
|
457
|
+
if (typeof first !== "number") break;
|
|
458
|
+
if (Array.isArray(next)) {
|
|
459
|
+
for (let k = 0; k + 2 < next.length; k += 3) {
|
|
460
|
+
const v = file.resolve(next[k] ?? PDF_NULL);
|
|
461
|
+
if (typeof v === "number") perCode.set(first + k / 3, v);
|
|
462
|
+
}
|
|
463
|
+
i += 2;
|
|
464
|
+
} else if (typeof next === "number" && typeof file.resolve(w2[i + 2] ?? PDF_NULL) === "number") {
|
|
465
|
+
const v = file.resolve(w2[i + 2]);
|
|
466
|
+
for (let c = first; c <= next && c - first < MAX_W2_RANGE; c++) perCode.set(c, v);
|
|
467
|
+
i += 5;
|
|
468
|
+
} else break;
|
|
469
|
+
}
|
|
470
|
+
return (code) => perCode.get(code) ?? stated;
|
|
471
|
+
}
|
|
472
|
+
/** §9.7.4.3 — `/DW2`'s default displacement: one em down the page. */
|
|
473
|
+
var DEFAULT_DW2_DISPLACEMENT = -1e3;
|
|
474
|
+
/** A `/W2` range wider than this is not walked out code by code. */
|
|
475
|
+
var MAX_W2_RANGE = 65536;
|
|
476
|
+
/**
|
|
477
|
+
* Annex D.2 — code → text for the encoding a simple font is read through, under
|
|
478
|
+
* whatever `/Differences` restates.
|
|
479
|
+
*
|
|
480
|
+
* `/Encoding` either names one of the base encodings, or is a dictionary that
|
|
481
|
+
* may name one in `/BaseEncoding`, or is absent — and absent means the encoding
|
|
482
|
+
* built into the FACE. Only the standard Latin faces can be answered for there:
|
|
483
|
+
* a file that names them embeds no program, and what they are is known
|
|
484
|
+
* (§9.6.2.2). An embedded font's built-in encoding lives in the program and a
|
|
485
|
+
* substituted face has none worth guessing at, so both keep the Latin-1
|
|
486
|
+
* reading, which is what the codes of such a file nearly always are.
|
|
487
|
+
*
|
|
488
|
+
* @returns Code → glyph NAME, or `undefined` where nothing better than Latin-1
|
|
489
|
+
* is known.
|
|
490
|
+
*/
|
|
491
|
+
function baseEncoding(file, fontDict) {
|
|
492
|
+
const encoding = file.resolve(fontDict.get("Encoding") ?? PDF_NULL);
|
|
493
|
+
const named = encoding instanceof PdfName ? encoding.value : encoding instanceof Map ? asName(file.resolve(encoding.get("BaseEncoding") ?? PDF_NULL)) : "";
|
|
494
|
+
return named.length > 0 ? baseEncodingTable(named) : isStandardLatinFace(asName(file.resolve(fontDict.get("BaseFont") ?? PDF_NULL))) ? standardEncodingTable() : void 0;
|
|
170
495
|
}
|
|
171
496
|
/** §9.6.6.1 `/Encoding` `/Differences` — code → glyph name, as the array runs. */
|
|
172
497
|
function differences(file, fontDict) {
|
|
@@ -183,9 +508,47 @@ function differences(file, fontDict) {
|
|
|
183
508
|
}
|
|
184
509
|
return out;
|
|
185
510
|
}
|
|
511
|
+
/**
|
|
512
|
+
* §9.8.2 — the family a run states: the file's own name for the face, plus what
|
|
513
|
+
* the file says the face IS.
|
|
514
|
+
*
|
|
515
|
+
* The name is a hint for the substitution and nothing more, unless the reader
|
|
516
|
+
* can lift the program itself (§9.9 `/FontFile2`) — then it is the key the face
|
|
517
|
+
* is filed under and must be left exactly as it is. Everything else is
|
|
518
|
+
* substituted by name, and TeX and PostScript producers embed Type 1 and CFF
|
|
519
|
+
* programs under names no table knows: `NimbusRomNo9L-Regu`, `LMRoman10`,
|
|
520
|
+
* `CMR10`, `stonesans`. Every one of those pages came back set in a grotesque
|
|
521
|
+
* — the loudest single difference between our page and a serif document's.
|
|
522
|
+
*
|
|
523
|
+
* The descriptor states the class outright, so where the name says nothing the
|
|
524
|
+
* flags do: bit 1 is FixedPitch and bit 2 Serif.
|
|
525
|
+
*
|
|
526
|
+
* @param file The document.
|
|
527
|
+
* @param fontDict The font dictionary.
|
|
528
|
+
* @param isType0 Whether it is a composite font, whose descendant owns the
|
|
529
|
+
* descriptor.
|
|
530
|
+
* @returns The family name for the run, or `undefined` where the font has none.
|
|
531
|
+
*/
|
|
532
|
+
function runFontName(file, fontDict, isType0) {
|
|
533
|
+
const name = embeddedFontName(file, fontDict);
|
|
534
|
+
if (name === void 0 || hasLiftableProgram(file, fontDict)) return name;
|
|
535
|
+
if (resolveFamilyStyle(name).key !== "arimo") return name;
|
|
536
|
+
const owner = isType0 ? descendantFont(file, fontDict) : fontDict;
|
|
537
|
+
const descriptor = file.resolve(owner.get("FontDescriptor") ?? PDF_NULL);
|
|
538
|
+
if (!(descriptor instanceof Map)) return name;
|
|
539
|
+
const flags = asNumber(file.resolve(descriptor.get("Flags") ?? PDF_NULL), 0);
|
|
540
|
+
if ((flags & FLAG_FIXED_PITCH) !== 0) return `${name} monospace`;
|
|
541
|
+
if ((flags & FLAG_SERIF) !== 0) return `${name} serif`;
|
|
542
|
+
return name;
|
|
543
|
+
}
|
|
544
|
+
/** §9.8.2 `/Flags` — bit 1 is FixedPitch, bit 2 Serif (bits numbered from 1). */
|
|
545
|
+
var FLAG_FIXED_PITCH = 1;
|
|
546
|
+
var FLAG_SERIF = 2;
|
|
186
547
|
/** §9.8.2 `/Flags` — bit 7 is Italic, bit 19 ForceBold (bits numbered from 1). */
|
|
187
548
|
var FLAG_ITALIC = 64;
|
|
188
549
|
var FLAG_FORCE_BOLD = 1 << 18;
|
|
550
|
+
/** §9.8.1 `/FontWeight` — the lightest a face may state; below it is no weight. */
|
|
551
|
+
var LIGHTEST_WEIGHT = 100;
|
|
189
552
|
/** §9.8.1 `/FontWeight` — 400 is normal, 700 bold; 600 is where "bold" begins. */
|
|
190
553
|
var BOLD_WEIGHT = 600;
|
|
191
554
|
/**
|
|
@@ -207,23 +570,38 @@ var BOLD_WEIGHT = 600;
|
|
|
207
570
|
function faceStyle(file, fontDict, isType0) {
|
|
208
571
|
const owner = isType0 ? descendantFont(file, fontDict) : fontDict;
|
|
209
572
|
const descriptor = file.resolve(owner.get("FontDescriptor") ?? PDF_NULL);
|
|
573
|
+
const named = styleFromName(asName(file.resolve(fontDict.get("BaseFont") ?? PDF_NULL)));
|
|
210
574
|
if (descriptor instanceof Map) {
|
|
211
575
|
const flags = asNumber(file.resolve(descriptor.get("Flags") ?? PDF_NULL), 0);
|
|
212
|
-
const
|
|
576
|
+
const weightVal = file.resolve(descriptor.get("FontWeight") ?? PDF_NULL);
|
|
213
577
|
const slant = asNumber(file.resolve(descriptor.get("ItalicAngle") ?? PDF_NULL), 0);
|
|
214
|
-
const bold =
|
|
215
|
-
const italic = slant !== 0 || (flags & FLAG_ITALIC) !== 0;
|
|
578
|
+
const bold = typeof weightVal === "number" && weightVal >= LIGHTEST_WEIGHT ? asNumber(weightVal, 0) >= BOLD_WEIGHT : (flags & FLAG_FORCE_BOLD) !== 0 || named.bold;
|
|
579
|
+
const italic = slant !== 0 || (flags & FLAG_ITALIC) !== 0 || named.italic;
|
|
216
580
|
return {
|
|
217
|
-
...bold ? { bold } : {},
|
|
218
|
-
...italic ? { italic } : {}
|
|
581
|
+
...bold ? { bold: true } : {},
|
|
582
|
+
...italic ? { italic: true } : {}
|
|
219
583
|
};
|
|
220
584
|
}
|
|
221
|
-
const name = asName(file.resolve(fontDict.get("BaseFont") ?? PDF_NULL)).replace(/^[A-Z]{6}\+/u, "");
|
|
222
|
-
const bold = /bold|black|heavy/iu.test(name);
|
|
223
|
-
const italic = /italic|oblique/iu.test(name);
|
|
224
585
|
return {
|
|
225
|
-
...bold ? { bold } : {},
|
|
226
|
-
...italic ? { italic } : {}
|
|
586
|
+
...named.bold ? { bold: true } : {},
|
|
587
|
+
...named.italic ? { italic: true } : {}
|
|
588
|
+
};
|
|
589
|
+
}
|
|
590
|
+
/**
|
|
591
|
+
* §9.6.2.2 — the style a font's NAME states, by the PostScript convention:
|
|
592
|
+
* `Family-Style`, or `Family,Style` as Word writes it.
|
|
593
|
+
*
|
|
594
|
+
* The separator is what makes this safe. A family whose name merely CONTAINS
|
|
595
|
+
* the word — "New Basrah Bold", "Damascus Bold", both real faces in
|
|
596
|
+
* ArabicCIDTrueType.pdf — is not a bold cut of anything, and reading it as one
|
|
597
|
+
* set two lines heavy that no reader sets heavy. `Times-Bold` is.
|
|
598
|
+
*/
|
|
599
|
+
function styleFromName(baseFont) {
|
|
600
|
+
const name = baseFont.replace(/^[A-Z]{6}\+/u, "");
|
|
601
|
+
const style = /[-,]([A-Za-z]+)$/u.exec(name)?.[1] ?? "";
|
|
602
|
+
return {
|
|
603
|
+
bold: /bold|black|heavy|semib|demi/iu.test(style),
|
|
604
|
+
italic: /italic|oblique/iu.test(style)
|
|
227
605
|
};
|
|
228
606
|
}
|
|
229
607
|
/** §9.7.4 — a `/Type0` font's one descendant CIDFont, which owns the descriptor. */
|
|
@@ -232,15 +610,19 @@ function descendantFont(file, fontDict) {
|
|
|
232
610
|
const first = Array.isArray(descFonts) ? file.resolve(descFonts[0] ?? PDF_NULL) : PDF_NULL;
|
|
233
611
|
return first instanceof Map ? first : /* @__PURE__ */ new Map();
|
|
234
612
|
}
|
|
235
|
-
function simpleWidths(file, fontDict) {
|
|
613
|
+
function simpleWidths(file, fontDict, decodeOne) {
|
|
236
614
|
const first = asNumber(file.resolve(fontDict.get("FirstChar") ?? PDF_NULL), 0);
|
|
237
615
|
const widthsVal = file.resolve(fontDict.get("Widths") ?? PDF_NULL);
|
|
238
616
|
const widths = Array.isArray(widthsVal) ? widthsVal : [];
|
|
239
617
|
const descriptor = file.resolve(fontDict.get("FontDescriptor") ?? PDF_NULL);
|
|
240
618
|
const missing = descriptor instanceof Map ? asNumber(file.resolve(descriptor.get("MissingWidth") ?? PDF_NULL), 0) : 0;
|
|
619
|
+
const face = standardFace(asName(file.resolve(fontDict.get("BaseFont") ?? PDF_NULL)));
|
|
241
620
|
return (code) => {
|
|
242
621
|
const w = widths[code - first];
|
|
243
|
-
|
|
622
|
+
if (typeof w === "number") return w;
|
|
623
|
+
const built = face === void 0 ? void 0 : standardWidth(face, code, decodeOne(code));
|
|
624
|
+
if (built !== void 0) return built;
|
|
625
|
+
return missing > 0 ? missing : 500;
|
|
244
626
|
};
|
|
245
627
|
}
|
|
246
628
|
function cidWidths(file, fontDict) {
|
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
import { PdfValue } from '../pdf/objects.js';
|
|
2
|
+
import { PdfFile } from './document.js';
|
|
3
|
+
/** §7.10 — m numbers in, n numbers out. */
|
|
4
|
+
export type PdfFunction = (inputs: ReadonlyArray<number>) => Array<number>;
|
|
5
|
+
/**
|
|
6
|
+
* Read a `/Function` entry into something callable (§7.10).
|
|
7
|
+
*
|
|
8
|
+
* The entry may also be an ARRAY of functions, each giving one output, which is
|
|
9
|
+
* what a `/DeviceN` with a per-colorant transform states; that comes back as
|
|
10
|
+
* one function returning all of them in order.
|
|
11
|
+
*
|
|
12
|
+
* @param file The owning file.
|
|
13
|
+
* @param value The `/Function` (or `/TintTransform`) entry, unresolved.
|
|
14
|
+
* @returns The function, or `undefined` for one this cannot run.
|
|
15
|
+
*/
|
|
16
|
+
export declare function readFunction(file: PdfFile, value: PdfValue | undefined): PdfFunction | undefined;
|