web-doc 0.9.0 → 0.10.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -49,5 +49,19 @@ The release artifact contains or depends on the following principal components.
49
49
  `packages/viewer/src/edit/pptx/`, `packages/viewer/src/edit/docx/` and
50
50
  `packages/viewer/src/edit/ai/` are web-doc's own; no GenOffice code is
51
51
  included.
52
+ - [Adobe Glyph List and Adobe Glyph List For New Fonts](https://github.com/adobe-type-tools/agl-aglfn) — BSD-3-Clause; the glyph names and code points in `packages/viewer/src/edit/pdf/engine/glyph-names.ts` (AGLFN 1.7 and nine AGL 2.0 entries), which give an embedded CFF font a Unicode cmap. The ISOAdobe glyph names in `packages/viewer/src/edit/pdf/engine/cff.ts` are the standard strings of the CFF specification (Adobe Technical Note 5176). Their license:
53
+
54
+ > Copyright 2002-2019 Adobe (http://www.adobe.com/).
55
+
56
+ > Redistribution and use in source and binary forms, with or without modification, are permitted provided that the following conditions are met:
57
+
58
+ > Redistributions of source code must retain the above copyright notice, this list of conditions and the following disclaimer.
59
+
60
+ > Redistributions in binary form must reproduce the above copyright notice, this list of conditions and the following disclaimer in the documentation and/or other materials provided with the distribution.
61
+
62
+ > Neither the name of Adobe nor the names of its contributors may be used to endorse or promote products derived from this software without specific prior written permission.
63
+
64
+ > THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT HOLDER OR CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
65
+
52
66
  - [Noto Sans](https://github.com/notofonts/noto-fonts) and [Noto Sans CJK](https://github.com/notofonts/noto-cjk) subset fonts — SIL Open Font License 1.1. The font manifest, complete OFL text, pinned source commits and SHA-256 hashes are included in `dist/fonts/`.
53
67
  No Microsoft proprietary font or copyleft runtime component is bundled. Transitive notices and license expressions are verified by `npm run licenses`.
@@ -32,7 +32,7 @@ export class OfficeDocumentAdapter {
32
32
  : (await import("../edit/pptx/provider.js")).loadPptxEditEngine(original, context, this.#options.edit ?? {}),
33
33
  createSession: (core, access) => core.format === "docx"
34
34
  ? new DocxSession(core, access)
35
- : new PptxSession(core),
35
+ : new PptxSession(core, access),
36
36
  };
37
37
  formats = [...MODERN_FORMATS, ...LEGACY_FORMATS];
38
38
  #options;
@@ -309,6 +309,7 @@ export class OfficeDocumentAdapter {
309
309
  y: run.shapeY + run.inShapeY,
310
310
  width: run.w,
311
311
  height: run.h,
312
+ shapeOrigin: { x: run.shapeX, y: run.shapeY },
312
313
  ...safeHyperlink(run.hyperlink, (ref) => handle.backend.resolveInternalTarget?.(ref, pageIndex)),
313
314
  direction: textDirection(run.text),
314
315
  }));
@@ -195,6 +195,15 @@ export interface TextRun {
195
195
  * source paragraph.
196
196
  */
197
197
  readonly paragraphId?: string;
198
+ /**
199
+ * PPTX: the top-left corner of the frame of the shape the run was laid out
200
+ * in, in the run coordinate space (slide CSS pixels). Text a shape paints
201
+ * past its frame, wrapped below a short box, still names its shape.
202
+ */
203
+ readonly shapeOrigin?: {
204
+ readonly x: number;
205
+ readonly y: number;
206
+ };
198
207
  }
199
208
  export type HyperlinkTarget = {
200
209
  readonly kind: "external";
@@ -0,0 +1,28 @@
1
+ /** What a browser font is built from. */
2
+ export interface CffFont {
3
+ /** The font's PostScript name, a subset tag included. */
4
+ readonly name: string;
5
+ /** CID-keyed: glyphs are numbered, not named, and only the PDF's CMap says what they draw. */
6
+ readonly cid: boolean;
7
+ /** Glyph names by glyph id; `undefined` where the name is not one this reader knows. */
8
+ readonly glyphNames: readonly (string | undefined)[];
9
+ /** Advance widths by glyph id, in font units. */
10
+ readonly widths: readonly number[];
11
+ readonly bbox: readonly [number, number, number, number];
12
+ readonly unitsPerEm: number;
13
+ }
14
+ /**
15
+ * The code point a glyph name stands for: an Adobe Glyph List name, or
16
+ * `uniXXXX` or `uXXXX[XX]` for one code point. Ligatures of several code
17
+ * points, suffixed variants such as `a.sc`, and noncharacters stand for none.
18
+ */
19
+ export declare function glyphUnicode(name: string | undefined): number | undefined;
20
+ /** Reads a CFF font program, or `undefined` when it is not one this reader can use. */
21
+ export declare function readCff(bytes: Uint8Array): CffFont | undefined;
22
+ /**
23
+ * A copy of the program whose font name holds only what a browser's font
24
+ * sanitizer accepts in a CFF table: printable ASCII without
25
+ * `[](){}<>/%` or a space, each other byte becoming `_`; `undefined` past
26
+ * the 127 characters it accepts or for a program that is not one font.
27
+ */
28
+ export declare function cffWithSafeName(bytes: Uint8Array): Uint8Array | undefined;
@@ -0,0 +1,416 @@
1
+ import { GLYPH_UNICODE } from "./glyph-names.js";
2
+ const ISO_ADOBE_NAMES = `
3
+ .notdef space exclam quotedbl numbersign dollar percent ampersand
4
+ quoteright parenleft parenright asterisk plus comma hyphen period slash
5
+ zero one two three four five six seven eight nine colon semicolon less
6
+ equal greater question at A B C D E F G H I J K L M N O P Q R S T U V W X Y
7
+ Z bracketleft backslash bracketright asciicircum underscore quoteleft a b c
8
+ d e f g h i j k l m n o p q r s t u v w x y z braceleft bar braceright
9
+ asciitilde exclamdown cent sterling fraction yen florin section currency
10
+ quotesingle quotedblleft guillemotleft guilsinglleft guilsinglright fi fl
11
+ endash dagger daggerdbl periodcentered paragraph bullet quotesinglbase
12
+ quotedblbase quotedblright guillemotright ellipsis perthousand questiondown
13
+ grave acute circumflex tilde macron breve dotaccent dieresis ring cedilla
14
+ hungarumlaut ogonek caron emdash AE ordfeminine Lslash Oslash OE
15
+ ordmasculine ae dotlessi lslash oslash oe germandbls onesuperior logicalnot
16
+ mu trademark Eth onehalf plusminus Thorn onequarter divide brokenbar degree
17
+ thorn threequarters twosuperior registered minus eth multiply threesuperior
18
+ copyright Aacute Acircumflex Adieresis Agrave Aring Atilde Ccedilla Eacute
19
+ Ecircumflex Edieresis Egrave Iacute Icircumflex Idieresis Igrave Ntilde
20
+ Oacute Ocircumflex Odieresis Ograve Otilde Scaron Uacute Ucircumflex
21
+ Udieresis Ugrave Yacute Ydieresis Zcaron aacute acircumflex adieresis
22
+ agrave aring atilde ccedilla eacute ecircumflex edieresis egrave iacute
23
+ icircumflex idieresis igrave ntilde oacute ocircumflex odieresis ograve
24
+ otilde scaron uacute ucircumflex udieresis ugrave yacute ydieresis zcaron
25
+ `
26
+ .trim()
27
+ .split(/\s+/);
28
+ /**
29
+ * The code point a glyph name stands for: an Adobe Glyph List name, or
30
+ * `uniXXXX` or `uXXXX[XX]` for one code point. Ligatures of several code
31
+ * points, suffixed variants such as `a.sc`, and noncharacters stand for none.
32
+ */
33
+ export function glyphUnicode(name) {
34
+ if (!name || name === ".notdef")
35
+ return undefined;
36
+ const known = GLYPH_UNICODE.get(name);
37
+ if (known !== undefined)
38
+ return known;
39
+ const match = /^uni([0-9A-F]{4})$|^u([0-9A-F]{4,6})$/.exec(name);
40
+ const hex = match?.[1] ?? match?.[2];
41
+ if (!hex)
42
+ return undefined;
43
+ const code = Number.parseInt(hex, 16);
44
+ const surrogate = code >= 0xd800 && code <= 0xdfff;
45
+ const nonCharacter = (code & 0xfffe) === 0xfffe;
46
+ return code <= 0x10ffff && !surrogate && !nonCharacter ? code : undefined;
47
+ }
48
+ /** Top and Private DICT operators this reader uses; two-byte ones as 1200 + their second byte. */
49
+ const OP = {
50
+ fontBBox: 5,
51
+ charset: 15,
52
+ charStrings: 17,
53
+ private: 18,
54
+ defaultWidthX: 20,
55
+ nominalWidthX: 21,
56
+ subrs: 19,
57
+ charstringType: 1206,
58
+ fontMatrix: 1207,
59
+ ros: 1230,
60
+ };
61
+ /** Standard strings come first; a string id from here on names an entry of the String INDEX. */
62
+ const STANDARD_STRINGS = 391;
63
+ /** Reads a CFF font program, or `undefined` when it is not one this reader can use. */
64
+ export function readCff(bytes) {
65
+ try {
66
+ const at = (offset) => {
67
+ if (!Number.isInteger(offset) || offset < 0 || offset >= bytes.length)
68
+ throw new RangeError("Outside the font");
69
+ return bytes[offset];
70
+ };
71
+ const u16 = (offset) => (at(offset) << 8) | at(offset + 1);
72
+ if (at(0) !== 1)
73
+ return undefined;
74
+ const names = readIndex(bytes, at(2), at, u16);
75
+ const tops = readIndex(bytes, names.end, at, u16);
76
+ const strings = readIndex(bytes, tops.end, at, u16);
77
+ const globalSubrs = readIndex(bytes, strings.end, at, u16);
78
+ // A PDF embeds one font per program.
79
+ if (names.items.length !== 1 || tops.items.length !== 1)
80
+ return undefined;
81
+ const top = readDict(bytes, ...tops.items[0], at);
82
+ const charStringsAt = top.get(OP.charStrings)?.[0];
83
+ if (charStringsAt === undefined)
84
+ return undefined;
85
+ if ((top.get(OP.charstringType)?.[0] ?? 2) !== 2)
86
+ return undefined;
87
+ const charStrings = readIndex(bytes, charStringsAt, at, u16);
88
+ const glyphCount = charStrings.items.length;
89
+ if (glyphCount === 0)
90
+ return undefined;
91
+ const [scale = 0.001, skewX = 0, skewY = 0, scaleY = 0.001] = top.get(OP.fontMatrix) ?? [];
92
+ if (skewX !== 0 || skewY !== 0 || scale !== scaleY || scale <= 0)
93
+ return undefined;
94
+ const unitsPerEm = Math.round(1 / scale);
95
+ if (unitsPerEm < 16 || unitsPerEm > 16384)
96
+ return undefined;
97
+ const [xMin = 0, yMin = 0, xMax = 0, yMax = 0] = top.get(OP.fontBBox) ?? [];
98
+ const cid = top.has(OP.ros);
99
+ const [nameStart, nameEnd] = names.items[0];
100
+ // A font name is 127 characters at most.
101
+ const name = latin1(bytes.subarray(nameStart, Math.min(nameEnd, nameStart + 127)));
102
+ if (cid)
103
+ return {
104
+ name,
105
+ cid,
106
+ glyphNames: [],
107
+ widths: [],
108
+ bbox: [xMin, yMin, xMax, yMax],
109
+ unitsPerEm,
110
+ };
111
+ // Glyphs may share a string; each is read once.
112
+ const custom = new Map();
113
+ const stringOf = (sid) => {
114
+ if (sid < ISO_ADOBE_NAMES.length)
115
+ return ISO_ADOBE_NAMES[sid];
116
+ const item = strings.items[sid - STANDARD_STRINGS];
117
+ if (!item)
118
+ return undefined;
119
+ if (!custom.has(sid))
120
+ custom.set(sid,
121
+ // A glyph name is 63 characters at most.
122
+ item[1] - item[0] <= 63 ? latin1(bytes.subarray(...item)) : undefined);
123
+ return custom.get(sid);
124
+ };
125
+ const glyphNames = charsetNames(top.get(OP.charset)?.[0] ?? 0, glyphCount, stringOf, at, u16);
126
+ const [privateSize, privateAt] = top.get(OP.private) ?? [0, 0];
127
+ const privateDict = privateSize
128
+ ? readDict(bytes, privateAt, privateAt + privateSize, at)
129
+ : new Map();
130
+ const localAt = privateDict.get(OP.subrs)?.[0];
131
+ const metrics = {
132
+ bytes,
133
+ defaultWidth: privateDict.get(OP.defaultWidthX)?.[0] ?? 0,
134
+ nominalWidth: privateDict.get(OP.nominalWidthX)?.[0] ?? 0,
135
+ global: globalSubrs.items,
136
+ local: localAt === undefined
137
+ ? []
138
+ : readIndex(bytes, privateAt + localAt, at, u16).items,
139
+ };
140
+ const widths = charStrings.items.map(([start, end]) => charstringWidth(metrics, start, end));
141
+ return {
142
+ name,
143
+ cid,
144
+ glyphNames,
145
+ widths,
146
+ bbox: [xMin, yMin, xMax, yMax],
147
+ unitsPerEm,
148
+ };
149
+ }
150
+ catch (error) {
151
+ if (error instanceof RangeError)
152
+ return undefined;
153
+ throw error;
154
+ }
155
+ }
156
+ /**
157
+ * A copy of the program whose font name holds only what a browser's font
158
+ * sanitizer accepts in a CFF table: printable ASCII without
159
+ * `[](){}<>/%` or a space, each other byte becoming `_`; `undefined` past
160
+ * the 127 characters it accepts or for a program that is not one font.
161
+ */
162
+ export function cffWithSafeName(bytes) {
163
+ try {
164
+ const at = (offset) => {
165
+ if (offset < 0 || offset >= bytes.length)
166
+ throw new RangeError("Outside the font");
167
+ return bytes[offset];
168
+ };
169
+ const u16 = (offset) => (at(offset) << 8) | at(offset + 1);
170
+ const names = readIndex(bytes, at(2), at, u16);
171
+ const [start, end] = names.items[0] ?? [0, 0];
172
+ if (names.items.length !== 1 || end - start > 127)
173
+ return undefined;
174
+ const safe = bytes.slice();
175
+ for (let offset = start; offset < end; offset += 1) {
176
+ const byte = safe[offset];
177
+ // A name that starts with NUL marks a deleted font, which is allowed.
178
+ if (offset === start && byte === 0)
179
+ continue;
180
+ if (byte < 33 || byte > 126 || UNSAFE_NAME_BYTES.has(byte))
181
+ safe[offset] = 0x5f;
182
+ }
183
+ return safe;
184
+ }
185
+ catch (error) {
186
+ if (error instanceof RangeError)
187
+ return undefined;
188
+ throw error;
189
+ }
190
+ }
191
+ const UNSAFE_NAME_BYTES = new Set([..."[](){}<>/%"].map((character) => character.charCodeAt(0)));
192
+ /** An INDEX: its items' byte ranges and where it ends. */
193
+ function readIndex(bytes, start, at, u16) {
194
+ const count = u16(start);
195
+ if (count === 0)
196
+ return { items: [], end: start + 2 };
197
+ const offSize = at(start + 2);
198
+ if (offSize < 1 || offSize > 4)
199
+ throw new RangeError("Bad offset size");
200
+ const offset = (index) => {
201
+ let value = 0;
202
+ for (let byte = 0; byte < offSize; byte += 1)
203
+ value = value * 256 + at(start + 3 + index * offSize + byte);
204
+ return value;
205
+ };
206
+ if (offset(0) !== 1)
207
+ throw new RangeError("Bad INDEX");
208
+ // Offsets count from the byte before the data.
209
+ const base = start + 2 + (count + 1) * offSize;
210
+ const items = [];
211
+ for (let index = 0; index < count; index += 1) {
212
+ const from = base + offset(index);
213
+ const to = base + offset(index + 1);
214
+ if (to < from || to > bytes.length)
215
+ throw new RangeError("Bad INDEX");
216
+ items.push([from, to]);
217
+ }
218
+ return { items, end: base + offset(count) };
219
+ }
220
+ /** A DICT's operands by operator. */
221
+ function readDict(bytes, start, end, at) {
222
+ if (end > bytes.length)
223
+ throw new RangeError("DICT past the font");
224
+ const entries = new Map();
225
+ let operands = [];
226
+ let offset = start;
227
+ while (offset < end) {
228
+ const byte = at(offset);
229
+ if (byte <= 21) {
230
+ const operator = byte === 12 ? 1200 + at(offset + 1) : byte;
231
+ offset += byte === 12 ? 2 : 1;
232
+ entries.set(operator, operands);
233
+ operands = [];
234
+ }
235
+ else if (byte === 28) {
236
+ operands.push(int16((at(offset + 1) << 8) | at(offset + 2)));
237
+ offset += 3;
238
+ }
239
+ else if (byte === 29) {
240
+ operands.push((at(offset + 1) << 24) |
241
+ (at(offset + 2) << 16) |
242
+ (at(offset + 3) << 8) |
243
+ at(offset + 4));
244
+ offset += 5;
245
+ }
246
+ else if (byte === 30) {
247
+ let text = "";
248
+ offset += 1;
249
+ for (let done = false; !done; offset += 1) {
250
+ const pair = at(offset);
251
+ for (const nibble of [pair >> 4, pair & 15]) {
252
+ if (nibble === 15) {
253
+ done = true;
254
+ break;
255
+ }
256
+ text += REAL_NIBBLES[nibble] ?? "";
257
+ }
258
+ }
259
+ operands.push(Number.parseFloat(text));
260
+ }
261
+ else if (byte >= 32 && byte <= 246) {
262
+ operands.push(byte - 139);
263
+ offset += 1;
264
+ }
265
+ else if (byte >= 247 && byte <= 250) {
266
+ operands.push((byte - 247) * 256 + at(offset + 1) + 108);
267
+ offset += 2;
268
+ }
269
+ else if (byte >= 251 && byte <= 254) {
270
+ operands.push(-(byte - 251) * 256 - at(offset + 1) - 108);
271
+ offset += 2;
272
+ }
273
+ else
274
+ throw new RangeError("Bad DICT byte");
275
+ }
276
+ return entries;
277
+ }
278
+ const REAL_NIBBLES = [
279
+ "0",
280
+ "1",
281
+ "2",
282
+ "3",
283
+ "4",
284
+ "5",
285
+ "6",
286
+ "7",
287
+ "8",
288
+ "9",
289
+ ".",
290
+ "E",
291
+ "E-",
292
+ "",
293
+ "-",
294
+ ];
295
+ /** Glyph names by glyph id from the charset at `offset`; 0 is the ISOAdobe set. */
296
+ function charsetNames(offset, glyphCount, stringOf, at, u16) {
297
+ const names = [".notdef"];
298
+ if (offset === 0) {
299
+ for (let glyph = 1; glyph < glyphCount; glyph += 1)
300
+ names.push(stringOf(glyph));
301
+ return names;
302
+ }
303
+ // The predefined expert sets name small capitals and old-style figures.
304
+ if (offset <= 2)
305
+ return [...names, ...Array(glyphCount - 1)];
306
+ const format = at(offset);
307
+ let cursor = offset + 1;
308
+ if (format === 0) {
309
+ for (let glyph = 1; glyph < glyphCount; glyph += 1, cursor += 2)
310
+ names.push(stringOf(u16(cursor)));
311
+ }
312
+ else if (format === 1 || format === 2) {
313
+ while (names.length < glyphCount) {
314
+ const first = u16(cursor);
315
+ const left = format === 1 ? at(cursor + 2) : u16(cursor + 2);
316
+ cursor += format === 1 ? 3 : 4;
317
+ for (let step = 0; step <= left && names.length < glyphCount; step += 1)
318
+ names.push(stringOf(first + step));
319
+ }
320
+ }
321
+ else
322
+ throw new RangeError("Bad charset format");
323
+ return names;
324
+ }
325
+ /**
326
+ * A Type 2 charstring's advance width (Adobe TN 5177): an extra first operand
327
+ * before the first operator that clears the stack, plus the nominal width;
328
+ * the default width without one. Subroutines called on the way are followed,
329
+ * ten deep at most. A charstring that breaks a rule gets the default width.
330
+ */
331
+ function charstringWidth(context, start, end) {
332
+ const { bytes, defaultWidth, nominalWidth } = context;
333
+ const stack = [];
334
+ // A width, or `undefined` to go on after a subroutine returns.
335
+ const walk = (from, to, depth) => {
336
+ let offset = from;
337
+ while (offset < to) {
338
+ const byte = bytes[offset];
339
+ const size = byte === 28 ? 3 : byte === 255 ? 5 : byte >= 247 && byte <= 254 ? 2 : 1;
340
+ if (offset + size > to)
341
+ return defaultWidth;
342
+ if (byte === 28 || byte >= 32) {
343
+ const next = (step) => bytes[offset + step];
344
+ stack.push(byte === 28
345
+ ? int16((next(1) << 8) | next(2))
346
+ : byte <= 246
347
+ ? byte - 139
348
+ : byte <= 250
349
+ ? (byte - 247) * 256 + next(1) + 108
350
+ : byte <= 254
351
+ ? -(byte - 251) * 256 - next(1) - 108
352
+ : ((next(1) << 24) |
353
+ (next(2) << 16) |
354
+ (next(3) << 8) |
355
+ next(4)) /
356
+ 65536);
357
+ // The Type 2 stack holds 48 operands.
358
+ if (stack.length > 48)
359
+ return defaultWidth;
360
+ offset += size;
361
+ continue;
362
+ }
363
+ offset += 1;
364
+ if (byte === 10 || byte === 29) {
365
+ const subrs = byte === 10 ? context.local : context.global;
366
+ const index = stack.pop();
367
+ const subr = index === undefined
368
+ ? undefined
369
+ : subrs[index + subrBias(subrs.length)];
370
+ if (!subr || depth >= 10)
371
+ return defaultWidth;
372
+ const width = walk(subr[0], subr[1], depth + 1);
373
+ if (width !== undefined)
374
+ return width;
375
+ continue;
376
+ }
377
+ if (byte === 11)
378
+ return depth > 0 ? undefined : defaultWidth;
379
+ const takes = STACK_CLEARING[byte];
380
+ if (takes === undefined)
381
+ return defaultWidth;
382
+ const count = stack.length;
383
+ const widthFirst = takes === "pairs"
384
+ ? count % 2 === 1
385
+ : takes === "endchar"
386
+ ? count === 1 || count === 5
387
+ : count > takes;
388
+ return widthFirst ? stack[0] + nominalWidth : defaultWidth;
389
+ }
390
+ return depth > 0 ? undefined : defaultWidth;
391
+ };
392
+ return walk(start, end, 0) ?? defaultWidth;
393
+ }
394
+ /** What a subroutine number is offset by, for a set of `count` subroutines. */
395
+ function subrBias(count) {
396
+ return count < 1240 ? 107 : count < 33900 ? 1131 : 32768;
397
+ }
398
+ /** Operators that clear the stack, and the operands they take without a width. */
399
+ const STACK_CLEARING = {
400
+ 1: "pairs", // hstem
401
+ 3: "pairs", // vstem
402
+ 18: "pairs", // hstemhm
403
+ 23: "pairs", // vstemhm
404
+ 19: "pairs", // hintmask, after implied vstems
405
+ 20: "pairs", // cntrmask
406
+ 21: 2, // rmoveto
407
+ 22: 1, // hmoveto
408
+ 4: 1, // vmoveto
409
+ 14: "endchar", // endchar: none, or four for an accented character
410
+ };
411
+ function int16(value) {
412
+ return value >= 0x8000 ? value - 0x10000 : value;
413
+ }
414
+ function latin1(bytes) {
415
+ return String.fromCharCode(...bytes);
416
+ }
@@ -1,6 +1,6 @@
1
1
  import type { EngineBatch, EngineChange, MaterializedDocument, RestoreTarget } from "../../engine.js";
2
2
  import type { EditFindOptions, EditOperation, ElementQuery, OperationIssue, PagePoint, PageRect, TextPosition, TextRange, TextTarget } from "../../types.js";
3
- import type { PageLayout, PdfElement, PdfOperation, TextLayout } from "../types.js";
3
+ import type { PageLayout, PdfElement, PdfOperation, TextFont, TextLayout } from "../types.js";
4
4
  import type { EditWorkerBitmap } from "../../../worker-protocol.js";
5
5
  import { FontLibrary } from "./fonts.js";
6
6
  import { ImageCache } from "./images.js";
@@ -78,6 +78,8 @@ export declare class PdfEditDocument {
78
78
  findText(query: string, options: EditFindOptions): TextTarget[];
79
79
  /** Lines, glyph boxes and styles of a text, text box or table element. */
80
80
  textLayout(elementId: string): TextLayout | undefined;
81
+ /** The browser face of the font a text or text box element is drawn in, see `TextFont`. */
82
+ textFont(elementId: string): TextFont | undefined;
81
83
  /** The layouts of every text element on a page, with the page's displayed size. */
82
84
  pageLayout(pageIndex: number): PageLayout | undefined;
83
85
  /** The caret position nearest to a page-space point; none on a page without text. */
@@ -1,8 +1,9 @@
1
1
  import { ViewerError } from "../../../errors.js";
2
2
  import { invalidOperationError, parseReference } from "../../operations.js";
3
3
  import { layoutOf, layoutsOf, positionIn, rectsOf, } from "./layout.js";
4
- import { markIsFresh, readMark, scanPage, } from "./elements.js";
4
+ import { markIsFresh, OBJECT_TEXT, readMark, scanPage, } from "./elements.js";
5
5
  import { FontLibrary, TextMeasurer } from "./fonts.js";
6
+ import { textFaceOf } from "./text-font.js";
6
7
  import { fontRequestsOf } from "./text-box.js";
7
8
  import { issueCollector, } from "./operations.js";
8
9
  import { insertTextBox } from "./text-box.js";
@@ -123,6 +124,8 @@ export class PdfEditDocument {
123
124
  #measurer;
124
125
  #pages;
125
126
  #batches = 0;
127
+ /** Browser faces of the document's fonts, by font program; see `textFont`. */
128
+ #faces = new Map();
126
129
  constructor(pdfium, original, fonts = new FontLibrary(async () => {
127
130
  throw new Error("No font source is configured");
128
131
  }), limits = defaultResourceLimits, assets = new AssetStore(), compact = compactPdf) {
@@ -402,6 +405,23 @@ export class PdfEditDocument {
402
405
  return element ? layoutOf(this.#pdfium, scan, element) : undefined;
403
406
  });
404
407
  }
408
+ /** The browser face of the font a text or text box element is drawn in, see `TextFont`. */
409
+ textFont(elementId) {
410
+ const location = this.#locate(elementId);
411
+ const element = location && this.getElement(elementId);
412
+ if (!element || (element.kind !== "text" && element.kind !== "textBox"))
413
+ return undefined;
414
+ const { lib } = this.#pdfium;
415
+ return this.#withPage(location.pageIndex, (page) => {
416
+ // A text box's lines share their font; its first line answers.
417
+ const object = location.indexes
418
+ .map((index) => lib.FPDFPage_GetObject(page, index))
419
+ .find((entry) => lib.FPDFPageObj_GetType(entry) === OBJECT_TEXT);
420
+ return object === undefined
421
+ ? undefined
422
+ : { elementId, ...textFaceOf(this.#pdfium, object, this.#faces) };
423
+ });
424
+ }
405
425
  /** The layouts of every text element on a page, with the page's displayed size. */
406
426
  pageLayout(pageIndex) {
407
427
  if (pageIndex < 0 || pageIndex >= this.#pages.length)
@@ -47,5 +47,12 @@ export declare function readMark(pdfium: Pdfium, object: number): MarkParams | u
47
47
  */
48
48
  export declare function markIsFresh(pdfium: Pdfium, page: number, textPage: number, geometry: PageGeometry, mark: MarkParams, indexes: readonly number[]): boolean;
49
49
  export declare function objectBounds(pdfium: Pdfium, object: number, geometry: PageGeometry): PageRect | undefined;
50
+ /**
51
+ * How much a text object's matrix scales its glyphs: the length of its
52
+ * vertical axis. Producers often write `1 Tf` and carry the size in the
53
+ * matrix, so the point size the text shows is its font size times this; a
54
+ * turn or a horizontal squeeze leaves it alone.
55
+ */
56
+ export declare function textScale(matrix: readonly number[] | undefined): number;
50
57
  export declare function textStyle(pdfium: Pdfium, object: number): PdfTextStyle;
51
58
  export declare function hex(rgba: readonly number[]): string;
@@ -232,6 +232,16 @@ const TABLE_OPERATIONS = Object.freeze([
232
232
  function textOf(pdfium, object, textPage) {
233
233
  return pdfium.readWideString((buffer, bytes) => pdfium.lib.FPDFTextObj_GetText(object, textPage, buffer, bytes));
234
234
  }
235
+ /**
236
+ * How much a text object's matrix scales its glyphs: the length of its
237
+ * vertical axis. Producers often write `1 Tf` and carry the size in the
238
+ * matrix, so the point size the text shows is its font size times this; a
239
+ * turn or a horizontal squeeze leaves it alone.
240
+ */
241
+ export function textScale(matrix) {
242
+ const scale = matrix ? Math.hypot(matrix[2], matrix[3]) : 1;
243
+ return Number.isFinite(scale) && scale > 0 ? scale : 1;
244
+ }
235
245
  export function textStyle(pdfium, object) {
236
246
  const { lib } = pdfium;
237
247
  const font = lib.FPDFTextObj_GetFont(object);
@@ -245,9 +255,10 @@ export function textStyle(pdfium, object) {
245
255
  const flags = lib.FPDFFont_GetFlags(font);
246
256
  const weight = lib.FPDFFont_GetWeight(font);
247
257
  const size = pdfium.readNumbers(1, "float", ([pointer]) => lib.FPDFTextObj_GetFontSize(object, pointer))?.[0] ?? 0;
258
+ const matrix = pdfium.readNumbers(6, "float", ([pointer]) => lib.FPDFPageObj_GetMatrix(object, pointer));
248
259
  return {
249
260
  fontFamily: family,
250
- fontSize: round(size),
261
+ fontSize: round(size * textScale(matrix)),
251
262
  bold: weight >= 600 ||
252
263
  (flags & FLAG_FORCE_BOLD) !== 0 ||
253
264
  /bold|black|heavy/i.test(baseName),
@@ -1,4 +1,4 @@
1
- import { objectBounds, OBJECT_TEXT } from "./elements.js";
1
+ import { objectBounds, OBJECT_TEXT, textScale } from "./elements.js";
2
2
  import { firstNonWinAnsi, isStandardFamily, parseCmap, } from "./fonts.js";
3
3
  import { pageToUser } from "./geometry.js";
4
4
  import { parseColor, setText, textBoxReplaceText, textBoxSetTextStyle, textBoxTarget, } from "./text-box.js";
@@ -255,7 +255,7 @@ function splitAround(context, target, before, middle, after) {
255
255
  const old = lib.FPDFPage_GetObject(page, index);
256
256
  const oldFont = lib.FPDFTextObj_GetFont(old);
257
257
  const matrix = pdfium.readNumbers(6, "float", ([pointer]) => lib.FPDFPageObj_GetMatrix(old, pointer)) ?? [1, 0, 0, 1, 0, 0];
258
- const size = pdfium.readNumbers(1, "float", ([pointer]) => lib.FPDFTextObj_GetFontSize(old, pointer))?.[0] ?? style.fontSize;
258
+ const size = pdfium.readNumbers(1, "float", ([pointer]) => lib.FPDFTextObj_GetFontSize(old, pointer))?.[0] ?? style.fontSize / textScale(matrix);
259
259
  const [a, b, c, d, e, f] = matrix;
260
260
  const [r, g, bl] = parseColor(style.color);
261
261
  const parts = [
@@ -349,7 +349,7 @@ function replaceWithFallback(context, target, text) {
349
349
  context.withPage(location.pageIndex, (page) => {
350
350
  const old = lib.FPDFPage_GetObject(page, index);
351
351
  const matrix = pdfium.readNumbers(6, "float", ([pointer]) => lib.FPDFPageObj_GetMatrix(old, pointer)) ?? [1, 0, 0, 1, 0, 0];
352
- const size = pdfium.readNumbers(1, "float", ([pointer]) => lib.FPDFTextObj_GetFontSize(old, pointer))?.[0] ?? style.fontSize;
352
+ const size = pdfium.readNumbers(1, "float", ([pointer]) => lib.FPDFTextObj_GetFontSize(old, pointer))?.[0] ?? style.fontSize / textScale(matrix);
353
353
  const object = lib.FPDFPageObj_CreateTextObj(context.document, font.handle, size);
354
354
  setText(pdfium, object, text);
355
355
  const [r, g, b] = parseColor(style.color);
@@ -399,7 +399,9 @@ function resize(context, target, fontSize) {
399
399
  }
400
400
  const matrix = pdfium.readNumbers(6, "float", ([pointer]) => lib.FPDFPageObj_GetMatrix(old, pointer)) ?? [1, 0, 0, 1, 0, 0];
401
401
  const color = pdfium.readNumbers(4, "i32", ([r, g, b, a]) => lib.FPDFPageObj_GetFillColor(old, r, g, b, a)) ?? [0, 0, 0, 255];
402
- const object = lib.FPDFPageObj_CreateTextObj(context.document, lib.FPDFTextObj_GetFont(old), fontSize);
402
+ // The size is the one the text shows: the old matrix, applied below,
403
+ // scales it again, so the font gets the size undone by that scale.
404
+ const object = lib.FPDFPageObj_CreateTextObj(context.document, lib.FPDFTextObj_GetFont(old), fontSize / textScale(matrix));
403
405
  setText(pdfium, object, text);
404
406
  lib.FPDFPageObj_SetFillColor(object, color[0], color[1], color[2], color[3]);
405
407
  const [a, b, c, d, e, f] = matrix;
@@ -0,0 +1,2 @@
1
+ /** Code points by glyph name. */
2
+ export declare const GLYPH_UNICODE: ReadonlyMap<string, number>;