reamkit 1.24.0 → 1.25.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +22 -11
- package/dist/esm/core/converter/facade.d.ts +3 -3
- package/dist/esm/core/converter/facade.js +12 -0
- package/dist/esm/core/converter/ream.d.ts +48 -4
- package/dist/esm/core/converter/ream.js +25 -4
- package/dist/esm/core/document-model/index.d.ts +1 -1
- package/dist/esm/core/document-model/types.d.ts +17 -0
- package/dist/esm/core/outline.d.ts +17 -0
- package/dist/esm/core/outline.js +30 -0
- package/dist/esm/core/style-cascade/resolver.js +1 -0
- package/dist/esm/core/style-cascade/types.d.ts +3 -1
- package/dist/esm/excel/sheet-to-flow.d.ts +10 -0
- package/dist/esm/excel/sheet-to-flow.js +14 -1
- package/dist/esm/html/html-writer.js +3 -2
- package/dist/esm/index.d.ts +3 -0
- package/dist/esm/index.js +2 -1
- package/dist/esm/layout/page-doc.js +1 -1
- package/dist/esm/layout/styled-layout.js +41 -15
- package/dist/esm/markdown/markdown-writer.d.ts +41 -0
- package/dist/esm/markdown/markdown-writer.js +733 -0
- package/dist/esm/pdf/styled-page-emitter.js +20 -1
- package/dist/esm/pdf-reader/annots.d.ts +24 -0
- package/dist/esm/pdf-reader/annots.js +126 -0
- package/dist/esm/pdf-reader/content.d.ts +131 -5
- package/dist/esm/pdf-reader/content.js +169 -12
- package/dist/esm/pdf-reader/display.d.ts +56 -0
- package/dist/esm/pdf-reader/display.js +162 -0
- package/dist/esm/pdf-reader/document.d.ts +36 -1
- package/dist/esm/pdf-reader/document.js +92 -25
- package/dist/esm/pdf-reader/embedded-fonts.d.ts +31 -0
- package/dist/esm/pdf-reader/embedded-fonts.js +94 -0
- package/dist/esm/pdf-reader/flow-build.d.ts +61 -6
- package/dist/esm/pdf-reader/flow-build.js +128 -22
- package/dist/esm/pdf-reader/font.js +185 -4
- package/dist/esm/pdf-reader/image-decode.js +55 -4
- package/dist/esm/pdf-reader/images.d.ts +6 -0
- package/dist/esm/pdf-reader/images.js +25 -5
- package/dist/esm/pdf-reader/jpeg.d.ts +18 -0
- package/dist/esm/pdf-reader/jpeg.js +419 -0
- package/dist/esm/pdf-reader/layout.d.ts +1 -1
- package/dist/esm/pdf-reader/layout.js +221 -32
- package/dist/esm/pdf-reader/pattern-tint.d.ts +17 -0
- package/dist/esm/pdf-reader/pattern-tint.js +181 -0
- package/dist/esm/pdf-reader/reader.d.ts +9 -1
- package/dist/esm/pdf-reader/reader.js +22 -6
- package/dist/esm/pdf-reader/shading.d.ts +14 -0
- package/dist/esm/pdf-reader/shading.js +27 -1
- package/dist/esm/pdf-reader/tagged.js +156 -17
- package/dist/esm/pdf-reader/text.d.ts +13 -1
- package/dist/esm/pdf-reader/text.js +70 -3
- package/dist/esm/pdf-reader/vector.d.ts +25 -1
- package/dist/esm/pdf-reader/vector.js +168 -12
- package/dist/esm/pptx/slide-parser.js +5 -0
- package/dist/esm/word/docx-writer.js +11 -1
- package/dist/esm/word/drawing-parser.js +7 -1
- package/package.json +1 -1
|
@@ -26,10 +26,11 @@ var PdfFile = class PdfFile {
|
|
|
26
26
|
objStmCache = /* @__PURE__ */ new Map();
|
|
27
27
|
decryptor;
|
|
28
28
|
encryptObjNum = -1;
|
|
29
|
-
constructor(buf, xref, trailer) {
|
|
29
|
+
constructor(buf, xref, trailer, filters = {}) {
|
|
30
30
|
this.buf = buf;
|
|
31
31
|
this.xref = xref;
|
|
32
32
|
this.trailer = trailer;
|
|
33
|
+
this.filters = filters;
|
|
33
34
|
}
|
|
34
35
|
/**
|
|
35
36
|
* Parse a whole PDF byte buffer: read the cross-reference chain (or brute-force
|
|
@@ -38,28 +39,31 @@ var PdfFile = class PdfFile {
|
|
|
38
39
|
* @param bytes The complete PDF file bytes.
|
|
39
40
|
* @param password The user password for an encrypted source (EP14); the empty
|
|
40
41
|
* string opens permissions-only encryption.
|
|
42
|
+
* @param filters Decoders for `/Filter` names this reader does not implement.
|
|
41
43
|
* @returns A ready-to-query {@link PdfFile}.
|
|
42
44
|
*/
|
|
43
|
-
static parse(bytes, password = "") {
|
|
45
|
+
static parse(bytes, password = "", filters = {}) {
|
|
46
|
+
const unknownFilters = /* @__PURE__ */ new Set();
|
|
44
47
|
let xref = /* @__PURE__ */ new Map();
|
|
45
48
|
let trailer = /* @__PURE__ */ new Map();
|
|
46
49
|
try {
|
|
47
50
|
const start = findStartXref(bytes);
|
|
48
51
|
if (start >= 0) {
|
|
49
|
-
const built = readXrefChain(bytes, start);
|
|
52
|
+
const built = readXrefChain(bytes, start, unknownFilters, filters);
|
|
50
53
|
xref = built.xref;
|
|
51
54
|
trailer = built.trailer;
|
|
52
55
|
}
|
|
53
56
|
} catch {}
|
|
54
57
|
if (xref.size === 0 || !(trailer.get("Root") instanceof PdfRef)) {
|
|
55
|
-
const scanned = bruteForceScan(bytes);
|
|
58
|
+
const scanned = bruteForceScan(bytes, unknownFilters, filters);
|
|
56
59
|
for (const [id, entry] of scanned.xref) if (!xref.has(id)) xref.set(id, entry);
|
|
57
60
|
if (!(trailer.get("Root") instanceof PdfRef) && scanned.root) {
|
|
58
61
|
trailer = new Map(trailer);
|
|
59
62
|
trailer.set("Root", scanned.root);
|
|
60
63
|
}
|
|
61
64
|
}
|
|
62
|
-
const file = new PdfFile(bytes, xref, trailer);
|
|
65
|
+
const file = new PdfFile(bytes, xref, trailer, filters);
|
|
66
|
+
for (const name of unknownFilters) file.unknownFilters.add(name);
|
|
63
67
|
file.initEncryption(password);
|
|
64
68
|
return file;
|
|
65
69
|
}
|
|
@@ -121,7 +125,7 @@ var PdfFile = class PdfFile {
|
|
|
121
125
|
const dec = this.decryptor.decrypt(stream, streamObj, obj.generation);
|
|
122
126
|
if (dec instanceof PdfStream) stream = dec;
|
|
123
127
|
}
|
|
124
|
-
const data = inflateStream(stream);
|
|
128
|
+
const data = inflateStream(stream, this.unknownFilters, this.filters);
|
|
125
129
|
const first = numOf(this.resolve(stream.dict.get("First") ?? PDF_NULL));
|
|
126
130
|
for (const member of objStmHeader(data, numOf(this.resolve(stream.dict.get("N") ?? PDF_NULL)))) out.set(member.id, parseObject(new Lexer(data, first + member.off), this.lengthResolver));
|
|
127
131
|
return out;
|
|
@@ -160,6 +164,7 @@ var PdfFile = class PdfFile {
|
|
|
160
164
|
const mediaBox = readRectangle(this.get(node, "MediaBox")) ?? inherited.mediaBox;
|
|
161
165
|
const resourcesVal = this.get(node, "Resources");
|
|
162
166
|
const resources = resourcesVal instanceof Map ? resourcesVal : inherited.resources;
|
|
167
|
+
const rotate = readRotate(this.get(node, "Rotate")) ?? inherited.rotate;
|
|
163
168
|
const type = node.get("Type");
|
|
164
169
|
const kids = this.get(node, "Kids");
|
|
165
170
|
if (type instanceof PdfName && type.value === "Pages" && Array.isArray(kids)) {
|
|
@@ -167,7 +172,8 @@ var PdfFile = class PdfFile {
|
|
|
167
172
|
const kidNode = this.resolve(kid);
|
|
168
173
|
if (kidNode instanceof Map) this.walkPageTree(kidNode, {
|
|
169
174
|
mediaBox,
|
|
170
|
-
resources
|
|
175
|
+
resources,
|
|
176
|
+
rotate
|
|
171
177
|
}, out, seen);
|
|
172
178
|
if (out.length >= MAX_PAGES) break;
|
|
173
179
|
}
|
|
@@ -176,6 +182,7 @@ var PdfFile = class PdfFile {
|
|
|
176
182
|
out.push({
|
|
177
183
|
dict: node,
|
|
178
184
|
mediaBox: mediaBox ?? DEFAULT_MEDIA_BOX,
|
|
185
|
+
rotate: rotate ?? 0,
|
|
179
186
|
resources
|
|
180
187
|
});
|
|
181
188
|
}
|
|
@@ -202,20 +209,57 @@ var PdfFile = class PdfFile {
|
|
|
202
209
|
const filter = this.resolve(stream.dict.get("Filter") ?? PDF_NULL);
|
|
203
210
|
const filters = Array.isArray(filter) ? filter : [filter];
|
|
204
211
|
let flate = false;
|
|
205
|
-
for (const f of filters)
|
|
206
|
-
|
|
207
|
-
|
|
208
|
-
|
|
212
|
+
for (const f of filters) {
|
|
213
|
+
if (!(f instanceof PdfName)) continue;
|
|
214
|
+
if (f.value === "FlateDecode" || f.value === "Fl") try {
|
|
215
|
+
data = unzlibSync(data);
|
|
216
|
+
flate = true;
|
|
217
|
+
} catch {}
|
|
218
|
+
else if (this.filters[f.value]) {
|
|
219
|
+
const decoded = runFilter(this.filters[f.value], data);
|
|
220
|
+
if (decoded) data = decoded;
|
|
221
|
+
else this.unknownFilters.add(f.value);
|
|
222
|
+
} else if (!PASSTHROUGH_FILTERS.has(f.value)) this.unknownFilters.add(f.value);
|
|
223
|
+
}
|
|
209
224
|
return flate ? applyStreamPredictor(this, stream.dict, data) : data;
|
|
210
225
|
}
|
|
226
|
+
/** Filter names met in this file that nothing here can undo. */
|
|
227
|
+
unknownFilters = /* @__PURE__ */ new Set();
|
|
211
228
|
};
|
|
229
|
+
/** Run a supplied filter, treating a throw as a filter that cannot decode. */
|
|
230
|
+
function runFilter(filter, data) {
|
|
231
|
+
if (!filter) return void 0;
|
|
232
|
+
try {
|
|
233
|
+
const out = filter(data);
|
|
234
|
+
return out.length > 0 ? out : void 0;
|
|
235
|
+
} catch {
|
|
236
|
+
return;
|
|
237
|
+
}
|
|
238
|
+
}
|
|
239
|
+
var PASSTHROUGH_FILTERS = new Set([
|
|
240
|
+
"LZWDecode",
|
|
241
|
+
"LZW",
|
|
242
|
+
"RunLengthDecode",
|
|
243
|
+
"RL",
|
|
244
|
+
"ASCII85Decode",
|
|
245
|
+
"A85",
|
|
246
|
+
"ASCIIHexDecode",
|
|
247
|
+
"AHx",
|
|
248
|
+
"DCTDecode",
|
|
249
|
+
"DCT",
|
|
250
|
+
"JPXDecode",
|
|
251
|
+
"JBIG2Decode",
|
|
252
|
+
"CCITTFaxDecode",
|
|
253
|
+
"CCF",
|
|
254
|
+
"Crypt"
|
|
255
|
+
]);
|
|
212
256
|
function findStartXref(buf) {
|
|
213
257
|
const tail = lastIndexOfAscii(buf, "startxref");
|
|
214
258
|
if (tail < 0) return -1;
|
|
215
259
|
const tok = new Lexer(buf, tail + 9).nextToken();
|
|
216
260
|
return tok.kind === "num" ? tok.value : -1;
|
|
217
261
|
}
|
|
218
|
-
function readXrefChain(buf, offset) {
|
|
262
|
+
function readXrefChain(buf, offset, unknown, filters = {}) {
|
|
219
263
|
const xref = /* @__PURE__ */ new Map();
|
|
220
264
|
let trailer = /* @__PURE__ */ new Map();
|
|
221
265
|
const visited = /* @__PURE__ */ new Set();
|
|
@@ -224,7 +268,7 @@ function readXrefChain(buf, offset) {
|
|
|
224
268
|
const at = queue.shift();
|
|
225
269
|
if (at < 0 || at >= buf.length || visited.has(at)) continue;
|
|
226
270
|
visited.add(at);
|
|
227
|
-
const section = readXrefAt(buf, at);
|
|
271
|
+
const section = readXrefAt(buf, at, unknown, filters);
|
|
228
272
|
if (!section) continue;
|
|
229
273
|
for (const [id, entry] of section.xref) if (!xref.has(id)) xref.set(id, entry);
|
|
230
274
|
if (trailer.size === 0) trailer = section.trailer;
|
|
@@ -238,12 +282,12 @@ function readXrefChain(buf, offset) {
|
|
|
238
282
|
trailer
|
|
239
283
|
};
|
|
240
284
|
}
|
|
241
|
-
function readXrefAt(buf, offset) {
|
|
285
|
+
function readXrefAt(buf, offset, unknown, filters = {}) {
|
|
242
286
|
const lexer = new Lexer(buf, offset);
|
|
243
287
|
const head = lexer.nextToken();
|
|
244
288
|
if (head.kind === "keyword" && head.value === "xref") return readClassicXref(lexer);
|
|
245
289
|
const obj = parseIndirectObject(new Lexer(buf, offset));
|
|
246
|
-
if (obj && obj.value instanceof PdfStream) return readXrefStream(obj.value);
|
|
290
|
+
if (obj && obj.value instanceof PdfStream) return readXrefStream(obj.value, unknown, filters);
|
|
247
291
|
}
|
|
248
292
|
function readClassicXref(lexer) {
|
|
249
293
|
const xref = /* @__PURE__ */ new Map();
|
|
@@ -272,7 +316,7 @@ function readClassicXref(lexer) {
|
|
|
272
316
|
trailer: trailerVal instanceof Map ? trailerVal : /* @__PURE__ */ new Map()
|
|
273
317
|
};
|
|
274
318
|
}
|
|
275
|
-
function readXrefStream(stream) {
|
|
319
|
+
function readXrefStream(stream, unknown, filters = {}) {
|
|
276
320
|
const dict = stream.dict;
|
|
277
321
|
const wv = dict.get("W");
|
|
278
322
|
if (!Array.isArray(wv) || wv.length < 3) return void 0;
|
|
@@ -281,7 +325,7 @@ function readXrefStream(stream) {
|
|
|
281
325
|
const w2 = numOf(wv[2]);
|
|
282
326
|
const rowLen = w0 + w1 + w2;
|
|
283
327
|
if (rowLen <= 0) return void 0;
|
|
284
|
-
const data = inflateStream(stream);
|
|
328
|
+
const data = inflateStream(stream, unknown, filters);
|
|
285
329
|
const size = numOf(dict.get("Size"));
|
|
286
330
|
const indexV = dict.get("Index");
|
|
287
331
|
const index = Array.isArray(indexV) ? indexV.map(numOf) : [0, size];
|
|
@@ -327,14 +371,22 @@ function objStmHeader(data, n) {
|
|
|
327
371
|
}
|
|
328
372
|
return out;
|
|
329
373
|
}
|
|
330
|
-
function inflateStream(stream) {
|
|
374
|
+
function inflateStream(stream, unknown, filters = {}) {
|
|
331
375
|
let data = stream.data;
|
|
332
376
|
const filter = stream.dict.get("Filter") ?? PDF_NULL;
|
|
333
|
-
const
|
|
334
|
-
for (const f of
|
|
335
|
-
|
|
336
|
-
|
|
337
|
-
|
|
377
|
+
const chain = Array.isArray(filter) ? filter : [filter];
|
|
378
|
+
for (const f of chain) {
|
|
379
|
+
if (!(f instanceof PdfName)) continue;
|
|
380
|
+
if (f.value === "FlateDecode" || f.value === "Fl") try {
|
|
381
|
+
data = unzlibSync(data);
|
|
382
|
+
} catch {
|
|
383
|
+
return new Uint8Array(0);
|
|
384
|
+
}
|
|
385
|
+
else if (!PASSTHROUGH_FILTERS.has(f.value)) {
|
|
386
|
+
const decoded = runFilter(filters[f.value], data);
|
|
387
|
+
if (decoded) data = decoded;
|
|
388
|
+
else unknown?.add(f.value);
|
|
389
|
+
}
|
|
338
390
|
}
|
|
339
391
|
const parmsVal = stream.dict.get("DecodeParms") ?? stream.dict.get("DP");
|
|
340
392
|
const parms = parmsVal instanceof Map ? parmsVal : Array.isArray(parmsVal) ? parmsVal.find((p) => p instanceof Map) : void 0;
|
|
@@ -362,7 +414,7 @@ function applyStreamPredictor(file, dict, data) {
|
|
|
362
414
|
columns: numOf(file.get(parms, "Columns")) || 1
|
|
363
415
|
});
|
|
364
416
|
}
|
|
365
|
-
function bruteForceScan(buf) {
|
|
417
|
+
function bruteForceScan(buf, unknown, filters = {}) {
|
|
366
418
|
const xref = /* @__PURE__ */ new Map();
|
|
367
419
|
const objStmObjs = [];
|
|
368
420
|
let root;
|
|
@@ -398,7 +450,7 @@ function bruteForceScan(buf) {
|
|
|
398
450
|
if (!entry || entry.kind !== "uncompressed") continue;
|
|
399
451
|
const obj = parseIndirectObject(new Lexer(buf, entry.offset));
|
|
400
452
|
if (!obj || !(obj.value instanceof PdfStream)) continue;
|
|
401
|
-
const data = inflateStream(obj.value);
|
|
453
|
+
const data = inflateStream(obj.value, unknown, filters);
|
|
402
454
|
const first = numOf(obj.value.dict.get("First") ?? PDF_NULL);
|
|
403
455
|
objStmHeader(data, numOf(obj.value.dict.get("N") ?? PDF_NULL)).forEach((member, index) => {
|
|
404
456
|
if (!xref.has(member.id)) xref.set(member.id, {
|
|
@@ -439,6 +491,21 @@ function readRectangle(value) {
|
|
|
439
491
|
nums[3]
|
|
440
492
|
];
|
|
441
493
|
}
|
|
494
|
+
/**
|
|
495
|
+
* §14.11.1 `/Rotate` — a multiple of 90, and the spec says so, but a file is
|
|
496
|
+
* free to write −90 or 450 and readers take both. Anything that is not a
|
|
497
|
+
* quarter turn is no turn at all.
|
|
498
|
+
*/
|
|
499
|
+
function readRotate(value) {
|
|
500
|
+
if (typeof value !== "number" || !Number.isFinite(value)) return void 0;
|
|
501
|
+
if (value % 90 !== 0) return void 0;
|
|
502
|
+
return [
|
|
503
|
+
0,
|
|
504
|
+
90,
|
|
505
|
+
180,
|
|
506
|
+
270
|
|
507
|
+
][(value / 90 % 4 + 4) % 4];
|
|
508
|
+
}
|
|
442
509
|
function concatWithSpaces(parts) {
|
|
443
510
|
if (parts.length === 1) return parts[0];
|
|
444
511
|
const total = parts.reduce((n, p) => n + p.length + 1, 0);
|
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
import { PdfDict } from '../pdf/objects.js';
|
|
2
|
+
import { PdfFile, PdfPage } from './document.js';
|
|
3
|
+
import { FontRegistry } from '../core/font/index.js';
|
|
4
|
+
/**
|
|
5
|
+
* Every embedded TrueType program the pages use, by the name a run will ask for
|
|
6
|
+
* — the `/BaseFont` lowercased, with the subset prefix dropped.
|
|
7
|
+
*
|
|
8
|
+
* Each PDF font object becomes its OWN entry rather than being folded into a
|
|
9
|
+
* family: `Arial-BoldMT` is a different program from `ArialMT` and the run that
|
|
10
|
+
* uses it names it exactly, so nothing has to guess at weights or match faces
|
|
11
|
+
* up. A face the reader cannot parse is left out and the writer substitutes for
|
|
12
|
+
* it as before.
|
|
13
|
+
*
|
|
14
|
+
* Only `/FontFile2` is read. A CFF program (`/FontFile3`) and a Type 1 one
|
|
15
|
+
* (`/FontFile`) are different formats that the layout's parser does not take;
|
|
16
|
+
* those keep their substitute.
|
|
17
|
+
*
|
|
18
|
+
* @param file The owning file.
|
|
19
|
+
* @param pages The pages whose fonts are wanted.
|
|
20
|
+
* @returns Name → a one-face registry holding that program.
|
|
21
|
+
*/
|
|
22
|
+
export declare function collectEmbeddedFonts(file: PdfFile, pages: ReadonlyArray<PdfPage>): Map<string, FontRegistry>;
|
|
23
|
+
/**
|
|
24
|
+
* The name a run set in `fontDict` will ask for: its `/BaseFont` without the
|
|
25
|
+
* six-capital subset prefix (§9.6.4), lowercased.
|
|
26
|
+
*
|
|
27
|
+
* @param file The owning file.
|
|
28
|
+
* @param fontDict A `/Font` dictionary.
|
|
29
|
+
* @returns The name, or `undefined` when the font states none.
|
|
30
|
+
*/
|
|
31
|
+
export declare function embeddedFontName(file: PdfFile, fontDict: PdfDict): string | undefined;
|
|
@@ -0,0 +1,94 @@
|
|
|
1
|
+
import { FontRegistry } from "../core/font/font-registry.js";
|
|
2
|
+
import { PDF_NULL, PdfName, PdfStream } from "../pdf/objects.js";
|
|
3
|
+
//#region src/pdf-reader/embedded-fonts.ts
|
|
4
|
+
var MAX_FORM_DEPTH = 8;
|
|
5
|
+
/**
|
|
6
|
+
* Every embedded TrueType program the pages use, by the name a run will ask for
|
|
7
|
+
* — the `/BaseFont` lowercased, with the subset prefix dropped.
|
|
8
|
+
*
|
|
9
|
+
* Each PDF font object becomes its OWN entry rather than being folded into a
|
|
10
|
+
* family: `Arial-BoldMT` is a different program from `ArialMT` and the run that
|
|
11
|
+
* uses it names it exactly, so nothing has to guess at weights or match faces
|
|
12
|
+
* up. A face the reader cannot parse is left out and the writer substitutes for
|
|
13
|
+
* it as before.
|
|
14
|
+
*
|
|
15
|
+
* Only `/FontFile2` is read. A CFF program (`/FontFile3`) and a Type 1 one
|
|
16
|
+
* (`/FontFile`) are different formats that the layout's parser does not take;
|
|
17
|
+
* those keep their substitute.
|
|
18
|
+
*
|
|
19
|
+
* @param file The owning file.
|
|
20
|
+
* @param pages The pages whose fonts are wanted.
|
|
21
|
+
* @returns Name → a one-face registry holding that program.
|
|
22
|
+
*/
|
|
23
|
+
function collectEmbeddedFonts(file, pages) {
|
|
24
|
+
const out = /* @__PURE__ */ new Map();
|
|
25
|
+
const seen = /* @__PURE__ */ new Set();
|
|
26
|
+
const visiting = /* @__PURE__ */ new Set();
|
|
27
|
+
const addFont = (fontDict) => {
|
|
28
|
+
if (seen.has(fontDict)) return;
|
|
29
|
+
seen.add(fontDict);
|
|
30
|
+
const name = embeddedFontName(file, fontDict);
|
|
31
|
+
if (name === void 0 || out.has(name)) return;
|
|
32
|
+
const program = fontProgram(file, fontDict);
|
|
33
|
+
if (!program) return;
|
|
34
|
+
try {
|
|
35
|
+
out.set(name, FontRegistry.fromBytes({ regular: program }));
|
|
36
|
+
} catch {}
|
|
37
|
+
};
|
|
38
|
+
const walk = (resources, depth) => {
|
|
39
|
+
if (!resources) return;
|
|
40
|
+
const fonts = file.get(resources, "Font");
|
|
41
|
+
if (fonts instanceof Map) for (const value of fonts.values()) {
|
|
42
|
+
const dict = file.resolve(value);
|
|
43
|
+
if (dict instanceof Map) addFont(dict);
|
|
44
|
+
}
|
|
45
|
+
if (depth >= MAX_FORM_DEPTH) return;
|
|
46
|
+
const xobjects = file.get(resources, "XObject");
|
|
47
|
+
if (!(xobjects instanceof Map)) return;
|
|
48
|
+
for (const value of xobjects.values()) {
|
|
49
|
+
const stream = file.resolve(value);
|
|
50
|
+
if (!(stream instanceof PdfStream) || visiting.has(stream)) continue;
|
|
51
|
+
const subtype = file.get(stream.dict, "Subtype");
|
|
52
|
+
if (!(subtype instanceof PdfName) || subtype.value !== "Form") continue;
|
|
53
|
+
visiting.add(stream);
|
|
54
|
+
const own = file.get(stream.dict, "Resources");
|
|
55
|
+
walk(own instanceof Map ? own : resources, depth + 1);
|
|
56
|
+
visiting.delete(stream);
|
|
57
|
+
}
|
|
58
|
+
};
|
|
59
|
+
for (const page of pages) walk(page.resources, 0);
|
|
60
|
+
return out;
|
|
61
|
+
}
|
|
62
|
+
/**
|
|
63
|
+
* The name a run set in `fontDict` will ask for: its `/BaseFont` without the
|
|
64
|
+
* six-capital subset prefix (§9.6.4), lowercased.
|
|
65
|
+
*
|
|
66
|
+
* @param file The owning file.
|
|
67
|
+
* @param fontDict A `/Font` dictionary.
|
|
68
|
+
* @returns The name, or `undefined` when the font states none.
|
|
69
|
+
*/
|
|
70
|
+
function embeddedFontName(file, fontDict) {
|
|
71
|
+
const base = file.resolve(fontDict.get("BaseFont") ?? PDF_NULL);
|
|
72
|
+
if (!(base instanceof PdfName)) return void 0;
|
|
73
|
+
const name = base.value.replace(/^[A-Z]{6}\+/u, "").trim();
|
|
74
|
+
return name.length > 0 ? name.toLowerCase() : void 0;
|
|
75
|
+
}
|
|
76
|
+
/** §9.9 `/FontFile2` — the TrueType program, off the font or its descendant. */
|
|
77
|
+
function fontProgram(file, fontDict) {
|
|
78
|
+
const owner = descendant(file, fontDict) ?? fontDict;
|
|
79
|
+
const descriptor = file.resolve(owner.get("FontDescriptor") ?? PDF_NULL);
|
|
80
|
+
if (!(descriptor instanceof Map)) return void 0;
|
|
81
|
+
const program = file.resolve(descriptor.get("FontFile2") ?? PDF_NULL);
|
|
82
|
+
if (!(program instanceof PdfStream)) return void 0;
|
|
83
|
+
const bytes = file.streamData(program);
|
|
84
|
+
return bytes.length > 0 ? bytes : void 0;
|
|
85
|
+
}
|
|
86
|
+
/** §9.7.4 — a `/Type0` font's descendant CIDFont, which owns the descriptor. */
|
|
87
|
+
function descendant(file, fontDict) {
|
|
88
|
+
const list = file.resolve(fontDict.get("DescendantFonts") ?? PDF_NULL);
|
|
89
|
+
if (!Array.isArray(list)) return void 0;
|
|
90
|
+
const first = file.resolve(list[0] ?? PDF_NULL);
|
|
91
|
+
return first instanceof Map ? first : void 0;
|
|
92
|
+
}
|
|
93
|
+
//#endregion
|
|
94
|
+
export { collectEmbeddedFonts, embeddedFontName };
|
|
@@ -1,5 +1,6 @@
|
|
|
1
|
-
import { BodyElement, SectionProperties } from '../core/document-model/index.js';
|
|
1
|
+
import { BodyElement, SectionProperties, TextOutline } from '../core/document-model/index.js';
|
|
2
2
|
import { FlowDoc } from '../core/ir/flow.js';
|
|
3
|
+
import { FontRegistry } from '../core/font/index.js';
|
|
3
4
|
import { Loss, ResourceStore } from '../core/ir/index.js';
|
|
4
5
|
import { PdfImage } from './images.js';
|
|
5
6
|
import { PdfPage } from './document.js';
|
|
@@ -12,6 +13,20 @@ export interface Reconstruction {
|
|
|
12
13
|
readonly doc: FlowDoc;
|
|
13
14
|
readonly losses: ReadonlyArray<Loss>;
|
|
14
15
|
}
|
|
16
|
+
/**
|
|
17
|
+
* The corner a page's marks are measured from, in the SHOWN page's own y-up
|
|
18
|
+
* frame — see `./display`, which puts every mark into it.
|
|
19
|
+
*
|
|
20
|
+
* The shown page starts at its own origin, so `left` is zero and `top` is its
|
|
21
|
+
* height; the two are kept as a pair because a caller that has not been through
|
|
22
|
+
* `display` (none, today) would state something else.
|
|
23
|
+
*/
|
|
24
|
+
export interface PageFrame {
|
|
25
|
+
/** Left edge — subtract it to get an offset from the page. */
|
|
26
|
+
readonly left: number;
|
|
27
|
+
/** Top edge — subtract from it to flip into a top-down frame. */
|
|
28
|
+
readonly top: number;
|
|
29
|
+
}
|
|
15
30
|
/**
|
|
16
31
|
* Build a paragraph {@link BodyElement} from a single plain-text string,
|
|
17
32
|
* optionally at the given outline (heading) level. Empty text yields a
|
|
@@ -22,6 +37,18 @@ export declare function paragraphBlock(text: string, outlineLevel?: number): Bod
|
|
|
22
37
|
export interface TextSpan {
|
|
23
38
|
readonly text: string;
|
|
24
39
|
readonly href?: string;
|
|
40
|
+
/** The size the glyphs were SHOWN at (§9.3.1 Tf), so the run keeps it. */
|
|
41
|
+
readonly sizePt?: number;
|
|
42
|
+
/** §8.6.8 — the colour they were painted in, when it is not plain black. */
|
|
43
|
+
readonly colorHex?: string;
|
|
44
|
+
/** §9.6.2 — the face's own name, for a document that embeds its programs. */
|
|
45
|
+
readonly fontName?: string;
|
|
46
|
+
/** §9.3.6 — a line round the glyphs, when the page asked for one. */
|
|
47
|
+
readonly outline?: TextOutline;
|
|
48
|
+
/** §9.8.1 — the face was a bold one. */
|
|
49
|
+
readonly bold?: boolean;
|
|
50
|
+
/** §9.8.1 — the face was a slanted one. */
|
|
51
|
+
readonly italic?: boolean;
|
|
25
52
|
}
|
|
26
53
|
/**
|
|
27
54
|
* Build a paragraph {@link BodyElement} from positioned {@link TextSpan}s,
|
|
@@ -35,17 +62,45 @@ export declare function paragraphFromRuns(spans: ReadonlyArray<TextSpan>, outlin
|
|
|
35
62
|
* {@link BodyElement} that references them, sized in points from the placement
|
|
36
63
|
* CTM. `alt` becomes the block's alt text when given.
|
|
37
64
|
*/
|
|
38
|
-
export declare function imageBlock(image: PdfImage, resources: ResourceStore, alt?: string): BodyElement;
|
|
65
|
+
export declare function imageBlock(image: PdfImage, resources: ResourceStore, alt?: string, frame?: PageFrame, zOrder?: number): BodyElement;
|
|
66
|
+
/**
|
|
67
|
+
* A line of text as an anchored box, standing where the page set it.
|
|
68
|
+
*
|
|
69
|
+
* The flowed reconstruction reads a document OUT of a page: paragraphs in
|
|
70
|
+
* reading order, re-flowable, free to land wherever the next medium puts them.
|
|
71
|
+
* A form is not that document. 160F-2019.pdf is a grid of ruled boxes with a
|
|
72
|
+
* label in each, and a label means nothing an inch from the box it labels — the
|
|
73
|
+
* artwork is placed absolutely, so text that flows beside it lines up with none
|
|
74
|
+
* of it.
|
|
75
|
+
*
|
|
76
|
+
* @param spans The line's runs.
|
|
77
|
+
* @param box Its page-space rectangle (y-up, as PDF measures).
|
|
78
|
+
* @param frame The page's own corner, to measure the box off.
|
|
79
|
+
* @param zOrder Its place in the page's painting order.
|
|
80
|
+
* @param rotation60k §20.1.7.6 — how far the box turns about its own centre,
|
|
81
|
+
* for a baseline the page did not set flat.
|
|
82
|
+
* @returns A shape carrying the text, anchored where the glyphs were.
|
|
83
|
+
*/
|
|
84
|
+
export declare function positionedText(spans: ReadonlyArray<TextSpan>, box: {
|
|
85
|
+
x: number;
|
|
86
|
+
y: number;
|
|
87
|
+
width: number;
|
|
88
|
+
height: number;
|
|
89
|
+
}, frame: PageFrame, zOrder: number, rotation60k?: number): BodyElement;
|
|
39
90
|
/** Collapse losses sharing a `detail` message (the same colour space dropped on many pages). */
|
|
40
91
|
export declare function dedupeLosses(losses: ReadonlyArray<Loss>): Array<Loss>;
|
|
41
92
|
/**
|
|
42
93
|
* Turn a lifted {@link PdfVector} path (filled EP10 / stroked EP11) into a
|
|
43
94
|
* custom-geometry shape {@link BodyElement}. Page-space points (y-up) become
|
|
44
95
|
* path-space (bbox-relative, y-down); the shape is sized from the bounding box
|
|
45
|
-
* (plus the stroke thickness)
|
|
46
|
-
*
|
|
96
|
+
* (plus the stroke thickness). A fill becomes a solid fill, a stroke the outline.
|
|
97
|
+
*
|
|
98
|
+
* Given the page's frame the shape is ANCHORED where the page drew it, behind
|
|
99
|
+
* the text, rather than taking a place of its own in the flow. A drawing is not
|
|
100
|
+
* a paragraph: 22060_A1_01_Plans.pdf is one A3 sheet of vectors, and stacking
|
|
101
|
+
* its forty-nine paths one under another spilled it onto a second page.
|
|
47
102
|
*/
|
|
48
|
-
export declare function shapeBlock(v: PdfVector): BodyElement;
|
|
103
|
+
export declare function shapeBlock(v: PdfVector, frame?: PageFrame, zOrder?: number): BodyElement;
|
|
49
104
|
/**
|
|
50
105
|
* Derive the {@link SectionProperties} geometry from the source pages so a
|
|
51
106
|
* reconstructed PDF re-renders at its real page size and orientation rather than
|
|
@@ -65,4 +120,4 @@ export declare function sectionFromPdfPages(pages: ReadonlyArray<PdfPage>): Sect
|
|
|
65
120
|
* both reconstruction paths (the tagged fast-path EP3 and the heuristic layout
|
|
66
121
|
* path EP4).
|
|
67
122
|
*/
|
|
68
|
-
export declare function buildFlowDoc(body: ReadonlyArray<BodyElement>, resources?: ResourceStore, section?: SectionProperties): FlowDoc;
|
|
123
|
+
export declare function buildFlowDoc(body: ReadonlyArray<BodyElement>, resources?: ResourceStore, section?: SectionProperties, embeddedFonts?: ReadonlyMap<string, FontRegistry>): FlowDoc;
|