reamkit 1.7.0 → 1.9.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +13 -1
- package/dist/esm/core/converter/ream.d.ts +1 -0
- package/dist/esm/core/converter/ream.js +1 -1
- package/dist/esm/core/document-model/types.d.ts +3 -1
- package/dist/esm/core/drawingml/shape-render.d.ts +3 -1
- package/dist/esm/core/drawingml/shape-render.js +27 -1
- package/dist/esm/core/vector.d.ts +10 -0
- package/dist/esm/html/html-writer.js +4 -2
- package/dist/esm/layout/styled-layout.js +5 -2
- package/dist/esm/pdf/shading.d.ts +12 -0
- package/dist/esm/pdf/shading.js +152 -0
- package/dist/esm/pdf/styled-page-emitter.js +29 -4
- package/dist/esm/pdf/vector-graphics.d.ts +1 -1
- package/dist/esm/pdf/vector-graphics.js +5 -3
- package/dist/esm/pdf-reader/ccitt.d.ts +15 -0
- package/dist/esm/pdf-reader/ccitt.js +394 -0
- package/dist/esm/pdf-reader/content.d.ts +6 -2
- package/dist/esm/pdf-reader/content.js +54 -13
- package/dist/esm/pdf-reader/decrypt.d.ts +1 -1
- package/dist/esm/pdf-reader/decrypt.js +20 -8
- package/dist/esm/pdf-reader/document.d.ts +1 -1
- package/dist/esm/pdf-reader/document.js +4 -4
- package/dist/esm/pdf-reader/flow-build.js +17 -6
- package/dist/esm/pdf-reader/image-decode.js +115 -7
- package/dist/esm/pdf-reader/layout.js +45 -7
- package/dist/esm/pdf-reader/reader.d.ts +1 -1
- package/dist/esm/pdf-reader/reader.js +5 -5
- package/dist/esm/pdf-reader/shading.d.ts +3 -0
- package/dist/esm/pdf-reader/shading.js +147 -0
- package/dist/esm/pdf-reader/text.js +47 -13
- package/dist/esm/pdf-reader/vector.d.ts +5 -1
- package/dist/esm/pdf-reader/vector.js +15 -6
- package/dist/esm/svg/svg-writer.js +12 -5
- package/dist/esm/word/docx-writer.js +8 -1
- package/dist/esm/word/drawing-parser.js +28 -17
- package/package.json +1 -1
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
import { PDF_NULL, PdfHexString, PdfName, PdfStream } from "../pdf/objects.js";
|
|
2
2
|
import { reversePredictor } from "./predictor.js";
|
|
3
|
+
import { decodeCcitt } from "./ccitt.js";
|
|
3
4
|
import { encodePng } from "./png-encode.js";
|
|
4
5
|
import { unzlibSync } from "fflate";
|
|
5
6
|
//#region src/pdf-reader/image-decode.ts
|
|
@@ -32,9 +33,8 @@ function decodePdfImage(file, stream) {
|
|
|
32
33
|
heightPx: height,
|
|
33
34
|
degraded: "JPEG 2000 image — limited viewer support"
|
|
34
35
|
};
|
|
35
|
-
if (last === "
|
|
36
|
-
|
|
37
|
-
const decoded = decodeToSamples(file, stream, filters, width, height);
|
|
36
|
+
if (last === "JBIG2Decode") return fail("dropped", "JBIG2-encoded image not decoded");
|
|
37
|
+
const decoded = last === "CCITTFaxDecode" || last === "CCF" ? decodeCcittImage(file, stream, filters, width, height) : decodeToSamples(file, stream, filters, width, height);
|
|
38
38
|
if (typeof decoded === "string") return fail("dropped", decoded);
|
|
39
39
|
const alpha = decodeSMask(file, d, width, height);
|
|
40
40
|
const { color, samples } = alpha ? combineAlpha(decoded, alpha) : decoded;
|
|
@@ -56,6 +56,30 @@ function decodeToSamples(file, stream, filters, width, height) {
|
|
|
56
56
|
const decodeArr = decodeArrayOf(file, d);
|
|
57
57
|
return toColor(cs, unpackSamples(raw, width, height, cs.components, bpc), width * height, bpc, decodeArr);
|
|
58
58
|
}
|
|
59
|
+
function decodeCcittImage(file, stream, filters, width, height) {
|
|
60
|
+
const parms = decodeParmsOf(file, stream.dict);
|
|
61
|
+
const columns = (parms ? intOf(file.get(parms, "Columns")) : 0) || 1728;
|
|
62
|
+
const packed = decodeCcitt(applyChainExceptLast(filters, stream.data), {
|
|
63
|
+
k: parms ? intOf(file.get(parms, "K")) : 0,
|
|
64
|
+
columns,
|
|
65
|
+
rows: height,
|
|
66
|
+
byteAlign: parms ? boolOf(file.get(parms, "EncodedByteAlign")) : false
|
|
67
|
+
});
|
|
68
|
+
if (!packed) return "CCITT fax image not decoded (Group 3 2-D or malformed)";
|
|
69
|
+
const rowBytes = columns + 7 >> 3;
|
|
70
|
+
const samples = new Uint8Array(width * height);
|
|
71
|
+
for (let y = 0; y < height; y++) {
|
|
72
|
+
const rowOff = y * rowBytes;
|
|
73
|
+
for (let x = 0; x < width; x++) {
|
|
74
|
+
const black = x < columns ? packed[rowOff + (x >> 3)] >> 7 - (x & 7) & 1 : 0;
|
|
75
|
+
samples[y * width + x] = black ? 0 : 255;
|
|
76
|
+
}
|
|
77
|
+
}
|
|
78
|
+
return {
|
|
79
|
+
color: "gray",
|
|
80
|
+
samples
|
|
81
|
+
};
|
|
82
|
+
}
|
|
59
83
|
function resolveColorSpace(file, csVal) {
|
|
60
84
|
const cs = file.resolve(csVal ?? PDF_NULL);
|
|
61
85
|
if (cs instanceof PdfName) return namedColorSpace(cs.value);
|
|
@@ -296,18 +320,23 @@ function filterNames(file, d) {
|
|
|
296
320
|
}
|
|
297
321
|
function decodeChain(file, stream, filters) {
|
|
298
322
|
let data = stream.data;
|
|
299
|
-
let
|
|
323
|
+
let mayPredict = false;
|
|
300
324
|
for (const f of filters) if (f === "FlateDecode" || f === "Fl") try {
|
|
301
325
|
data = unzlibSync(data);
|
|
302
|
-
|
|
326
|
+
mayPredict = true;
|
|
303
327
|
} catch {
|
|
304
328
|
return;
|
|
305
329
|
}
|
|
306
|
-
else if (f === "
|
|
330
|
+
else if (f === "LZWDecode" || f === "LZW") {
|
|
331
|
+
const dec = lzwDecode(data, lzwEarlyChange(file, stream.dict));
|
|
332
|
+
if (!dec) return void 0;
|
|
333
|
+
data = dec;
|
|
334
|
+
mayPredict = true;
|
|
335
|
+
} else if (f === "RunLengthDecode" || f === "RL") data = runLengthDecode(data);
|
|
307
336
|
else if (f === "ASCII85Decode" || f === "A85") data = ascii85Decode(data);
|
|
308
337
|
else if (f === "ASCIIHexDecode" || f === "AHx") data = asciiHexDecode(data);
|
|
309
338
|
else return;
|
|
310
|
-
return
|
|
339
|
+
return mayPredict ? applyPredictor(file, stream.dict, data) : data;
|
|
311
340
|
}
|
|
312
341
|
function applyChainExceptLast(filters, raw) {
|
|
313
342
|
let data = raw;
|
|
@@ -316,6 +345,7 @@ function applyChainExceptLast(filters, raw) {
|
|
|
316
345
|
if (f === "FlateDecode" || f === "Fl") try {
|
|
317
346
|
data = unzlibSync(data);
|
|
318
347
|
} catch {}
|
|
348
|
+
else if (f === "LZWDecode" || f === "LZW") data = lzwDecode(data, 1) ?? data;
|
|
319
349
|
else if (f === "RunLengthDecode" || f === "RL") data = runLengthDecode(data);
|
|
320
350
|
else if (f === "ASCII85Decode" || f === "A85") data = ascii85Decode(data);
|
|
321
351
|
else if (f === "ASCIIHexDecode" || f === "AHx") data = asciiHexDecode(data);
|
|
@@ -334,6 +364,84 @@ function applyPredictor(file, d, data) {
|
|
|
334
364
|
columns: intOf(file.get(parms, "Columns")) || 1
|
|
335
365
|
});
|
|
336
366
|
}
|
|
367
|
+
var MAX_LZW_OUT = MAX_PIXELS * 4;
|
|
368
|
+
function lzwDecode(data, earlyChange) {
|
|
369
|
+
let out = new Uint8Array(4096);
|
|
370
|
+
let outLen = 0;
|
|
371
|
+
const emit = (e) => {
|
|
372
|
+
if (outLen + e.length > MAX_LZW_OUT) return false;
|
|
373
|
+
if (outLen + e.length > out.length) {
|
|
374
|
+
let cap = out.length * 2;
|
|
375
|
+
while (cap < outLen + e.length) cap *= 2;
|
|
376
|
+
const grown = new Uint8Array(cap);
|
|
377
|
+
grown.set(out.subarray(0, outLen));
|
|
378
|
+
out = grown;
|
|
379
|
+
}
|
|
380
|
+
out.set(e, outLen);
|
|
381
|
+
outLen += e.length;
|
|
382
|
+
return true;
|
|
383
|
+
};
|
|
384
|
+
let bitBuffer = 0;
|
|
385
|
+
let bitCount = 0;
|
|
386
|
+
let pos = 0;
|
|
387
|
+
let codeLength = 9;
|
|
388
|
+
const readCode = () => {
|
|
389
|
+
while (bitCount < codeLength) {
|
|
390
|
+
if (pos >= data.length) return -1;
|
|
391
|
+
bitBuffer = (bitBuffer << 8 | data[pos++]) >>> 0;
|
|
392
|
+
bitCount += 8;
|
|
393
|
+
}
|
|
394
|
+
bitCount -= codeLength;
|
|
395
|
+
return bitBuffer >>> bitCount & (1 << codeLength) - 1;
|
|
396
|
+
};
|
|
397
|
+
const dict = new Array(4096);
|
|
398
|
+
let nextCode = 258;
|
|
399
|
+
let prev = -1;
|
|
400
|
+
const reset = () => {
|
|
401
|
+
for (let i = 0; i < 256; i++) dict[i] = Uint8Array.of(i);
|
|
402
|
+
nextCode = 258;
|
|
403
|
+
codeLength = 9;
|
|
404
|
+
prev = -1;
|
|
405
|
+
};
|
|
406
|
+
reset();
|
|
407
|
+
for (;;) {
|
|
408
|
+
const code = readCode();
|
|
409
|
+
if (code < 0 || code === 257) break;
|
|
410
|
+
if (code === 256) {
|
|
411
|
+
reset();
|
|
412
|
+
continue;
|
|
413
|
+
}
|
|
414
|
+
if (prev < 0) {
|
|
415
|
+
const first = dict[code];
|
|
416
|
+
if (!first || !emit(first)) break;
|
|
417
|
+
prev = code;
|
|
418
|
+
continue;
|
|
419
|
+
}
|
|
420
|
+
const prevEntry = dict[prev];
|
|
421
|
+
let entry = code < nextCode ? dict[code] : void 0;
|
|
422
|
+
if (!entry) {
|
|
423
|
+
entry = new Uint8Array(prevEntry.length + 1);
|
|
424
|
+
entry.set(prevEntry);
|
|
425
|
+
entry[prevEntry.length] = prevEntry[0];
|
|
426
|
+
}
|
|
427
|
+
if (!emit(entry)) break;
|
|
428
|
+
if (nextCode < 4096) {
|
|
429
|
+
const added = new Uint8Array(prevEntry.length + 1);
|
|
430
|
+
added.set(prevEntry);
|
|
431
|
+
added[prevEntry.length] = entry[0];
|
|
432
|
+
dict[nextCode++] = added;
|
|
433
|
+
if (nextCode + earlyChange === 512) codeLength = 10;
|
|
434
|
+
else if (nextCode + earlyChange === 1024) codeLength = 11;
|
|
435
|
+
else if (nextCode + earlyChange === 2048) codeLength = 12;
|
|
436
|
+
}
|
|
437
|
+
prev = code;
|
|
438
|
+
}
|
|
439
|
+
return out.subarray(0, outLen);
|
|
440
|
+
}
|
|
441
|
+
function lzwEarlyChange(file, d) {
|
|
442
|
+
const ec = decodeParmsOf(file, d)?.get("EarlyChange");
|
|
443
|
+
return ec !== void 0 && file.resolve(ec) === 0 ? 0 : 1;
|
|
444
|
+
}
|
|
337
445
|
function runLengthDecode(data) {
|
|
338
446
|
const out = [];
|
|
339
447
|
let i = 0;
|
|
@@ -6,28 +6,42 @@ import { collectPageVectors } from "./vector.js";
|
|
|
6
6
|
//#region src/pdf-reader/layout.ts
|
|
7
7
|
function reconstructByLayout(file) {
|
|
8
8
|
const pages = file.pages();
|
|
9
|
-
const
|
|
10
|
-
const medianFont = median(
|
|
9
|
+
const pageRuns = pages.map((page) => extractPageText(file, page));
|
|
10
|
+
const medianFont = median(pageRuns.flat().map((r) => r.fontSizePt).filter((s) => s > 0)) || 12;
|
|
11
11
|
const resources = new ResourceStore();
|
|
12
12
|
const losses = [];
|
|
13
13
|
const body = [];
|
|
14
14
|
pages.forEach((page, i) => {
|
|
15
|
+
const runs = pageRuns[i];
|
|
16
|
+
const [px0, , px1] = page.mediaBox;
|
|
17
|
+
const gutter = detectGutter(runs, Math.abs(px1 - px0));
|
|
15
18
|
const blocks = [];
|
|
16
|
-
|
|
17
|
-
|
|
18
|
-
|
|
19
|
-
|
|
19
|
+
const addColumn = (colRuns, col) => {
|
|
20
|
+
const lines = groupIntoLines(colRuns).filter((l) => l.text.length > 0);
|
|
21
|
+
for (const para of groupIntoParagraphs(lines)) blocks.push({
|
|
22
|
+
col,
|
|
23
|
+
top: para.top,
|
|
24
|
+
el: paragraphFromRuns(para.spans, headingLevel(para.fontSize, medianFont))
|
|
25
|
+
});
|
|
26
|
+
};
|
|
27
|
+
if (gutter !== void 0) {
|
|
28
|
+
addColumn(runs.filter((r) => r.x < gutter), 0);
|
|
29
|
+
addColumn(runs.filter((r) => r.x >= gutter), 1);
|
|
30
|
+
} else addColumn(runs, 0);
|
|
31
|
+
const colOf = (centerX) => gutter !== void 0 && centerX >= gutter ? 1 : 0;
|
|
20
32
|
const imgs = collectPageImages(file, page);
|
|
21
33
|
losses.push(...imgs.losses);
|
|
22
34
|
for (const img of imgs.images) blocks.push({
|
|
35
|
+
col: colOf(img.x + img.widthPt / 2),
|
|
23
36
|
top: img.y + img.heightPt,
|
|
24
37
|
el: imageBlock(img, resources)
|
|
25
38
|
});
|
|
26
39
|
for (const v of collectPageVectors(file, page)) blocks.push({
|
|
40
|
+
col: colOf((v.minX + v.maxX) / 2),
|
|
27
41
|
top: v.maxY,
|
|
28
42
|
el: shapeBlock(v)
|
|
29
43
|
});
|
|
30
|
-
blocks.sort((a, b) => b.top - a.top);
|
|
44
|
+
blocks.sort((a, b) => a.col - b.col || b.top - a.top);
|
|
31
45
|
for (const block of blocks) body.push(block.el);
|
|
32
46
|
});
|
|
33
47
|
return {
|
|
@@ -35,6 +49,30 @@ function reconstructByLayout(file) {
|
|
|
35
49
|
losses: dedupeLosses(losses)
|
|
36
50
|
};
|
|
37
51
|
}
|
|
52
|
+
function detectGutter(runs, pageWidth) {
|
|
53
|
+
if (runs.length < 30 || pageWidth <= 0) return void 0;
|
|
54
|
+
const fontSize = median(runs.map((r) => r.fontSizePt).filter((s) => s > 0)) || 10;
|
|
55
|
+
const right = (r) => r.x + Math.max(1, r.text.length) * (r.fontSizePt || fontSize) * .5;
|
|
56
|
+
const intervals = runs.map((r) => [r.x, right(r)]).sort((a, b) => a[0] - b[0]);
|
|
57
|
+
const minX = intervals[0][0];
|
|
58
|
+
const span = Math.max(...intervals.map((iv) => iv[1])) - minX;
|
|
59
|
+
if (span < pageWidth * .5) return void 0;
|
|
60
|
+
let curEnd = intervals[0][1];
|
|
61
|
+
let gapMid = 0;
|
|
62
|
+
let gapW = 0;
|
|
63
|
+
for (const [l, r] of intervals) {
|
|
64
|
+
if (l - curEnd > gapW) {
|
|
65
|
+
gapW = l - curEnd;
|
|
66
|
+
gapMid = (curEnd + l) / 2;
|
|
67
|
+
}
|
|
68
|
+
if (r > curEnd) curEnd = r;
|
|
69
|
+
}
|
|
70
|
+
const frac = (gapMid - minX) / span;
|
|
71
|
+
if (gapW < fontSize * 3 || frac < .35 || frac > .65) return void 0;
|
|
72
|
+
const left = runs.filter((r) => r.x < gapMid).length;
|
|
73
|
+
if (left < runs.length * .25 || left > runs.length * .75) return void 0;
|
|
74
|
+
return gapMid;
|
|
75
|
+
}
|
|
38
76
|
function groupIntoLines(runs) {
|
|
39
77
|
const sorted = [...runs].sort((a, b) => b.y - a.y || a.x - b.x);
|
|
40
78
|
const clusters = [];
|
|
@@ -1,4 +1,4 @@
|
|
|
1
1
|
import { DocumentReader, ReadResult } from '../core/ir/adapters.js';
|
|
2
2
|
import { FlowDoc } from '../core/ir/flow.js';
|
|
3
|
-
export declare function readPdf(bytes: Uint8Array): ReadResult<FlowDoc>;
|
|
3
|
+
export declare function readPdf(bytes: Uint8Array, password?: string): ReadResult<FlowDoc>;
|
|
4
4
|
export declare const pdfReader: DocumentReader<FlowDoc>;
|
|
@@ -8,13 +8,13 @@ function sniffPdf(bytes) {
|
|
|
8
8
|
for (let i = 0; i <= limit; i++) if (bytes[i] === 37 && bytes[i + 1] === 80 && bytes[i + 2] === 68 && bytes[i + 3] === 70 && bytes[i + 4] === 45) return true;
|
|
9
9
|
return false;
|
|
10
10
|
}
|
|
11
|
-
function readPdf(bytes) {
|
|
12
|
-
const file = PdfFile.parse(bytes);
|
|
11
|
+
function readPdf(bytes, password = "") {
|
|
12
|
+
const file = PdfFile.parse(bytes, password);
|
|
13
13
|
const losses = [];
|
|
14
14
|
if (file.encryptionUnsupported) losses.push({
|
|
15
15
|
severity: "dropped",
|
|
16
16
|
feature: FEATURES.text,
|
|
17
|
-
detail: "encrypted PDF —
|
|
17
|
+
detail: "encrypted PDF — the user password was missing or incorrect, or the handler is unsupported"
|
|
18
18
|
});
|
|
19
19
|
const tagged = reconstructTaggedPdf(file);
|
|
20
20
|
const reconstruction = tagged ?? reconstructByLayout(file);
|
|
@@ -27,7 +27,7 @@ function readPdf(bytes) {
|
|
|
27
27
|
losses.push({
|
|
28
28
|
severity: "dropped",
|
|
29
29
|
feature: FEATURES.images,
|
|
30
|
-
detail: "PDF
|
|
30
|
+
detail: "PDF clipping paths and bare-shading (sh) vector regions are not reconstructed"
|
|
31
31
|
});
|
|
32
32
|
return {
|
|
33
33
|
doc: reconstruction.doc,
|
|
@@ -44,7 +44,7 @@ var pdfReader = {
|
|
|
44
44
|
FEATURES.images
|
|
45
45
|
]),
|
|
46
46
|
sniff: sniffPdf,
|
|
47
|
-
read: (bytes) => readPdf(bytes)
|
|
47
|
+
read: (bytes, opts) => readPdf(bytes, typeof opts?.password === "string" ? opts.password : "")
|
|
48
48
|
};
|
|
49
49
|
//#endregion
|
|
50
50
|
export { pdfReader };
|
|
@@ -0,0 +1,147 @@
|
|
|
1
|
+
import { PDF_NULL, PdfStream } from "../pdf/objects.js";
|
|
2
|
+
//#region src/pdf-reader/shading.ts
|
|
3
|
+
function buildShadingMap(file, page) {
|
|
4
|
+
const out = /* @__PURE__ */ new Map();
|
|
5
|
+
if (!page.resources) return out;
|
|
6
|
+
const patterns = file.get(page.resources, "Pattern");
|
|
7
|
+
if (!(patterns instanceof Map)) return out;
|
|
8
|
+
for (const [nm, value] of patterns) {
|
|
9
|
+
const pat = file.resolve(value);
|
|
10
|
+
if (!(pat instanceof Map)) continue;
|
|
11
|
+
const shading = dictOf(file.resolve(pat.get("Shading") ?? PDF_NULL));
|
|
12
|
+
if (!shading) continue;
|
|
13
|
+
const gradient = parseShading(file, shading);
|
|
14
|
+
if (gradient) out.set(nm, gradient);
|
|
15
|
+
}
|
|
16
|
+
return out;
|
|
17
|
+
}
|
|
18
|
+
function parseShading(file, sh) {
|
|
19
|
+
const type = numOf(file.get(sh, "ShadingType"));
|
|
20
|
+
if (type !== 2 && type !== 3) return void 0;
|
|
21
|
+
const stops = parseFunction(file, sh.get("Function"));
|
|
22
|
+
if (!stops || stops.length === 0) return void 0;
|
|
23
|
+
if (type === 3) return {
|
|
24
|
+
kind: "radial",
|
|
25
|
+
stops
|
|
26
|
+
};
|
|
27
|
+
const c = numArray(file, sh.get("Coords"));
|
|
28
|
+
return {
|
|
29
|
+
kind: "linear",
|
|
30
|
+
angle: c && c.length >= 4 ? (Math.atan2(-(c[3] - c[1]), c[2] - c[0]) * 180 / Math.PI % 360 + 360) % 360 : 0,
|
|
31
|
+
stops
|
|
32
|
+
};
|
|
33
|
+
}
|
|
34
|
+
function parseFunction(file, value) {
|
|
35
|
+
const resolved = value !== void 0 ? file.resolve(value) : void 0;
|
|
36
|
+
const dict = dictOf(resolved);
|
|
37
|
+
if (!dict) return void 0;
|
|
38
|
+
const type = numOf(file.get(dict, "FunctionType"));
|
|
39
|
+
if (type === 2) {
|
|
40
|
+
const c0 = colorOf(numArray(file, dict.get("C0")) ?? [0]);
|
|
41
|
+
const c1 = colorOf(numArray(file, dict.get("C1")) ?? [1]);
|
|
42
|
+
return [{
|
|
43
|
+
offset: 0,
|
|
44
|
+
colorHex: c0
|
|
45
|
+
}, {
|
|
46
|
+
offset: 1,
|
|
47
|
+
colorHex: c1
|
|
48
|
+
}];
|
|
49
|
+
}
|
|
50
|
+
if (type === 3) {
|
|
51
|
+
const fns = dict.get("Functions");
|
|
52
|
+
const subs = Array.isArray(fns) ? fns : void 0;
|
|
53
|
+
if (!subs) return void 0;
|
|
54
|
+
const bounds = numArray(file, dict.get("Bounds")) ?? [];
|
|
55
|
+
const domain = numArray(file, dict.get("Domain")) ?? [0, 1];
|
|
56
|
+
const d0 = domain[0] ?? 0;
|
|
57
|
+
const d1 = domain[domain.length - 1] ?? 1;
|
|
58
|
+
const edges = [
|
|
59
|
+
d0,
|
|
60
|
+
...bounds,
|
|
61
|
+
d1
|
|
62
|
+
];
|
|
63
|
+
const stops = [];
|
|
64
|
+
const span = d1 - d0 || 1;
|
|
65
|
+
for (let i = 0; i < subs.length; i++) {
|
|
66
|
+
const sub = parseFunction(file, subs[i]);
|
|
67
|
+
if (!sub || sub.length < 2) continue;
|
|
68
|
+
pushStop(stops, ((edges[i] ?? d0) - d0) / span, sub[0].colorHex);
|
|
69
|
+
pushStop(stops, ((edges[i + 1] ?? d1) - d0) / span, sub[sub.length - 1].colorHex);
|
|
70
|
+
}
|
|
71
|
+
return stops.length > 0 ? stops : void 0;
|
|
72
|
+
}
|
|
73
|
+
if (type === 0 && resolved instanceof PdfStream) return sampleFunction(file, resolved);
|
|
74
|
+
}
|
|
75
|
+
function sampleFunction(file, stream) {
|
|
76
|
+
const d = stream.dict;
|
|
77
|
+
const size = numArray(file, d.get("Size"));
|
|
78
|
+
const range = numArray(file, d.get("Range"));
|
|
79
|
+
const bps = numOf(file.get(d, "BitsPerSample"));
|
|
80
|
+
if (!size || !range || size.length < 1 || range.length < 2 || bps < 1) return void 0;
|
|
81
|
+
const n = size[0];
|
|
82
|
+
const comps = range.length / 2;
|
|
83
|
+
const data = file.streamData(stream);
|
|
84
|
+
const maxv = 2 ** bps - 1;
|
|
85
|
+
const bitAt = (sampleIdx, comp) => {
|
|
86
|
+
let bit = (sampleIdx * comps + comp) * bps;
|
|
87
|
+
let v = 0;
|
|
88
|
+
for (let k = 0; k < bps; k++) {
|
|
89
|
+
const byte = data[bit >> 3] ?? 0;
|
|
90
|
+
v = v << 1 | byte >> 7 - (bit & 7) & 1;
|
|
91
|
+
bit++;
|
|
92
|
+
}
|
|
93
|
+
return v;
|
|
94
|
+
};
|
|
95
|
+
const count = Math.min(Math.max(n, 2), 16);
|
|
96
|
+
const stops = [];
|
|
97
|
+
for (let s = 0; s < count; s++) {
|
|
98
|
+
const off = s / (count - 1);
|
|
99
|
+
const j = Math.round(off * (n - 1));
|
|
100
|
+
const c = [];
|
|
101
|
+
for (let comp = 0; comp < comps; comp++) {
|
|
102
|
+
const lo = range[comp * 2];
|
|
103
|
+
const hi = range[comp * 2 + 1];
|
|
104
|
+
c.push(lo + bitAt(j, comp) / maxv * (hi - lo));
|
|
105
|
+
}
|
|
106
|
+
pushStop(stops, off, colorOf(c));
|
|
107
|
+
}
|
|
108
|
+
return stops.length > 0 ? stops : void 0;
|
|
109
|
+
}
|
|
110
|
+
function dictOf(v) {
|
|
111
|
+
if (v instanceof PdfStream) return v.dict;
|
|
112
|
+
if (v instanceof Map) return v;
|
|
113
|
+
}
|
|
114
|
+
function numOf(v) {
|
|
115
|
+
return typeof v === "number" ? v : 0;
|
|
116
|
+
}
|
|
117
|
+
function numArray(file, v) {
|
|
118
|
+
const r = v !== void 0 ? file.resolve(v) : void 0;
|
|
119
|
+
if (!Array.isArray(r)) return void 0;
|
|
120
|
+
return r.map((x) => typeof x === "number" ? x : 0);
|
|
121
|
+
}
|
|
122
|
+
function colorOf(c) {
|
|
123
|
+
if (c.length >= 4) {
|
|
124
|
+
const k = c[3];
|
|
125
|
+
return hex255(255 * (1 - c[0]) * (1 - k), 255 * (1 - c[1]) * (1 - k), 255 * (1 - c[2]) * (1 - k));
|
|
126
|
+
}
|
|
127
|
+
if (c.length === 3) return hex255(c[0] * 255, c[1] * 255, c[2] * 255);
|
|
128
|
+
const g = (c[0] ?? 0) * 255;
|
|
129
|
+
return hex255(g, g, g);
|
|
130
|
+
}
|
|
131
|
+
function hex255(r, g, b) {
|
|
132
|
+
const h = (x) => Math.max(0, Math.min(255, Math.round(x))).toString(16).padStart(2, "0");
|
|
133
|
+
return (h(r) + h(g) + h(b)).toUpperCase();
|
|
134
|
+
}
|
|
135
|
+
function pushStop(stops, offset, colorHex) {
|
|
136
|
+
const o = Math.max(0, Math.min(1, offset));
|
|
137
|
+
const last = stops[stops.length - 1];
|
|
138
|
+
if (last && Math.abs(last.offset - o) < 1e-6) {
|
|
139
|
+
if (last.colorHex === colorHex) return;
|
|
140
|
+
}
|
|
141
|
+
stops.push({
|
|
142
|
+
offset: o,
|
|
143
|
+
colorHex
|
|
144
|
+
});
|
|
145
|
+
}
|
|
146
|
+
//#endregion
|
|
147
|
+
export { buildShadingMap };
|
|
@@ -1,19 +1,11 @@
|
|
|
1
|
-
import { PdfName } from "../pdf/objects.js";
|
|
2
|
-
import { interpretContent } from "./content.js";
|
|
1
|
+
import { PDF_NULL, PdfName, PdfStream } from "../pdf/objects.js";
|
|
2
|
+
import { IDENTITY, interpretContent, multiply } from "./content.js";
|
|
3
3
|
import { buildContentFont } from "./font.js";
|
|
4
4
|
//#region src/pdf-reader/text.ts
|
|
5
|
+
var MAX_FORM_DEPTH = 8;
|
|
5
6
|
function extractPageText(file, page) {
|
|
6
|
-
const
|
|
7
|
-
|
|
8
|
-
const fontContainer = file.get(page.resources, "Font");
|
|
9
|
-
if (fontContainer instanceof Map) for (const [fontName, fontRef] of fontContainer) {
|
|
10
|
-
const fontDict = file.resolve(fontRef);
|
|
11
|
-
if (fontDict instanceof Map) try {
|
|
12
|
-
fonts.set(fontName, buildContentFont(file, fontDict));
|
|
13
|
-
} catch {}
|
|
14
|
-
}
|
|
15
|
-
}
|
|
16
|
-
const runs = interpretContent(file.pageContent(page), fonts).texts;
|
|
7
|
+
const runs = [];
|
|
8
|
+
collectRuns(file, page.resources, file.pageContent(page), IDENTITY, 0, /* @__PURE__ */ new Set(), runs);
|
|
17
9
|
const links = collectLinks(file, page);
|
|
18
10
|
if (links.length === 0) return runs;
|
|
19
11
|
return runs.map((run) => {
|
|
@@ -24,6 +16,48 @@ function extractPageText(file, page) {
|
|
|
24
16
|
} : run;
|
|
25
17
|
});
|
|
26
18
|
}
|
|
19
|
+
function collectRuns(file, resources, content, baseCtm, depth, visiting, out) {
|
|
20
|
+
const result = interpretContent(content, buildFonts(file, resources), baseCtm);
|
|
21
|
+
out.push(...result.texts);
|
|
22
|
+
if (depth >= MAX_FORM_DEPTH || !resources) return;
|
|
23
|
+
const xobjects = file.get(resources, "XObject");
|
|
24
|
+
if (!(xobjects instanceof Map)) return;
|
|
25
|
+
for (const placement of result.images) {
|
|
26
|
+
const stream = file.resolve(xobjects.get(placement.name) ?? PDF_NULL);
|
|
27
|
+
if (!(stream instanceof PdfStream) || visiting.has(stream)) continue;
|
|
28
|
+
const sub = file.get(stream.dict, "Subtype");
|
|
29
|
+
if (!(sub instanceof PdfName) || sub.value !== "Form") continue;
|
|
30
|
+
visiting.add(stream);
|
|
31
|
+
const formRes = file.get(stream.dict, "Resources");
|
|
32
|
+
collectRuns(file, formRes instanceof Map ? formRes : resources, file.streamData(stream), multiply(matrixOf(file, stream.dict), placement.ctm), depth + 1, visiting, out);
|
|
33
|
+
visiting.delete(stream);
|
|
34
|
+
}
|
|
35
|
+
}
|
|
36
|
+
function buildFonts(file, resources) {
|
|
37
|
+
const fonts = /* @__PURE__ */ new Map();
|
|
38
|
+
if (!resources) return fonts;
|
|
39
|
+
const fontContainer = file.get(resources, "Font");
|
|
40
|
+
if (!(fontContainer instanceof Map)) return fonts;
|
|
41
|
+
for (const [fontName, fontRef] of fontContainer) {
|
|
42
|
+
const fontDict = file.resolve(fontRef);
|
|
43
|
+
if (fontDict instanceof Map) try {
|
|
44
|
+
fonts.set(fontName, buildContentFont(file, fontDict));
|
|
45
|
+
} catch {}
|
|
46
|
+
}
|
|
47
|
+
return fonts;
|
|
48
|
+
}
|
|
49
|
+
function matrixOf(file, dict) {
|
|
50
|
+
const m = file.resolve(dict.get("Matrix") ?? PDF_NULL);
|
|
51
|
+
if (Array.isArray(m) && m.length >= 6 && m.every((v) => typeof v === "number")) return [
|
|
52
|
+
m[0],
|
|
53
|
+
m[1],
|
|
54
|
+
m[2],
|
|
55
|
+
m[3],
|
|
56
|
+
m[4],
|
|
57
|
+
m[5]
|
|
58
|
+
];
|
|
59
|
+
return IDENTITY;
|
|
60
|
+
}
|
|
27
61
|
function collectLinks(file, page) {
|
|
28
62
|
const annots = file.get(page.dict, "Annots");
|
|
29
63
|
if (!Array.isArray(annots)) return [];
|
|
@@ -1,8 +1,12 @@
|
|
|
1
1
|
import { PathSeg } from './content.js';
|
|
2
|
+
import { ShapeGradient } from '../core/vector.js';
|
|
2
3
|
import { PdfFile, PdfPage } from './document.js';
|
|
3
4
|
export interface PdfVector {
|
|
4
5
|
readonly segs: ReadonlyArray<PathSeg>;
|
|
5
|
-
readonly fillHex
|
|
6
|
+
readonly fillHex?: string;
|
|
7
|
+
readonly gradient?: ShapeGradient;
|
|
8
|
+
readonly strokeHex?: string;
|
|
9
|
+
readonly lineWidth?: number;
|
|
6
10
|
readonly minX: number;
|
|
7
11
|
readonly minY: number;
|
|
8
12
|
readonly maxX: number;
|
|
@@ -1,25 +1,34 @@
|
|
|
1
|
-
import { interpretContent } from "./content.js";
|
|
1
|
+
import { IDENTITY, interpretContent } from "./content.js";
|
|
2
|
+
import { buildShadingMap } from "./shading.js";
|
|
2
3
|
//#region src/pdf-reader/vector.ts
|
|
3
4
|
var NO_FONTS = /* @__PURE__ */ new Map();
|
|
4
5
|
var MIN_SIDE = 2;
|
|
5
6
|
var MIN_AREA = 16;
|
|
7
|
+
var MIN_STROKE_LEN = 6;
|
|
6
8
|
var MAX_VECTORS = 2e3;
|
|
7
9
|
function collectPageVectors(file, page) {
|
|
8
10
|
const [px0, py0, px1, py1] = page.mediaBox;
|
|
9
11
|
const pageArea = Math.max(1, Math.abs((px1 - px0) * (py1 - py0)));
|
|
12
|
+
const shadings = buildShadingMap(file, page);
|
|
10
13
|
const out = [];
|
|
11
|
-
for (const v of interpretContent(file.pageContent(page), NO_FONTS).vectors) {
|
|
14
|
+
for (const v of interpretContent(file.pageContent(page), NO_FONTS, IDENTITY, shadings).vectors) {
|
|
12
15
|
if (out.length >= MAX_VECTORS) break;
|
|
13
|
-
if (v.fillHex === "FFFFFF") continue;
|
|
14
16
|
const b = bbox(v.segs);
|
|
15
17
|
if (!b) continue;
|
|
16
18
|
const w = b.maxX - b.minX;
|
|
17
19
|
const h = b.maxY - b.minY;
|
|
18
|
-
|
|
19
|
-
|
|
20
|
+
const area = w * h;
|
|
21
|
+
const solidFill = v.fillHex !== void 0 && v.fillHex !== "FFFFFF";
|
|
22
|
+
const filled = (v.gradient !== void 0 || solidFill) && w >= MIN_SIDE && h >= MIN_SIDE && area >= MIN_AREA && area <= .85 * pageArea;
|
|
23
|
+
const stroked = v.strokeHex !== void 0 && v.strokeHex !== "FFFFFF" && Math.max(w, h) >= MIN_STROKE_LEN && area <= .85 * pageArea;
|
|
24
|
+
if (!filled && !stroked) continue;
|
|
20
25
|
out.push({
|
|
21
26
|
segs: v.segs,
|
|
22
|
-
fillHex: v.fillHex,
|
|
27
|
+
...filled ? v.gradient ? { gradient: v.gradient } : v.fillHex !== void 0 ? { fillHex: v.fillHex } : {} : {},
|
|
28
|
+
...stroked ? {
|
|
29
|
+
strokeHex: v.strokeHex,
|
|
30
|
+
...v.lineWidth !== void 0 ? { lineWidth: v.lineWidth } : {}
|
|
31
|
+
} : {},
|
|
23
32
|
...b,
|
|
24
33
|
...v.mcid !== void 0 ? { mcid: v.mcid } : {}
|
|
25
34
|
});
|
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
import { FEATURES } from "../core/ir/features.js";
|
|
2
2
|
import { svgPathData } from "../core/vector.js";
|
|
3
|
+
import { gradientSvgDef } from "../core/drawingml/shape-render.js";
|
|
3
4
|
import { paintPlan } from "../layout/page-doc.js";
|
|
4
5
|
import { toBase64 } from "../core/bytes.js";
|
|
5
6
|
//#region src/svg/svg-writer.ts
|
|
@@ -12,10 +13,11 @@ function writeSvg(laid, opts = {}) {
|
|
|
12
13
|
const parts = [];
|
|
13
14
|
parts.push(`<svg xmlns="http://www.w3.org/2000/svg" width="${fmt(width)}" height="${fmt(height)}" viewBox="0 0 ${fmt(width)} ${fmt(height)}">`);
|
|
14
15
|
let yOffset = 0;
|
|
16
|
+
const idc = { n: 0 };
|
|
15
17
|
laid.pages.forEach((page, i) => {
|
|
16
18
|
parts.push(`<g transform="translate(0 ${fmt(yOffset)})" data-page="${i + 1}">`);
|
|
17
19
|
parts.push(`<rect x="0" y="0" width="${fmt(page.width)}" height="${fmt(page.height)}" fill="#ffffff" stroke="#cccccc"/>`);
|
|
18
|
-
emitPage(parts, page, laid, losses);
|
|
20
|
+
emitPage(parts, page, laid, losses, idc);
|
|
19
21
|
parts.push("</g>");
|
|
20
22
|
yOffset += page.height + gap;
|
|
21
23
|
});
|
|
@@ -36,7 +38,7 @@ var svgWriter = {
|
|
|
36
38
|
]),
|
|
37
39
|
write: (doc, opts) => writeSvg(doc, opts ?? {})
|
|
38
40
|
};
|
|
39
|
-
function emitPage(out, page, laid, losses) {
|
|
41
|
+
function emitPage(out, page, laid, losses, idc) {
|
|
40
42
|
const plan = paintPlan(page.commands);
|
|
41
43
|
for (const f of plan.fills) out.push(`<rect x="${fmt(f.x)}" y="${fmt(f.y)}" width="${fmt(f.width)}" height="${fmt(f.height)}" fill="#${f.fillColorHex}"/>`);
|
|
42
44
|
for (const img of plan.images) {
|
|
@@ -71,7 +73,7 @@ function emitPage(out, page, laid, losses) {
|
|
|
71
73
|
];
|
|
72
74
|
out.push(`<line x1="${fmt(ax)}" y1="${fmt(ay)}" x2="${fmt(bx)}" y2="${fmt(by)}" stroke="#${b.borderColorHex}" stroke-width="${fmt(b.borderSizePt)}"/>`);
|
|
73
75
|
}
|
|
74
|
-
for (const sh of plan.shapes) emitShape(out, sh.shape);
|
|
76
|
+
for (const sh of plan.shapes) emitShape(out, sh.shape, idc);
|
|
75
77
|
for (const t of plan.lines) emitTextLine(out, t, losses);
|
|
76
78
|
}
|
|
77
79
|
function emitTextLine(out, item, losses) {
|
|
@@ -95,10 +97,15 @@ function emitTextLine(out, item, losses) {
|
|
|
95
97
|
x += tok.widthPt;
|
|
96
98
|
}
|
|
97
99
|
}
|
|
98
|
-
function emitShape(out, shape) {
|
|
100
|
+
function emitShape(out, shape, idc) {
|
|
99
101
|
const [a, b, c, d, e, f] = shape.transform;
|
|
100
102
|
const transform = `matrix(${fmt(a)} ${fmt(b)} ${fmt(c)} ${fmt(d)} ${fmt(e)} ${fmt(f)})`;
|
|
101
|
-
|
|
103
|
+
let fill;
|
|
104
|
+
if (shape.fillGradient) {
|
|
105
|
+
const id = `grad${idc.n++}`;
|
|
106
|
+
out.push(gradientSvgDef(id, shape.fillGradient));
|
|
107
|
+
fill = `url(#${id})`;
|
|
108
|
+
} else fill = shape.fillColorHex ? `#${shape.fillColorHex}` : "none";
|
|
102
109
|
const stroke = shape.stroke ? ` stroke="#${shape.stroke.colorHex}" stroke-width="${fmt(shape.stroke.widthPt)}"` : "";
|
|
103
110
|
for (const path of shape.paths) {
|
|
104
111
|
const d2 = pathData(path.segments);
|
|
@@ -359,7 +359,14 @@ function geomXml(g) {
|
|
|
359
359
|
return `<a:prstGeom prst="${escapeAttr(g.preset ?? "rect")}"><a:avLst>${gds}</a:avLst></a:prstGeom>`;
|
|
360
360
|
}
|
|
361
361
|
function fillXml(f) {
|
|
362
|
-
|
|
362
|
+
if (f.kind === "solid" && f.colorHex) return `<a:solidFill><a:srgbClr val="${f.colorHex}"/></a:solidFill>`;
|
|
363
|
+
if (f.kind === "gradient" && f.gradient) return gradFillXml(f.gradient);
|
|
364
|
+
return "<a:noFill/>";
|
|
365
|
+
}
|
|
366
|
+
function gradFillXml(g) {
|
|
367
|
+
return `<a:gradFill><a:gsLst>${g.stops.map((s) => {
|
|
368
|
+
return `<a:gs pos="${Math.round(Math.max(0, Math.min(1, s.offset)) * 1e5)}"><a:srgbClr val="${s.colorHex}"/></a:gs>`;
|
|
369
|
+
}).join("")}</a:gsLst>${g.kind === "radial" ? "<a:path path=\"circle\"/>" : `<a:lin ang="${Math.round(((g.angle ?? 0) % 360 + 360) % 360 * 6e4)}" scaled="1"/>`}</a:gradFill>`;
|
|
363
370
|
}
|
|
364
371
|
function lineXml(l) {
|
|
365
372
|
const w = l.width !== void 0 ? ` w="${Math.round(l.width * EMU_PER_PT)}"` : "";
|