reamkit 1.6.0 → 1.9.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +13 -1
- package/dist/esm/core/converter/ream.d.ts +1 -0
- package/dist/esm/core/converter/ream.js +1 -1
- package/dist/esm/core/document-model/types.d.ts +3 -1
- package/dist/esm/core/drawingml/chart-serializer.d.ts +2 -0
- package/dist/esm/core/drawingml/chart-serializer.js +53 -0
- package/dist/esm/core/drawingml/shape-render.d.ts +3 -1
- package/dist/esm/core/drawingml/shape-render.js +27 -1
- package/dist/esm/core/vector.d.ts +10 -0
- package/dist/esm/excel/print-model.js +6 -3
- package/dist/esm/excel/xlsx-writer.js +81 -17
- package/dist/esm/html/html-writer.js +4 -2
- package/dist/esm/layout/styled-layout.js +5 -2
- package/dist/esm/pdf/shading.d.ts +12 -0
- package/dist/esm/pdf/shading.js +152 -0
- package/dist/esm/pdf/styled-page-emitter.js +29 -4
- package/dist/esm/pdf/vector-graphics.d.ts +1 -1
- package/dist/esm/pdf/vector-graphics.js +5 -3
- package/dist/esm/pdf-reader/ccitt.d.ts +15 -0
- package/dist/esm/pdf-reader/ccitt.js +394 -0
- package/dist/esm/pdf-reader/content.d.ts +43 -1
- package/dist/esm/pdf-reader/content.js +173 -4
- package/dist/esm/pdf-reader/crypto.d.ts +7 -0
- package/dist/esm/pdf-reader/crypto.js +609 -0
- package/dist/esm/pdf-reader/decrypt.d.ts +5 -0
- package/dist/esm/pdf-reader/decrypt.js +211 -0
- package/dist/esm/pdf-reader/document.d.ts +8 -1
- package/dist/esm/pdf-reader/document.js +210 -27
- package/dist/esm/pdf-reader/flow-build.d.ts +16 -1
- package/dist/esm/pdf-reader/flow-build.js +113 -3
- package/dist/esm/pdf-reader/image-decode.d.ts +15 -0
- package/dist/esm/pdf-reader/image-decode.js +550 -0
- package/dist/esm/pdf-reader/images.d.ts +16 -0
- package/dist/esm/pdf-reader/images.js +90 -0
- package/dist/esm/pdf-reader/layout.d.ts +2 -2
- package/dist/esm/pdf-reader/layout.js +86 -14
- package/dist/esm/pdf-reader/png-encode.d.ts +2 -0
- package/dist/esm/pdf-reader/png-encode.js +98 -0
- package/dist/esm/pdf-reader/predictor.d.ts +7 -0
- package/dist/esm/pdf-reader/predictor.js +61 -0
- package/dist/esm/pdf-reader/reader.d.ts +1 -1
- package/dist/esm/pdf-reader/reader.js +14 -7
- package/dist/esm/pdf-reader/shading.d.ts +3 -0
- package/dist/esm/pdf-reader/shading.js +147 -0
- package/dist/esm/pdf-reader/tagged.d.ts +2 -2
- package/dist/esm/pdf-reader/tagged.js +60 -7
- package/dist/esm/pdf-reader/text.js +91 -10
- package/dist/esm/pdf-reader/vector.d.ts +16 -0
- package/dist/esm/pdf-reader/vector.js +63 -0
- package/dist/esm/svg/svg-writer.js +12 -5
- package/dist/esm/word/docx-writer.js +104 -8
- package/dist/esm/word/drawing-parser.js +28 -17
- package/dist/esm/word/omml-serializer.d.ts +2 -0
- package/dist/esm/word/omml-serializer.js +48 -0
- package/package.json +1 -1
|
@@ -0,0 +1,90 @@
|
|
|
1
|
+
import { PDF_NULL, PdfName, PdfStream } from "../pdf/objects.js";
|
|
2
|
+
import { FEATURES } from "../core/ir/features.js";
|
|
3
|
+
import { interpretContent, multiply } from "./content.js";
|
|
4
|
+
import { decodePdfImage } from "./image-decode.js";
|
|
5
|
+
//#region src/pdf-reader/images.ts
|
|
6
|
+
var NO_FONTS = /* @__PURE__ */ new Map();
|
|
7
|
+
var MAX_FORM_DEPTH = 12;
|
|
8
|
+
var MAX_IMAGES = 4096;
|
|
9
|
+
function collectPageImages(file, page) {
|
|
10
|
+
const images = [];
|
|
11
|
+
const lossByDetail = /* @__PURE__ */ new Map();
|
|
12
|
+
const visiting = /* @__PURE__ */ new Set();
|
|
13
|
+
const addLoss = (severity, detail) => {
|
|
14
|
+
if (!lossByDetail.has(detail)) lossByDetail.set(detail, {
|
|
15
|
+
severity,
|
|
16
|
+
feature: FEATURES.images,
|
|
17
|
+
detail
|
|
18
|
+
});
|
|
19
|
+
};
|
|
20
|
+
const walk = (resources, content, baseCtm, depth, inheritedMcid) => {
|
|
21
|
+
const xobjects = resources ? file.get(resources, "XObject") : PDF_NULL;
|
|
22
|
+
const xobjDict = xobjects instanceof Map ? xobjects : void 0;
|
|
23
|
+
for (const placement of interpretContent(content, NO_FONTS, baseCtm).images) {
|
|
24
|
+
if (images.length >= MAX_IMAGES) return;
|
|
25
|
+
const stream = xobjDict ? file.resolve(xobjDict.get(placement.name) ?? PDF_NULL) : PDF_NULL;
|
|
26
|
+
if (!(stream instanceof PdfStream)) continue;
|
|
27
|
+
const subtype = nameOf(file.get(stream.dict, "Subtype"));
|
|
28
|
+
const mcid = placement.mcid ?? inheritedMcid;
|
|
29
|
+
if (subtype === "Image") {
|
|
30
|
+
const decoded = decodePdfImage(file, stream);
|
|
31
|
+
if (decoded.ok) {
|
|
32
|
+
images.push(geometry(placement.ctm, decoded, mcid));
|
|
33
|
+
if (decoded.degraded) addLoss("degraded", decoded.degraded);
|
|
34
|
+
} else addLoss(decoded.severity, decoded.detail);
|
|
35
|
+
} else if (subtype === "Form" && depth < MAX_FORM_DEPTH && !visiting.has(stream)) {
|
|
36
|
+
visiting.add(stream);
|
|
37
|
+
const formRes = file.get(stream.dict, "Resources");
|
|
38
|
+
walk(formRes instanceof Map ? formRes : resources, file.streamData(stream), multiply(matrixOf(file, stream.dict), placement.ctm), depth + 1, mcid);
|
|
39
|
+
visiting.delete(stream);
|
|
40
|
+
}
|
|
41
|
+
}
|
|
42
|
+
};
|
|
43
|
+
walk(page.resources, file.pageContent(page), [
|
|
44
|
+
1,
|
|
45
|
+
0,
|
|
46
|
+
0,
|
|
47
|
+
1,
|
|
48
|
+
0,
|
|
49
|
+
0
|
|
50
|
+
], 0, void 0);
|
|
51
|
+
return {
|
|
52
|
+
images,
|
|
53
|
+
losses: [...lossByDetail.values()]
|
|
54
|
+
};
|
|
55
|
+
}
|
|
56
|
+
function geometry(ctm, decoded, mcid) {
|
|
57
|
+
return {
|
|
58
|
+
bytes: decoded.bytes,
|
|
59
|
+
format: decoded.format,
|
|
60
|
+
widthPt: Math.hypot(ctm[0], ctm[1]) || 1,
|
|
61
|
+
heightPt: Math.hypot(ctm[2], ctm[3]) || 1,
|
|
62
|
+
x: ctm[4],
|
|
63
|
+
y: ctm[5],
|
|
64
|
+
...mcid !== void 0 ? { mcid } : {}
|
|
65
|
+
};
|
|
66
|
+
}
|
|
67
|
+
function matrixOf(file, dict) {
|
|
68
|
+
const m = file.resolve(dict.get("Matrix") ?? PDF_NULL);
|
|
69
|
+
if (Array.isArray(m) && m.length >= 6 && m.every((v) => typeof v === "number")) return [
|
|
70
|
+
m[0],
|
|
71
|
+
m[1],
|
|
72
|
+
m[2],
|
|
73
|
+
m[3],
|
|
74
|
+
m[4],
|
|
75
|
+
m[5]
|
|
76
|
+
];
|
|
77
|
+
return [
|
|
78
|
+
1,
|
|
79
|
+
0,
|
|
80
|
+
0,
|
|
81
|
+
1,
|
|
82
|
+
0,
|
|
83
|
+
0
|
|
84
|
+
];
|
|
85
|
+
}
|
|
86
|
+
function nameOf(v) {
|
|
87
|
+
return v instanceof PdfName ? v.value : "";
|
|
88
|
+
}
|
|
89
|
+
//#endregion
|
|
90
|
+
export { collectPageImages };
|
|
@@ -1,3 +1,3 @@
|
|
|
1
|
-
import { FlowDoc } from '../core/ir/flow.js';
|
|
2
1
|
import { PdfFile } from './document.js';
|
|
3
|
-
|
|
2
|
+
import { Reconstruction } from './flow-build.js';
|
|
3
|
+
export declare function reconstructByLayout(file: PdfFile): Reconstruction;
|
|
@@ -1,12 +1,77 @@
|
|
|
1
|
-
import {
|
|
1
|
+
import { ResourceStore } from "../core/ir/resources.js";
|
|
2
|
+
import { buildFlowDoc, dedupeLosses, imageBlock, paragraphFromRuns, shapeBlock } from "./flow-build.js";
|
|
3
|
+
import { collectPageImages } from "./images.js";
|
|
2
4
|
import { extractPageText } from "./text.js";
|
|
5
|
+
import { collectPageVectors } from "./vector.js";
|
|
3
6
|
//#region src/pdf-reader/layout.ts
|
|
4
7
|
function reconstructByLayout(file) {
|
|
5
|
-
const pages = file.pages()
|
|
6
|
-
const
|
|
8
|
+
const pages = file.pages();
|
|
9
|
+
const pageRuns = pages.map((page) => extractPageText(file, page));
|
|
10
|
+
const medianFont = median(pageRuns.flat().map((r) => r.fontSizePt).filter((s) => s > 0)) || 12;
|
|
11
|
+
const resources = new ResourceStore();
|
|
12
|
+
const losses = [];
|
|
7
13
|
const body = [];
|
|
8
|
-
|
|
9
|
-
|
|
14
|
+
pages.forEach((page, i) => {
|
|
15
|
+
const runs = pageRuns[i];
|
|
16
|
+
const [px0, , px1] = page.mediaBox;
|
|
17
|
+
const gutter = detectGutter(runs, Math.abs(px1 - px0));
|
|
18
|
+
const blocks = [];
|
|
19
|
+
const addColumn = (colRuns, col) => {
|
|
20
|
+
const lines = groupIntoLines(colRuns).filter((l) => l.text.length > 0);
|
|
21
|
+
for (const para of groupIntoParagraphs(lines)) blocks.push({
|
|
22
|
+
col,
|
|
23
|
+
top: para.top,
|
|
24
|
+
el: paragraphFromRuns(para.spans, headingLevel(para.fontSize, medianFont))
|
|
25
|
+
});
|
|
26
|
+
};
|
|
27
|
+
if (gutter !== void 0) {
|
|
28
|
+
addColumn(runs.filter((r) => r.x < gutter), 0);
|
|
29
|
+
addColumn(runs.filter((r) => r.x >= gutter), 1);
|
|
30
|
+
} else addColumn(runs, 0);
|
|
31
|
+
const colOf = (centerX) => gutter !== void 0 && centerX >= gutter ? 1 : 0;
|
|
32
|
+
const imgs = collectPageImages(file, page);
|
|
33
|
+
losses.push(...imgs.losses);
|
|
34
|
+
for (const img of imgs.images) blocks.push({
|
|
35
|
+
col: colOf(img.x + img.widthPt / 2),
|
|
36
|
+
top: img.y + img.heightPt,
|
|
37
|
+
el: imageBlock(img, resources)
|
|
38
|
+
});
|
|
39
|
+
for (const v of collectPageVectors(file, page)) blocks.push({
|
|
40
|
+
col: colOf((v.minX + v.maxX) / 2),
|
|
41
|
+
top: v.maxY,
|
|
42
|
+
el: shapeBlock(v)
|
|
43
|
+
});
|
|
44
|
+
blocks.sort((a, b) => a.col - b.col || b.top - a.top);
|
|
45
|
+
for (const block of blocks) body.push(block.el);
|
|
46
|
+
});
|
|
47
|
+
return {
|
|
48
|
+
doc: buildFlowDoc(body, resources),
|
|
49
|
+
losses: dedupeLosses(losses)
|
|
50
|
+
};
|
|
51
|
+
}
|
|
52
|
+
function detectGutter(runs, pageWidth) {
|
|
53
|
+
if (runs.length < 30 || pageWidth <= 0) return void 0;
|
|
54
|
+
const fontSize = median(runs.map((r) => r.fontSizePt).filter((s) => s > 0)) || 10;
|
|
55
|
+
const right = (r) => r.x + Math.max(1, r.text.length) * (r.fontSizePt || fontSize) * .5;
|
|
56
|
+
const intervals = runs.map((r) => [r.x, right(r)]).sort((a, b) => a[0] - b[0]);
|
|
57
|
+
const minX = intervals[0][0];
|
|
58
|
+
const span = Math.max(...intervals.map((iv) => iv[1])) - minX;
|
|
59
|
+
if (span < pageWidth * .5) return void 0;
|
|
60
|
+
let curEnd = intervals[0][1];
|
|
61
|
+
let gapMid = 0;
|
|
62
|
+
let gapW = 0;
|
|
63
|
+
for (const [l, r] of intervals) {
|
|
64
|
+
if (l - curEnd > gapW) {
|
|
65
|
+
gapW = l - curEnd;
|
|
66
|
+
gapMid = (curEnd + l) / 2;
|
|
67
|
+
}
|
|
68
|
+
if (r > curEnd) curEnd = r;
|
|
69
|
+
}
|
|
70
|
+
const frac = (gapMid - minX) / span;
|
|
71
|
+
if (gapW < fontSize * 3 || frac < .35 || frac > .65) return void 0;
|
|
72
|
+
const left = runs.filter((r) => r.x < gapMid).length;
|
|
73
|
+
if (left < runs.length * .25 || left > runs.length * .75) return void 0;
|
|
74
|
+
return gapMid;
|
|
10
75
|
}
|
|
11
76
|
function groupIntoLines(runs) {
|
|
12
77
|
const sorted = [...runs].sort((a, b) => b.y - a.y || a.x - b.x);
|
|
@@ -25,22 +90,28 @@ function groupIntoLines(runs) {
|
|
|
25
90
|
}
|
|
26
91
|
return clusters.map((c) => {
|
|
27
92
|
const ordered = c.runs.sort((a, b) => a.x - b.x);
|
|
93
|
+
const fontSize = c.fontSize || 10;
|
|
94
|
+
const spans = lineSpans(ordered, fontSize);
|
|
28
95
|
return {
|
|
29
96
|
y: c.y,
|
|
30
|
-
fontSize
|
|
31
|
-
text:
|
|
97
|
+
fontSize,
|
|
98
|
+
text: spans.map((s) => s.text).join("").replace(/\s+/g, " ").trim(),
|
|
99
|
+
spans
|
|
32
100
|
};
|
|
33
101
|
});
|
|
34
102
|
}
|
|
35
|
-
function
|
|
36
|
-
|
|
103
|
+
function lineSpans(runs, fontSize) {
|
|
104
|
+
const spans = [];
|
|
37
105
|
let prevEnd;
|
|
38
106
|
for (const run of runs) {
|
|
39
|
-
if (prevEnd !== void 0 && run.x - prevEnd > fontSize * .25) text
|
|
40
|
-
|
|
107
|
+
if (prevEnd !== void 0 && run.x - prevEnd > fontSize * .25) spans.push({ text: " " });
|
|
108
|
+
spans.push(run.href !== void 0 ? {
|
|
109
|
+
text: run.text,
|
|
110
|
+
href: run.href
|
|
111
|
+
} : { text: run.text });
|
|
41
112
|
prevEnd = run.x + run.text.length * (run.fontSizePt || fontSize) * .5;
|
|
42
113
|
}
|
|
43
|
-
return
|
|
114
|
+
return spans;
|
|
44
115
|
}
|
|
45
116
|
function groupIntoParagraphs(lines) {
|
|
46
117
|
const groups = [];
|
|
@@ -52,8 +123,9 @@ function groupIntoParagraphs(lines) {
|
|
|
52
123
|
prevY = line.y;
|
|
53
124
|
}
|
|
54
125
|
return groups.map((g) => ({
|
|
55
|
-
|
|
56
|
-
fontSize: Math.max(...g.map((l) => l.fontSize))
|
|
126
|
+
spans: g.flatMap((l, i) => i > 0 ? [{ text: " " }, ...l.spans] : [...l.spans]),
|
|
127
|
+
fontSize: Math.max(...g.map((l) => l.fontSize)),
|
|
128
|
+
top: g[0].y
|
|
57
129
|
}));
|
|
58
130
|
}
|
|
59
131
|
function headingLevel(fontSize, medianFont) {
|
|
@@ -0,0 +1,98 @@
|
|
|
1
|
+
import { zlibSync } from "fflate";
|
|
2
|
+
//#region src/pdf-reader/png-encode.ts
|
|
3
|
+
var COLOR_TYPE = {
|
|
4
|
+
gray: 0,
|
|
5
|
+
rgb: 2,
|
|
6
|
+
"gray-alpha": 4,
|
|
7
|
+
rgba: 6
|
|
8
|
+
};
|
|
9
|
+
var CHANNELS = {
|
|
10
|
+
gray: 1,
|
|
11
|
+
rgb: 3,
|
|
12
|
+
"gray-alpha": 2,
|
|
13
|
+
rgba: 4
|
|
14
|
+
};
|
|
15
|
+
var SIGNATURE = Uint8Array.from([
|
|
16
|
+
137,
|
|
17
|
+
80,
|
|
18
|
+
78,
|
|
19
|
+
71,
|
|
20
|
+
13,
|
|
21
|
+
10,
|
|
22
|
+
26,
|
|
23
|
+
10
|
|
24
|
+
]);
|
|
25
|
+
function encodePng(width, height, color, samples) {
|
|
26
|
+
const stride = width * CHANNELS[color];
|
|
27
|
+
const raw = new Uint8Array(height * (stride + 1));
|
|
28
|
+
for (let y = 0; y < height; y++) {
|
|
29
|
+
const dst = y * (stride + 1);
|
|
30
|
+
raw[dst] = 0;
|
|
31
|
+
raw.set(samples.subarray(y * stride, y * stride + stride), dst + 1);
|
|
32
|
+
}
|
|
33
|
+
const idat = zlibSync(raw);
|
|
34
|
+
const ihdr = new Uint8Array(13);
|
|
35
|
+
writeU32(ihdr, 0, width);
|
|
36
|
+
writeU32(ihdr, 4, height);
|
|
37
|
+
ihdr[8] = 8;
|
|
38
|
+
ihdr[9] = COLOR_TYPE[color];
|
|
39
|
+
ihdr[10] = 0;
|
|
40
|
+
ihdr[11] = 0;
|
|
41
|
+
ihdr[12] = 0;
|
|
42
|
+
return concat([
|
|
43
|
+
SIGNATURE,
|
|
44
|
+
chunk("IHDR", ihdr),
|
|
45
|
+
chunk("IDAT", idat),
|
|
46
|
+
chunk("IEND", new Uint8Array(0))
|
|
47
|
+
]);
|
|
48
|
+
}
|
|
49
|
+
function chunk(type, data) {
|
|
50
|
+
const typeBytes = Uint8Array.from([
|
|
51
|
+
type.charCodeAt(0),
|
|
52
|
+
type.charCodeAt(1),
|
|
53
|
+
type.charCodeAt(2),
|
|
54
|
+
type.charCodeAt(3)
|
|
55
|
+
]);
|
|
56
|
+
const out = new Uint8Array(12 + data.length);
|
|
57
|
+
writeU32(out, 0, data.length);
|
|
58
|
+
out.set(typeBytes, 4);
|
|
59
|
+
out.set(data, 8);
|
|
60
|
+
const crcInput = new Uint8Array(4 + data.length);
|
|
61
|
+
crcInput.set(typeBytes, 0);
|
|
62
|
+
crcInput.set(data, 4);
|
|
63
|
+
writeU32(out, 8 + data.length, crc32(crcInput));
|
|
64
|
+
return out;
|
|
65
|
+
}
|
|
66
|
+
function writeU32(buf, offset, value) {
|
|
67
|
+
buf[offset] = value >>> 24 & 255;
|
|
68
|
+
buf[offset + 1] = value >>> 16 & 255;
|
|
69
|
+
buf[offset + 2] = value >>> 8 & 255;
|
|
70
|
+
buf[offset + 3] = value & 255;
|
|
71
|
+
}
|
|
72
|
+
function concat(parts) {
|
|
73
|
+
let total = 0;
|
|
74
|
+
for (const p of parts) total += p.length;
|
|
75
|
+
const out = new Uint8Array(total);
|
|
76
|
+
let off = 0;
|
|
77
|
+
for (const p of parts) {
|
|
78
|
+
out.set(p, off);
|
|
79
|
+
off += p.length;
|
|
80
|
+
}
|
|
81
|
+
return out;
|
|
82
|
+
}
|
|
83
|
+
var crcTable;
|
|
84
|
+
function crc32(bytes) {
|
|
85
|
+
if (!crcTable) {
|
|
86
|
+
crcTable = new Uint32Array(256);
|
|
87
|
+
for (let n = 0; n < 256; n++) {
|
|
88
|
+
let c = n;
|
|
89
|
+
for (let k = 0; k < 8; k++) c = c & 1 ? 3988292384 ^ c >>> 1 : c >>> 1;
|
|
90
|
+
crcTable[n] = c >>> 0;
|
|
91
|
+
}
|
|
92
|
+
}
|
|
93
|
+
let crc = 4294967295;
|
|
94
|
+
for (const b of bytes) crc = crcTable[(crc ^ b) & 255] ^ crc >>> 8;
|
|
95
|
+
return (crc ^ 4294967295) >>> 0;
|
|
96
|
+
}
|
|
97
|
+
//#endregion
|
|
98
|
+
export { encodePng };
|
|
@@ -0,0 +1,61 @@
|
|
|
1
|
+
//#region src/pdf-reader/predictor.ts
|
|
2
|
+
function reversePredictor(data, p) {
|
|
3
|
+
if (p.predictor < 2) return data;
|
|
4
|
+
const bpp = Math.max(1, Math.ceil(p.colors * p.bitsPerComponent / 8));
|
|
5
|
+
const rowBytes = Math.ceil(p.colors * p.bitsPerComponent * p.columns / 8);
|
|
6
|
+
if (rowBytes <= 0) return data;
|
|
7
|
+
if (p.predictor === 2) {
|
|
8
|
+
if (p.bitsPerComponent !== 8) return data;
|
|
9
|
+
const rows = Math.floor(data.length / rowBytes);
|
|
10
|
+
const out = data.slice(0, rows * rowBytes);
|
|
11
|
+
for (let r = 0; r < rows; r++) {
|
|
12
|
+
const off = r * rowBytes;
|
|
13
|
+
for (let i = bpp; i < rowBytes; i++) out[off + i] = out[off + i] + out[off + i - bpp] & 255;
|
|
14
|
+
}
|
|
15
|
+
return out;
|
|
16
|
+
}
|
|
17
|
+
const stride = rowBytes + 1;
|
|
18
|
+
const rows = Math.floor(data.length / stride);
|
|
19
|
+
const out = new Uint8Array(rows * rowBytes);
|
|
20
|
+
let prev = new Uint8Array(rowBytes);
|
|
21
|
+
for (let r = 0; r < rows; r++) {
|
|
22
|
+
const ft = data[r * stride];
|
|
23
|
+
const src = r * stride + 1;
|
|
24
|
+
const dst = r * rowBytes;
|
|
25
|
+
for (let i = 0; i < rowBytes; i++) {
|
|
26
|
+
const x = data[src + i];
|
|
27
|
+
const a = i >= bpp ? out[dst + i - bpp] : 0;
|
|
28
|
+
const b = prev[i];
|
|
29
|
+
const c = i >= bpp ? prev[i - bpp] : 0;
|
|
30
|
+
let v;
|
|
31
|
+
switch (ft) {
|
|
32
|
+
case 1:
|
|
33
|
+
v = x + a;
|
|
34
|
+
break;
|
|
35
|
+
case 2:
|
|
36
|
+
v = x + b;
|
|
37
|
+
break;
|
|
38
|
+
case 3:
|
|
39
|
+
v = x + (a + b >> 1);
|
|
40
|
+
break;
|
|
41
|
+
case 4:
|
|
42
|
+
v = x + paeth(a, b, c);
|
|
43
|
+
break;
|
|
44
|
+
default: v = x;
|
|
45
|
+
}
|
|
46
|
+
out[dst + i] = v & 255;
|
|
47
|
+
}
|
|
48
|
+
prev = out.subarray(dst, dst + rowBytes);
|
|
49
|
+
}
|
|
50
|
+
return out;
|
|
51
|
+
}
|
|
52
|
+
function paeth(a, b, c) {
|
|
53
|
+
const p = a + b - c;
|
|
54
|
+
const pa = Math.abs(p - a);
|
|
55
|
+
const pb = Math.abs(p - b);
|
|
56
|
+
const pc = Math.abs(p - c);
|
|
57
|
+
if (pa <= pb && pa <= pc) return a;
|
|
58
|
+
return pb <= pc ? b : c;
|
|
59
|
+
}
|
|
60
|
+
//#endregion
|
|
61
|
+
export { reversePredictor };
|
|
@@ -1,4 +1,4 @@
|
|
|
1
1
|
import { DocumentReader, ReadResult } from '../core/ir/adapters.js';
|
|
2
2
|
import { FlowDoc } from '../core/ir/flow.js';
|
|
3
|
-
export declare function readPdf(bytes: Uint8Array): ReadResult<FlowDoc>;
|
|
3
|
+
export declare function readPdf(bytes: Uint8Array, password?: string): ReadResult<FlowDoc>;
|
|
4
4
|
export declare const pdfReader: DocumentReader<FlowDoc>;
|
|
@@ -8,23 +8,29 @@ function sniffPdf(bytes) {
|
|
|
8
8
|
for (let i = 0; i <= limit; i++) if (bytes[i] === 37 && bytes[i + 1] === 80 && bytes[i + 2] === 68 && bytes[i + 3] === 70 && bytes[i + 4] === 45) return true;
|
|
9
9
|
return false;
|
|
10
10
|
}
|
|
11
|
-
function readPdf(bytes) {
|
|
12
|
-
const file = PdfFile.parse(bytes);
|
|
11
|
+
function readPdf(bytes, password = "") {
|
|
12
|
+
const file = PdfFile.parse(bytes, password);
|
|
13
13
|
const losses = [];
|
|
14
|
+
if (file.encryptionUnsupported) losses.push({
|
|
15
|
+
severity: "dropped",
|
|
16
|
+
feature: FEATURES.text,
|
|
17
|
+
detail: "encrypted PDF — the user password was missing or incorrect, or the handler is unsupported"
|
|
18
|
+
});
|
|
14
19
|
const tagged = reconstructTaggedPdf(file);
|
|
15
|
-
const
|
|
20
|
+
const reconstruction = tagged ?? reconstructByLayout(file);
|
|
16
21
|
if (!tagged) losses.push({
|
|
17
22
|
severity: "degraded",
|
|
18
23
|
feature: FEATURES.text,
|
|
19
24
|
detail: "untagged PDF — text and headings reconstructed heuristically from glyph positions; structure is approximate"
|
|
20
25
|
});
|
|
26
|
+
losses.push(...reconstruction.losses);
|
|
21
27
|
losses.push({
|
|
22
28
|
severity: "dropped",
|
|
23
29
|
feature: FEATURES.images,
|
|
24
|
-
detail: "PDF
|
|
30
|
+
detail: "PDF clipping paths and bare-shading (sh) vector regions are not reconstructed"
|
|
25
31
|
});
|
|
26
32
|
return {
|
|
27
|
-
doc,
|
|
33
|
+
doc: reconstruction.doc,
|
|
28
34
|
losses
|
|
29
35
|
};
|
|
30
36
|
}
|
|
@@ -34,10 +40,11 @@ var pdfReader = {
|
|
|
34
40
|
supports: new Set([
|
|
35
41
|
FEATURES.text,
|
|
36
42
|
FEATURES.tables,
|
|
37
|
-
FEATURES.lists
|
|
43
|
+
FEATURES.lists,
|
|
44
|
+
FEATURES.images
|
|
38
45
|
]),
|
|
39
46
|
sniff: sniffPdf,
|
|
40
|
-
read: (bytes) => readPdf(bytes)
|
|
47
|
+
read: (bytes, opts) => readPdf(bytes, typeof opts?.password === "string" ? opts.password : "")
|
|
41
48
|
};
|
|
42
49
|
//#endregion
|
|
43
50
|
export { pdfReader };
|
|
@@ -0,0 +1,147 @@
|
|
|
1
|
+
import { PDF_NULL, PdfStream } from "../pdf/objects.js";
|
|
2
|
+
//#region src/pdf-reader/shading.ts
|
|
3
|
+
function buildShadingMap(file, page) {
|
|
4
|
+
const out = /* @__PURE__ */ new Map();
|
|
5
|
+
if (!page.resources) return out;
|
|
6
|
+
const patterns = file.get(page.resources, "Pattern");
|
|
7
|
+
if (!(patterns instanceof Map)) return out;
|
|
8
|
+
for (const [nm, value] of patterns) {
|
|
9
|
+
const pat = file.resolve(value);
|
|
10
|
+
if (!(pat instanceof Map)) continue;
|
|
11
|
+
const shading = dictOf(file.resolve(pat.get("Shading") ?? PDF_NULL));
|
|
12
|
+
if (!shading) continue;
|
|
13
|
+
const gradient = parseShading(file, shading);
|
|
14
|
+
if (gradient) out.set(nm, gradient);
|
|
15
|
+
}
|
|
16
|
+
return out;
|
|
17
|
+
}
|
|
18
|
+
function parseShading(file, sh) {
|
|
19
|
+
const type = numOf(file.get(sh, "ShadingType"));
|
|
20
|
+
if (type !== 2 && type !== 3) return void 0;
|
|
21
|
+
const stops = parseFunction(file, sh.get("Function"));
|
|
22
|
+
if (!stops || stops.length === 0) return void 0;
|
|
23
|
+
if (type === 3) return {
|
|
24
|
+
kind: "radial",
|
|
25
|
+
stops
|
|
26
|
+
};
|
|
27
|
+
const c = numArray(file, sh.get("Coords"));
|
|
28
|
+
return {
|
|
29
|
+
kind: "linear",
|
|
30
|
+
angle: c && c.length >= 4 ? (Math.atan2(-(c[3] - c[1]), c[2] - c[0]) * 180 / Math.PI % 360 + 360) % 360 : 0,
|
|
31
|
+
stops
|
|
32
|
+
};
|
|
33
|
+
}
|
|
34
|
+
function parseFunction(file, value) {
|
|
35
|
+
const resolved = value !== void 0 ? file.resolve(value) : void 0;
|
|
36
|
+
const dict = dictOf(resolved);
|
|
37
|
+
if (!dict) return void 0;
|
|
38
|
+
const type = numOf(file.get(dict, "FunctionType"));
|
|
39
|
+
if (type === 2) {
|
|
40
|
+
const c0 = colorOf(numArray(file, dict.get("C0")) ?? [0]);
|
|
41
|
+
const c1 = colorOf(numArray(file, dict.get("C1")) ?? [1]);
|
|
42
|
+
return [{
|
|
43
|
+
offset: 0,
|
|
44
|
+
colorHex: c0
|
|
45
|
+
}, {
|
|
46
|
+
offset: 1,
|
|
47
|
+
colorHex: c1
|
|
48
|
+
}];
|
|
49
|
+
}
|
|
50
|
+
if (type === 3) {
|
|
51
|
+
const fns = dict.get("Functions");
|
|
52
|
+
const subs = Array.isArray(fns) ? fns : void 0;
|
|
53
|
+
if (!subs) return void 0;
|
|
54
|
+
const bounds = numArray(file, dict.get("Bounds")) ?? [];
|
|
55
|
+
const domain = numArray(file, dict.get("Domain")) ?? [0, 1];
|
|
56
|
+
const d0 = domain[0] ?? 0;
|
|
57
|
+
const d1 = domain[domain.length - 1] ?? 1;
|
|
58
|
+
const edges = [
|
|
59
|
+
d0,
|
|
60
|
+
...bounds,
|
|
61
|
+
d1
|
|
62
|
+
];
|
|
63
|
+
const stops = [];
|
|
64
|
+
const span = d1 - d0 || 1;
|
|
65
|
+
for (let i = 0; i < subs.length; i++) {
|
|
66
|
+
const sub = parseFunction(file, subs[i]);
|
|
67
|
+
if (!sub || sub.length < 2) continue;
|
|
68
|
+
pushStop(stops, ((edges[i] ?? d0) - d0) / span, sub[0].colorHex);
|
|
69
|
+
pushStop(stops, ((edges[i + 1] ?? d1) - d0) / span, sub[sub.length - 1].colorHex);
|
|
70
|
+
}
|
|
71
|
+
return stops.length > 0 ? stops : void 0;
|
|
72
|
+
}
|
|
73
|
+
if (type === 0 && resolved instanceof PdfStream) return sampleFunction(file, resolved);
|
|
74
|
+
}
|
|
75
|
+
function sampleFunction(file, stream) {
|
|
76
|
+
const d = stream.dict;
|
|
77
|
+
const size = numArray(file, d.get("Size"));
|
|
78
|
+
const range = numArray(file, d.get("Range"));
|
|
79
|
+
const bps = numOf(file.get(d, "BitsPerSample"));
|
|
80
|
+
if (!size || !range || size.length < 1 || range.length < 2 || bps < 1) return void 0;
|
|
81
|
+
const n = size[0];
|
|
82
|
+
const comps = range.length / 2;
|
|
83
|
+
const data = file.streamData(stream);
|
|
84
|
+
const maxv = 2 ** bps - 1;
|
|
85
|
+
const bitAt = (sampleIdx, comp) => {
|
|
86
|
+
let bit = (sampleIdx * comps + comp) * bps;
|
|
87
|
+
let v = 0;
|
|
88
|
+
for (let k = 0; k < bps; k++) {
|
|
89
|
+
const byte = data[bit >> 3] ?? 0;
|
|
90
|
+
v = v << 1 | byte >> 7 - (bit & 7) & 1;
|
|
91
|
+
bit++;
|
|
92
|
+
}
|
|
93
|
+
return v;
|
|
94
|
+
};
|
|
95
|
+
const count = Math.min(Math.max(n, 2), 16);
|
|
96
|
+
const stops = [];
|
|
97
|
+
for (let s = 0; s < count; s++) {
|
|
98
|
+
const off = s / (count - 1);
|
|
99
|
+
const j = Math.round(off * (n - 1));
|
|
100
|
+
const c = [];
|
|
101
|
+
for (let comp = 0; comp < comps; comp++) {
|
|
102
|
+
const lo = range[comp * 2];
|
|
103
|
+
const hi = range[comp * 2 + 1];
|
|
104
|
+
c.push(lo + bitAt(j, comp) / maxv * (hi - lo));
|
|
105
|
+
}
|
|
106
|
+
pushStop(stops, off, colorOf(c));
|
|
107
|
+
}
|
|
108
|
+
return stops.length > 0 ? stops : void 0;
|
|
109
|
+
}
|
|
110
|
+
function dictOf(v) {
|
|
111
|
+
if (v instanceof PdfStream) return v.dict;
|
|
112
|
+
if (v instanceof Map) return v;
|
|
113
|
+
}
|
|
114
|
+
function numOf(v) {
|
|
115
|
+
return typeof v === "number" ? v : 0;
|
|
116
|
+
}
|
|
117
|
+
function numArray(file, v) {
|
|
118
|
+
const r = v !== void 0 ? file.resolve(v) : void 0;
|
|
119
|
+
if (!Array.isArray(r)) return void 0;
|
|
120
|
+
return r.map((x) => typeof x === "number" ? x : 0);
|
|
121
|
+
}
|
|
122
|
+
function colorOf(c) {
|
|
123
|
+
if (c.length >= 4) {
|
|
124
|
+
const k = c[3];
|
|
125
|
+
return hex255(255 * (1 - c[0]) * (1 - k), 255 * (1 - c[1]) * (1 - k), 255 * (1 - c[2]) * (1 - k));
|
|
126
|
+
}
|
|
127
|
+
if (c.length === 3) return hex255(c[0] * 255, c[1] * 255, c[2] * 255);
|
|
128
|
+
const g = (c[0] ?? 0) * 255;
|
|
129
|
+
return hex255(g, g, g);
|
|
130
|
+
}
|
|
131
|
+
function hex255(r, g, b) {
|
|
132
|
+
const h = (x) => Math.max(0, Math.min(255, Math.round(x))).toString(16).padStart(2, "0");
|
|
133
|
+
return (h(r) + h(g) + h(b)).toUpperCase();
|
|
134
|
+
}
|
|
135
|
+
function pushStop(stops, offset, colorHex) {
|
|
136
|
+
const o = Math.max(0, Math.min(1, offset));
|
|
137
|
+
const last = stops[stops.length - 1];
|
|
138
|
+
if (last && Math.abs(last.offset - o) < 1e-6) {
|
|
139
|
+
if (last.colorHex === colorHex) return;
|
|
140
|
+
}
|
|
141
|
+
stops.push({
|
|
142
|
+
offset: o,
|
|
143
|
+
colorHex
|
|
144
|
+
});
|
|
145
|
+
}
|
|
146
|
+
//#endregion
|
|
147
|
+
export { buildShadingMap };
|
|
@@ -1,3 +1,3 @@
|
|
|
1
|
-
import { FlowDoc } from '../core/ir/flow.js';
|
|
2
1
|
import { PdfFile } from './document.js';
|
|
3
|
-
|
|
2
|
+
import { Reconstruction } from './flow-build.js';
|
|
3
|
+
export declare function reconstructTaggedPdf(file: PdfFile): Reconstruction | undefined;
|