reamkit 1.5.0 → 1.7.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +2 -2
- package/dist/esm/core/converter/facade.js +6 -1
- package/dist/esm/core/converter/ream.js +2 -1
- package/dist/esm/core/document-model/types.d.ts +9 -1
- package/dist/esm/core/drawingml/chart-serializer.d.ts +2 -0
- package/dist/esm/core/drawingml/chart-serializer.js +53 -0
- package/dist/esm/core/spreadsheet-model/index.d.ts +1 -1
- package/dist/esm/core/spreadsheet-model/types.d.ts +5 -0
- package/dist/esm/excel/column-bands.d.ts +7 -0
- package/dist/esm/excel/column-bands.js +87 -0
- package/dist/esm/excel/conditional-format.js +22 -0
- package/dist/esm/excel/print-model.js +33 -11
- package/dist/esm/excel/worksheet-parser.js +26 -0
- package/dist/esm/excel/xlsx-writer.js +88 -17
- package/dist/esm/html/html-writer.js +100 -5
- package/dist/esm/layout/styled-layout.js +88 -19
- package/dist/esm/pdf-reader/cmap.d.ts +5 -0
- package/dist/esm/pdf-reader/cmap.js +72 -0
- package/dist/esm/pdf-reader/content.d.ts +52 -0
- package/dist/esm/pdf-reader/content.js +420 -0
- package/dist/esm/pdf-reader/crypto.d.ts +7 -0
- package/dist/esm/pdf-reader/crypto.js +609 -0
- package/dist/esm/pdf-reader/decrypt.d.ts +5 -0
- package/dist/esm/pdf-reader/decrypt.js +199 -0
- package/dist/esm/pdf-reader/document.d.ts +30 -0
- package/dist/esm/pdf-reader/document.js +413 -0
- package/dist/esm/pdf-reader/flow-build.d.ts +19 -0
- package/dist/esm/pdf-reader/flow-build.js +126 -0
- package/dist/esm/pdf-reader/font.d.ts +4 -0
- package/dist/esm/pdf-reader/font.js +69 -0
- package/dist/esm/pdf-reader/image-decode.d.ts +15 -0
- package/dist/esm/pdf-reader/image-decode.js +442 -0
- package/dist/esm/pdf-reader/images.d.ts +16 -0
- package/dist/esm/pdf-reader/images.js +90 -0
- package/dist/esm/pdf-reader/layout.d.ts +3 -0
- package/dist/esm/pdf-reader/layout.js +103 -0
- package/dist/esm/pdf-reader/lexer.d.ts +43 -0
- package/dist/esm/pdf-reader/lexer.js +250 -0
- package/dist/esm/pdf-reader/parser.d.ts +10 -0
- package/dist/esm/pdf-reader/parser.js +86 -0
- package/dist/esm/pdf-reader/png-encode.d.ts +2 -0
- package/dist/esm/pdf-reader/png-encode.js +98 -0
- package/dist/esm/pdf-reader/predictor.d.ts +7 -0
- package/dist/esm/pdf-reader/predictor.js +61 -0
- package/dist/esm/pdf-reader/reader.d.ts +4 -0
- package/dist/esm/pdf-reader/reader.js +50 -0
- package/dist/esm/pdf-reader/struct-tree.d.ts +14 -0
- package/dist/esm/pdf-reader/struct-tree.js +92 -0
- package/dist/esm/pdf-reader/tagged.d.ts +3 -0
- package/dist/esm/pdf-reader/tagged.js +141 -0
- package/dist/esm/pdf-reader/text.d.ts +3 -0
- package/dist/esm/pdf-reader/text.js +65 -0
- package/dist/esm/pdf-reader/vector.d.ts +12 -0
- package/dist/esm/pdf-reader/vector.js +54 -0
- package/dist/esm/word/docx-writer.js +96 -7
- package/dist/esm/word/omml-serializer.d.ts +2 -0
- package/dist/esm/word/omml-serializer.js +48 -0
- package/package.json +2 -2
|
@@ -0,0 +1,61 @@
|
|
|
1
|
+
//#region src/pdf-reader/predictor.ts
|
|
2
|
+
function reversePredictor(data, p) {
|
|
3
|
+
if (p.predictor < 2) return data;
|
|
4
|
+
const bpp = Math.max(1, Math.ceil(p.colors * p.bitsPerComponent / 8));
|
|
5
|
+
const rowBytes = Math.ceil(p.colors * p.bitsPerComponent * p.columns / 8);
|
|
6
|
+
if (rowBytes <= 0) return data;
|
|
7
|
+
if (p.predictor === 2) {
|
|
8
|
+
if (p.bitsPerComponent !== 8) return data;
|
|
9
|
+
const rows = Math.floor(data.length / rowBytes);
|
|
10
|
+
const out = data.slice(0, rows * rowBytes);
|
|
11
|
+
for (let r = 0; r < rows; r++) {
|
|
12
|
+
const off = r * rowBytes;
|
|
13
|
+
for (let i = bpp; i < rowBytes; i++) out[off + i] = out[off + i] + out[off + i - bpp] & 255;
|
|
14
|
+
}
|
|
15
|
+
return out;
|
|
16
|
+
}
|
|
17
|
+
const stride = rowBytes + 1;
|
|
18
|
+
const rows = Math.floor(data.length / stride);
|
|
19
|
+
const out = new Uint8Array(rows * rowBytes);
|
|
20
|
+
let prev = new Uint8Array(rowBytes);
|
|
21
|
+
for (let r = 0; r < rows; r++) {
|
|
22
|
+
const ft = data[r * stride];
|
|
23
|
+
const src = r * stride + 1;
|
|
24
|
+
const dst = r * rowBytes;
|
|
25
|
+
for (let i = 0; i < rowBytes; i++) {
|
|
26
|
+
const x = data[src + i];
|
|
27
|
+
const a = i >= bpp ? out[dst + i - bpp] : 0;
|
|
28
|
+
const b = prev[i];
|
|
29
|
+
const c = i >= bpp ? prev[i - bpp] : 0;
|
|
30
|
+
let v;
|
|
31
|
+
switch (ft) {
|
|
32
|
+
case 1:
|
|
33
|
+
v = x + a;
|
|
34
|
+
break;
|
|
35
|
+
case 2:
|
|
36
|
+
v = x + b;
|
|
37
|
+
break;
|
|
38
|
+
case 3:
|
|
39
|
+
v = x + (a + b >> 1);
|
|
40
|
+
break;
|
|
41
|
+
case 4:
|
|
42
|
+
v = x + paeth(a, b, c);
|
|
43
|
+
break;
|
|
44
|
+
default: v = x;
|
|
45
|
+
}
|
|
46
|
+
out[dst + i] = v & 255;
|
|
47
|
+
}
|
|
48
|
+
prev = out.subarray(dst, dst + rowBytes);
|
|
49
|
+
}
|
|
50
|
+
return out;
|
|
51
|
+
}
|
|
52
|
+
function paeth(a, b, c) {
|
|
53
|
+
const p = a + b - c;
|
|
54
|
+
const pa = Math.abs(p - a);
|
|
55
|
+
const pb = Math.abs(p - b);
|
|
56
|
+
const pc = Math.abs(p - c);
|
|
57
|
+
if (pa <= pb && pa <= pc) return a;
|
|
58
|
+
return pb <= pc ? b : c;
|
|
59
|
+
}
|
|
60
|
+
//#endregion
|
|
61
|
+
export { reversePredictor };
|
|
@@ -0,0 +1,50 @@
|
|
|
1
|
+
import { FEATURES } from "../core/ir/features.js";
|
|
2
|
+
import { PdfFile } from "./document.js";
|
|
3
|
+
import { reconstructByLayout } from "./layout.js";
|
|
4
|
+
import { reconstructTaggedPdf } from "./tagged.js";
|
|
5
|
+
//#region src/pdf-reader/reader.ts
|
|
6
|
+
function sniffPdf(bytes) {
|
|
7
|
+
const limit = Math.min(bytes.length - 5, 1024);
|
|
8
|
+
for (let i = 0; i <= limit; i++) if (bytes[i] === 37 && bytes[i + 1] === 80 && bytes[i + 2] === 68 && bytes[i + 3] === 70 && bytes[i + 4] === 45) return true;
|
|
9
|
+
return false;
|
|
10
|
+
}
|
|
11
|
+
function readPdf(bytes) {
|
|
12
|
+
const file = PdfFile.parse(bytes);
|
|
13
|
+
const losses = [];
|
|
14
|
+
if (file.encryptionUnsupported) losses.push({
|
|
15
|
+
severity: "dropped",
|
|
16
|
+
feature: FEATURES.text,
|
|
17
|
+
detail: "encrypted PDF — a user password is required and was not supplied"
|
|
18
|
+
});
|
|
19
|
+
const tagged = reconstructTaggedPdf(file);
|
|
20
|
+
const reconstruction = tagged ?? reconstructByLayout(file);
|
|
21
|
+
if (!tagged) losses.push({
|
|
22
|
+
severity: "degraded",
|
|
23
|
+
feature: FEATURES.text,
|
|
24
|
+
detail: "untagged PDF — text and headings reconstructed heuristically from glyph positions; structure is approximate"
|
|
25
|
+
});
|
|
26
|
+
losses.push(...reconstruction.losses);
|
|
27
|
+
losses.push({
|
|
28
|
+
severity: "dropped",
|
|
29
|
+
feature: FEATURES.images,
|
|
30
|
+
detail: "PDF stroked / shaded vector graphics (lines, gradients, clips) are not reconstructed"
|
|
31
|
+
});
|
|
32
|
+
return {
|
|
33
|
+
doc: reconstruction.doc,
|
|
34
|
+
losses
|
|
35
|
+
};
|
|
36
|
+
}
|
|
37
|
+
var pdfReader = {
|
|
38
|
+
id: "pdf",
|
|
39
|
+
produces: "flow",
|
|
40
|
+
supports: new Set([
|
|
41
|
+
FEATURES.text,
|
|
42
|
+
FEATURES.tables,
|
|
43
|
+
FEATURES.lists,
|
|
44
|
+
FEATURES.images
|
|
45
|
+
]),
|
|
46
|
+
sniff: sniffPdf,
|
|
47
|
+
read: (bytes) => readPdf(bytes)
|
|
48
|
+
};
|
|
49
|
+
//#endregion
|
|
50
|
+
export { pdfReader };
|
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
import { PdfFile } from './document.js';
|
|
2
|
+
export interface StructMcid {
|
|
3
|
+
readonly page: number;
|
|
4
|
+
readonly mcid: number;
|
|
5
|
+
}
|
|
6
|
+
export interface StructNode {
|
|
7
|
+
readonly type: string;
|
|
8
|
+
readonly mcids: ReadonlyArray<StructMcid>;
|
|
9
|
+
readonly children: ReadonlyArray<StructNode>;
|
|
10
|
+
readonly alt?: string;
|
|
11
|
+
readonly colSpan?: number;
|
|
12
|
+
readonly rowSpan?: number;
|
|
13
|
+
}
|
|
14
|
+
export declare function readStructTree(file: PdfFile): StructNode | undefined;
|
|
@@ -0,0 +1,92 @@
|
|
|
1
|
+
import { PDF_NULL, PdfName } from "../pdf/objects.js";
|
|
2
|
+
//#region src/pdf-reader/struct-tree.ts
|
|
3
|
+
var MAX_NODES = 2e5;
|
|
4
|
+
function readStructTree(file) {
|
|
5
|
+
const stRoot = file.get(file.catalog, "StructTreeRoot");
|
|
6
|
+
if (!(stRoot instanceof Map)) return void 0;
|
|
7
|
+
const pageMap = /* @__PURE__ */ new Map();
|
|
8
|
+
file.pages().forEach((p, i) => pageMap.set(p.dict, i));
|
|
9
|
+
const pageIndexOf = (pgVal) => {
|
|
10
|
+
const pg = file.resolve(pgVal);
|
|
11
|
+
return pg instanceof Map ? pageMap.get(pg) : void 0;
|
|
12
|
+
};
|
|
13
|
+
const seen = /* @__PURE__ */ new Set();
|
|
14
|
+
const read = (value, parentPage) => {
|
|
15
|
+
const elem = file.resolve(value);
|
|
16
|
+
if (!(elem instanceof Map) || seen.has(elem) || seen.size > MAX_NODES) return void 0;
|
|
17
|
+
seen.add(elem);
|
|
18
|
+
const ownPage = pageIndexOf(elem.get("Pg") ?? PDF_NULL) ?? parentPage;
|
|
19
|
+
const mcids = [];
|
|
20
|
+
const children = [];
|
|
21
|
+
for (const kid of kidList(file, elem.get("K"))) {
|
|
22
|
+
const rk = file.resolve(kid);
|
|
23
|
+
if (typeof rk === "number") {
|
|
24
|
+
if (ownPage >= 0) mcids.push({
|
|
25
|
+
page: ownPage,
|
|
26
|
+
mcid: rk
|
|
27
|
+
});
|
|
28
|
+
} else if (rk instanceof Map) {
|
|
29
|
+
const kind = nameOf(rk.get("Type"));
|
|
30
|
+
if (kind === "MCR") {
|
|
31
|
+
const m = rk.get("MCID");
|
|
32
|
+
const page = pageIndexOf(rk.get("Pg") ?? PDF_NULL) ?? ownPage;
|
|
33
|
+
if (typeof m === "number" && page >= 0) mcids.push({
|
|
34
|
+
page,
|
|
35
|
+
mcid: m
|
|
36
|
+
});
|
|
37
|
+
} else if (kind === "OBJR") {} else {
|
|
38
|
+
const child = read(rk, ownPage);
|
|
39
|
+
if (child) children.push(child);
|
|
40
|
+
}
|
|
41
|
+
}
|
|
42
|
+
}
|
|
43
|
+
const alt = elem.get("Alt");
|
|
44
|
+
const { colSpan, rowSpan } = readSpans(file, elem.get("A") ?? PDF_NULL);
|
|
45
|
+
return {
|
|
46
|
+
type: nameOf(elem.get("S")),
|
|
47
|
+
mcids,
|
|
48
|
+
children,
|
|
49
|
+
...typeof alt === "string" ? { alt } : {},
|
|
50
|
+
...colSpan > 1 ? { colSpan } : {},
|
|
51
|
+
...rowSpan > 1 ? { rowSpan } : {}
|
|
52
|
+
};
|
|
53
|
+
};
|
|
54
|
+
const roots = kidList(file, stRoot.get("K")).map((k) => read(k, -1)).filter((n) => n !== void 0);
|
|
55
|
+
if (roots.length === 1) return roots[0];
|
|
56
|
+
return {
|
|
57
|
+
type: "Document",
|
|
58
|
+
mcids: [],
|
|
59
|
+
children: roots
|
|
60
|
+
};
|
|
61
|
+
}
|
|
62
|
+
function kidList(file, kVal) {
|
|
63
|
+
if (kVal === void 0) return [];
|
|
64
|
+
const k = file.resolve(kVal);
|
|
65
|
+
if (Array.isArray(k)) return k;
|
|
66
|
+
if (k === PDF_NULL) return [];
|
|
67
|
+
return [k];
|
|
68
|
+
}
|
|
69
|
+
function nameOf(v) {
|
|
70
|
+
return v instanceof PdfName ? v.value : "";
|
|
71
|
+
}
|
|
72
|
+
function readSpans(file, aVal) {
|
|
73
|
+
let colSpan = 1;
|
|
74
|
+
let rowSpan = 1;
|
|
75
|
+
const a = file.resolve(aVal);
|
|
76
|
+
const attrs = Array.isArray(a) ? a : [a];
|
|
77
|
+
for (const entry of attrs) {
|
|
78
|
+
const d = file.resolve(entry);
|
|
79
|
+
if (d instanceof Map) {
|
|
80
|
+
const cs = d.get("ColSpan");
|
|
81
|
+
const rs = d.get("RowSpan");
|
|
82
|
+
if (typeof cs === "number") colSpan = cs;
|
|
83
|
+
if (typeof rs === "number") rowSpan = rs;
|
|
84
|
+
}
|
|
85
|
+
}
|
|
86
|
+
return {
|
|
87
|
+
colSpan,
|
|
88
|
+
rowSpan
|
|
89
|
+
};
|
|
90
|
+
}
|
|
91
|
+
//#endregion
|
|
92
|
+
export { readStructTree };
|
|
@@ -0,0 +1,141 @@
|
|
|
1
|
+
import { pt } from "../core/ir/units.js";
|
|
2
|
+
import { ResourceStore } from "../core/ir/resources.js";
|
|
3
|
+
import { buildFlowDoc, dedupeLosses, imageBlock, paragraphBlock, paragraphFromRuns } from "./flow-build.js";
|
|
4
|
+
import { collectPageImages } from "./images.js";
|
|
5
|
+
import { extractPageText } from "./text.js";
|
|
6
|
+
import { readStructTree } from "./struct-tree.js";
|
|
7
|
+
//#region src/pdf-reader/tagged.ts
|
|
8
|
+
var ASSUMED_CONTENT_WIDTH_PT = 468;
|
|
9
|
+
function reconstructTaggedPdf(file) {
|
|
10
|
+
const root = readStructTree(file);
|
|
11
|
+
if (!root) return void 0;
|
|
12
|
+
const pages = file.pages();
|
|
13
|
+
const pageRuns = pages.map((page) => {
|
|
14
|
+
const byMcid = /* @__PURE__ */ new Map();
|
|
15
|
+
for (const run of extractPageText(file, page)) {
|
|
16
|
+
if (run.mcid === void 0) continue;
|
|
17
|
+
const list = byMcid.get(run.mcid);
|
|
18
|
+
if (list) list.push(run);
|
|
19
|
+
else byMcid.set(run.mcid, [run]);
|
|
20
|
+
}
|
|
21
|
+
return byMcid;
|
|
22
|
+
});
|
|
23
|
+
const runsOfMcid = (page, mcid) => pageRuns[page]?.get(mcid) ?? [];
|
|
24
|
+
const resources = new ResourceStore();
|
|
25
|
+
const pageImages = pages.map((page) => collectPageImages(file, page));
|
|
26
|
+
const imageLosses = dedupeLosses(pageImages.flatMap((p) => p.losses));
|
|
27
|
+
const imagesByMcid = pageImages.map((p) => {
|
|
28
|
+
const byMcid = /* @__PURE__ */ new Map();
|
|
29
|
+
for (const img of p.images) {
|
|
30
|
+
if (img.mcid === void 0) continue;
|
|
31
|
+
const list = byMcid.get(img.mcid);
|
|
32
|
+
if (list) list.push(img);
|
|
33
|
+
else byMcid.set(img.mcid, [img]);
|
|
34
|
+
}
|
|
35
|
+
return byMcid;
|
|
36
|
+
});
|
|
37
|
+
const emitted = /* @__PURE__ */ new Set();
|
|
38
|
+
const imagesForNode = (node) => node.mcids.flatMap(({ page, mcid }) => imagesByMcid[page]?.get(mcid) ?? []);
|
|
39
|
+
const textOf = (node) => squash(node.mcids.map(({ page, mcid }) => runsOfMcid(page, mcid).map((r) => r.text).join("")).join(" "));
|
|
40
|
+
const spansOf = (node) => {
|
|
41
|
+
const spans = [];
|
|
42
|
+
node.mcids.forEach(({ page, mcid }, i) => {
|
|
43
|
+
if (i > 0) spans.push({ text: " " });
|
|
44
|
+
for (const run of runsOfMcid(page, mcid)) spans.push(run.href !== void 0 ? {
|
|
45
|
+
text: run.text,
|
|
46
|
+
href: run.href
|
|
47
|
+
} : { text: run.text });
|
|
48
|
+
});
|
|
49
|
+
return spans;
|
|
50
|
+
};
|
|
51
|
+
const collectText = (node) => squash([textOf(node), ...node.children.map(collectText)].join(" "));
|
|
52
|
+
function emit(node, out) {
|
|
53
|
+
if (node.type === "Table") {
|
|
54
|
+
const table = buildTable(node);
|
|
55
|
+
if (table) out.push(table);
|
|
56
|
+
return;
|
|
57
|
+
}
|
|
58
|
+
if (node.type === "Figure") {
|
|
59
|
+
for (const img of imagesForNode(node)) {
|
|
60
|
+
emitted.add(img);
|
|
61
|
+
out.push(imageBlock(img, resources, node.alt));
|
|
62
|
+
}
|
|
63
|
+
return;
|
|
64
|
+
}
|
|
65
|
+
if (node.type === "LI") {
|
|
66
|
+
const text = collectText(node);
|
|
67
|
+
if (text.length > 0) out.push(paragraphBlock(text, void 0));
|
|
68
|
+
return;
|
|
69
|
+
}
|
|
70
|
+
if (node.children.length === 0) {
|
|
71
|
+
if (textOf(node).length > 0) out.push(paragraphFromRuns(spansOf(node), headingLevel(node.type)));
|
|
72
|
+
return;
|
|
73
|
+
}
|
|
74
|
+
for (const child of node.children) emit(child, out);
|
|
75
|
+
}
|
|
76
|
+
function buildTable(tableNode) {
|
|
77
|
+
const rows = [];
|
|
78
|
+
const collectRows = (n) => {
|
|
79
|
+
for (const child of n.children) if (child.type === "TR") rows.push(buildRow(child));
|
|
80
|
+
else if (child.type === "THead" || child.type === "TBody" || child.type === "TFoot") collectRows(child);
|
|
81
|
+
};
|
|
82
|
+
collectRows(tableNode);
|
|
83
|
+
if (rows.length === 0) return void 0;
|
|
84
|
+
const numCols = Math.max(1, ...rows.map((r) => r.cells.reduce((s, c) => s + (c.properties.colSpan ?? 1), 0)));
|
|
85
|
+
const colWidth = pt(Math.max(1, ASSUMED_CONTENT_WIDTH_PT / numCols));
|
|
86
|
+
return {
|
|
87
|
+
kind: "table",
|
|
88
|
+
table: {
|
|
89
|
+
properties: {},
|
|
90
|
+
grid: Array.from({ length: numCols }, () => colWidth),
|
|
91
|
+
rows
|
|
92
|
+
}
|
|
93
|
+
};
|
|
94
|
+
}
|
|
95
|
+
function buildRow(trNode) {
|
|
96
|
+
const cells = [];
|
|
97
|
+
let allHeader = false;
|
|
98
|
+
for (const cell of trNode.children) {
|
|
99
|
+
if (cell.type !== "TH" && cell.type !== "TD") continue;
|
|
100
|
+
if (cells.length === 0) allHeader = true;
|
|
101
|
+
if (cell.type !== "TH") allHeader = false;
|
|
102
|
+
const content = [];
|
|
103
|
+
for (const child of cell.children) emit(child, content);
|
|
104
|
+
if (content.length === 0) content.push(paragraphBlock(textOf(cell), void 0));
|
|
105
|
+
const colSpan = cell.colSpan ?? 1;
|
|
106
|
+
cells.push({
|
|
107
|
+
properties: colSpan > 1 ? { colSpan } : {},
|
|
108
|
+
content
|
|
109
|
+
});
|
|
110
|
+
}
|
|
111
|
+
return {
|
|
112
|
+
properties: allHeader && cells.length > 0 ? { isHeader: true } : {},
|
|
113
|
+
cells
|
|
114
|
+
};
|
|
115
|
+
}
|
|
116
|
+
const body = [];
|
|
117
|
+
emit(root, body);
|
|
118
|
+
const orphans = [];
|
|
119
|
+
pageImages.forEach((p, page) => {
|
|
120
|
+
for (const img of p.images) if (!emitted.has(img)) orphans.push({
|
|
121
|
+
page,
|
|
122
|
+
img
|
|
123
|
+
});
|
|
124
|
+
});
|
|
125
|
+
orphans.sort((a, b) => a.page - b.page || b.img.y - a.img.y);
|
|
126
|
+
for (const { img } of orphans) body.push(imageBlock(img, resources));
|
|
127
|
+
if (body.length === 0) return void 0;
|
|
128
|
+
return {
|
|
129
|
+
doc: buildFlowDoc(body, resources),
|
|
130
|
+
losses: imageLosses
|
|
131
|
+
};
|
|
132
|
+
}
|
|
133
|
+
function squash(text) {
|
|
134
|
+
return text.replace(/\s+/g, " ").trim();
|
|
135
|
+
}
|
|
136
|
+
function headingLevel(type) {
|
|
137
|
+
const m = /^H([1-6])$/.exec(type);
|
|
138
|
+
return m ? Number(m[1]) - 1 : void 0;
|
|
139
|
+
}
|
|
140
|
+
//#endregion
|
|
141
|
+
export { reconstructTaggedPdf };
|
|
@@ -0,0 +1,65 @@
|
|
|
1
|
+
import { PdfName } from "../pdf/objects.js";
|
|
2
|
+
import { interpretContent } from "./content.js";
|
|
3
|
+
import { buildContentFont } from "./font.js";
|
|
4
|
+
//#region src/pdf-reader/text.ts
|
|
5
|
+
function extractPageText(file, page) {
|
|
6
|
+
const fonts = /* @__PURE__ */ new Map();
|
|
7
|
+
if (page.resources) {
|
|
8
|
+
const fontContainer = file.get(page.resources, "Font");
|
|
9
|
+
if (fontContainer instanceof Map) for (const [fontName, fontRef] of fontContainer) {
|
|
10
|
+
const fontDict = file.resolve(fontRef);
|
|
11
|
+
if (fontDict instanceof Map) try {
|
|
12
|
+
fonts.set(fontName, buildContentFont(file, fontDict));
|
|
13
|
+
} catch {}
|
|
14
|
+
}
|
|
15
|
+
}
|
|
16
|
+
const runs = interpretContent(file.pageContent(page), fonts).texts;
|
|
17
|
+
const links = collectLinks(file, page);
|
|
18
|
+
if (links.length === 0) return runs;
|
|
19
|
+
return runs.map((run) => {
|
|
20
|
+
const link = links.find((l) => inRect(run.x, run.y, l.rect));
|
|
21
|
+
return link ? {
|
|
22
|
+
...run,
|
|
23
|
+
href: link.href
|
|
24
|
+
} : run;
|
|
25
|
+
});
|
|
26
|
+
}
|
|
27
|
+
function collectLinks(file, page) {
|
|
28
|
+
const annots = file.get(page.dict, "Annots");
|
|
29
|
+
if (!Array.isArray(annots)) return [];
|
|
30
|
+
const out = [];
|
|
31
|
+
for (const a of annots) {
|
|
32
|
+
const annot = file.resolve(a);
|
|
33
|
+
if (!(annot instanceof Map)) continue;
|
|
34
|
+
const sub = file.get(annot, "Subtype");
|
|
35
|
+
if (!(sub instanceof PdfName) || sub.value !== "Link") continue;
|
|
36
|
+
const rect = normRect(file.get(annot, "Rect"));
|
|
37
|
+
if (!rect) continue;
|
|
38
|
+
const action = file.get(annot, "A");
|
|
39
|
+
if (!(action instanceof Map)) continue;
|
|
40
|
+
const s = file.get(action, "S");
|
|
41
|
+
if (!(s instanceof PdfName) || s.value !== "URI") continue;
|
|
42
|
+
const uri = file.get(action, "URI");
|
|
43
|
+
if (typeof uri === "string" && uri.length > 0) out.push({
|
|
44
|
+
rect,
|
|
45
|
+
href: uri
|
|
46
|
+
});
|
|
47
|
+
}
|
|
48
|
+
return out;
|
|
49
|
+
}
|
|
50
|
+
function normRect(v) {
|
|
51
|
+
if (!Array.isArray(v) || v.length < 4) return void 0;
|
|
52
|
+
const n = v.slice(0, 4).map((x) => typeof x === "number" ? x : NaN);
|
|
53
|
+
if (n.some((x) => !Number.isFinite(x))) return void 0;
|
|
54
|
+
return [
|
|
55
|
+
Math.min(n[0], n[2]),
|
|
56
|
+
Math.min(n[1], n[3]),
|
|
57
|
+
Math.max(n[0], n[2]),
|
|
58
|
+
Math.max(n[1], n[3])
|
|
59
|
+
];
|
|
60
|
+
}
|
|
61
|
+
function inRect(x, y, r) {
|
|
62
|
+
return x >= r[0] - 1 && x <= r[2] + 1 && y >= r[1] - 2 && y <= r[3] + 2;
|
|
63
|
+
}
|
|
64
|
+
//#endregion
|
|
65
|
+
export { extractPageText };
|
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
import { PathSeg } from './content.js';
|
|
2
|
+
import { PdfFile, PdfPage } from './document.js';
|
|
3
|
+
export interface PdfVector {
|
|
4
|
+
readonly segs: ReadonlyArray<PathSeg>;
|
|
5
|
+
readonly fillHex: string;
|
|
6
|
+
readonly minX: number;
|
|
7
|
+
readonly minY: number;
|
|
8
|
+
readonly maxX: number;
|
|
9
|
+
readonly maxY: number;
|
|
10
|
+
readonly mcid?: number;
|
|
11
|
+
}
|
|
12
|
+
export declare function collectPageVectors(file: PdfFile, page: PdfPage): Array<PdfVector>;
|
|
@@ -0,0 +1,54 @@
|
|
|
1
|
+
import { interpretContent } from "./content.js";
|
|
2
|
+
//#region src/pdf-reader/vector.ts
|
|
3
|
+
var NO_FONTS = /* @__PURE__ */ new Map();
|
|
4
|
+
var MIN_SIDE = 2;
|
|
5
|
+
var MIN_AREA = 16;
|
|
6
|
+
var MAX_VECTORS = 2e3;
|
|
7
|
+
function collectPageVectors(file, page) {
|
|
8
|
+
const [px0, py0, px1, py1] = page.mediaBox;
|
|
9
|
+
const pageArea = Math.max(1, Math.abs((px1 - px0) * (py1 - py0)));
|
|
10
|
+
const out = [];
|
|
11
|
+
for (const v of interpretContent(file.pageContent(page), NO_FONTS).vectors) {
|
|
12
|
+
if (out.length >= MAX_VECTORS) break;
|
|
13
|
+
if (v.fillHex === "FFFFFF") continue;
|
|
14
|
+
const b = bbox(v.segs);
|
|
15
|
+
if (!b) continue;
|
|
16
|
+
const w = b.maxX - b.minX;
|
|
17
|
+
const h = b.maxY - b.minY;
|
|
18
|
+
if (w < MIN_SIDE || h < MIN_SIDE || w * h < MIN_AREA) continue;
|
|
19
|
+
if (w * h > .85 * pageArea) continue;
|
|
20
|
+
out.push({
|
|
21
|
+
segs: v.segs,
|
|
22
|
+
fillHex: v.fillHex,
|
|
23
|
+
...b,
|
|
24
|
+
...v.mcid !== void 0 ? { mcid: v.mcid } : {}
|
|
25
|
+
});
|
|
26
|
+
}
|
|
27
|
+
return out;
|
|
28
|
+
}
|
|
29
|
+
function bbox(segs) {
|
|
30
|
+
let minX = Infinity;
|
|
31
|
+
let minY = Infinity;
|
|
32
|
+
let maxX = -Infinity;
|
|
33
|
+
let maxY = -Infinity;
|
|
34
|
+
const add = (x, y) => {
|
|
35
|
+
minX = Math.min(minX, x);
|
|
36
|
+
minY = Math.min(minY, y);
|
|
37
|
+
maxX = Math.max(maxX, x);
|
|
38
|
+
maxY = Math.max(maxY, y);
|
|
39
|
+
};
|
|
40
|
+
for (const s of segs) if (s.op === "move" || s.op === "line") add(s.x, s.y);
|
|
41
|
+
else if (s.op === "cubic") {
|
|
42
|
+
add(s.x1, s.y1);
|
|
43
|
+
add(s.x2, s.y2);
|
|
44
|
+
add(s.x, s.y);
|
|
45
|
+
}
|
|
46
|
+
return Number.isFinite(minX) ? {
|
|
47
|
+
minX,
|
|
48
|
+
minY,
|
|
49
|
+
maxX,
|
|
50
|
+
maxY
|
|
51
|
+
} : void 0;
|
|
52
|
+
}
|
|
53
|
+
//#endregion
|
|
54
|
+
export { collectPageVectors };
|
|
@@ -4,6 +4,8 @@ import { EMPTY_STYLE_SHEET, resolveParagraphProperties, resolveRunProperties } f
|
|
|
4
4
|
import "../core/style-cascade/index.js";
|
|
5
5
|
import { buildOpcPackage } from "../core/opc/opc-writer.js";
|
|
6
6
|
import "../core/opc/index.js";
|
|
7
|
+
import { omathXml } from "./omml-serializer.js";
|
|
8
|
+
import { chartSpaceXml } from "../core/drawingml/chart-serializer.js";
|
|
7
9
|
//#region src/word/docx-writer.ts
|
|
8
10
|
var encoder = new TextEncoder();
|
|
9
11
|
var DEFAULT_RUN = resolveRunProperties({}, {}, EMPTY_STYLE_SHEET);
|
|
@@ -15,6 +17,14 @@ var REL_NUMBERING = "http://schemas.openxmlformats.org/officeDocument/2006/relat
|
|
|
15
17
|
var REL_HYPERLINK = "http://schemas.openxmlformats.org/officeDocument/2006/relationships/hyperlink";
|
|
16
18
|
var REL_IMAGE = "http://schemas.openxmlformats.org/officeDocument/2006/relationships/image";
|
|
17
19
|
var NUMBERING_PART = "word/numbering.xml";
|
|
20
|
+
var REL_FOOTNOTES = "http://schemas.openxmlformats.org/officeDocument/2006/relationships/footnotes";
|
|
21
|
+
var REL_ENDNOTES = "http://schemas.openxmlformats.org/officeDocument/2006/relationships/endnotes";
|
|
22
|
+
var FOOTNOTES_CONTENT_TYPE = "application/vnd.openxmlformats-officedocument.wordprocessingml.footnotes+xml";
|
|
23
|
+
var ENDNOTES_CONTENT_TYPE = "application/vnd.openxmlformats-officedocument.wordprocessingml.endnotes+xml";
|
|
24
|
+
var FOOTNOTES_PART = "word/footnotes.xml";
|
|
25
|
+
var ENDNOTES_PART = "word/endnotes.xml";
|
|
26
|
+
var REL_CHART = "http://schemas.openxmlformats.org/officeDocument/2006/relationships/chart";
|
|
27
|
+
var CHART_CONTENT_TYPE = "application/vnd.openxmlformats-officedocument.drawingml.chart+xml";
|
|
18
28
|
var EMU_PER_PT = 12700;
|
|
19
29
|
var RASTER_MEDIA = {
|
|
20
30
|
png: {
|
|
@@ -78,7 +88,10 @@ function writeDocx(flow) {
|
|
|
78
88
|
mediaParts: [],
|
|
79
89
|
mediaFileByResource: /* @__PURE__ */ new Map(),
|
|
80
90
|
bookmarkSeq: 0,
|
|
81
|
-
drawingSeq: 0
|
|
91
|
+
drawingSeq: 0,
|
|
92
|
+
chartParts: [],
|
|
93
|
+
chartSeq: 0,
|
|
94
|
+
...flow.charts ? { charts: flow.charts } : {}
|
|
82
95
|
};
|
|
83
96
|
const docScope = newScope();
|
|
84
97
|
const extraParts = [];
|
|
@@ -112,6 +125,24 @@ function writeDocx(flow) {
|
|
|
112
125
|
target: "numbering.xml",
|
|
113
126
|
targetMode: "Internal"
|
|
114
127
|
});
|
|
128
|
+
emitNotes(flow.footnotes, {
|
|
129
|
+
noteKind: "footnote",
|
|
130
|
+
partPath: FOOTNOTES_PART,
|
|
131
|
+
rootTag: "w:footnotes",
|
|
132
|
+
noteTag: "w:footnote",
|
|
133
|
+
contentType: FOOTNOTES_CONTENT_TYPE,
|
|
134
|
+
relType: REL_FOOTNOTES,
|
|
135
|
+
target: "footnotes.xml"
|
|
136
|
+
}, state, losses, docScope, extraParts, extraPartRels);
|
|
137
|
+
emitNotes(flow.endnotes, {
|
|
138
|
+
noteKind: "endnote",
|
|
139
|
+
partPath: ENDNOTES_PART,
|
|
140
|
+
rootTag: "w:endnotes",
|
|
141
|
+
noteTag: "w:endnote",
|
|
142
|
+
contentType: ENDNOTES_CONTENT_TYPE,
|
|
143
|
+
relType: REL_ENDNOTES,
|
|
144
|
+
target: "endnotes.xml"
|
|
145
|
+
}, state, losses, docScope, extraParts, extraPartRels);
|
|
115
146
|
const partRelationships = [...docScope.rels.length > 0 ? [{
|
|
116
147
|
sourcePart: "word/document.xml",
|
|
117
148
|
relationships: docScope.rels
|
|
@@ -126,6 +157,7 @@ function writeDocx(flow) {
|
|
|
126
157
|
},
|
|
127
158
|
...numberingPart ? [numberingPart] : [],
|
|
128
159
|
...extraParts,
|
|
160
|
+
...state.chartParts,
|
|
129
161
|
...state.mediaParts
|
|
130
162
|
],
|
|
131
163
|
rootRelationships: [{
|
|
@@ -139,6 +171,34 @@ function writeDocx(flow) {
|
|
|
139
171
|
losses
|
|
140
172
|
};
|
|
141
173
|
}
|
|
174
|
+
function emitNotes(notes, cfg, state, losses, docScope, extraParts, extraPartRels) {
|
|
175
|
+
if (!notes || notes.size === 0) return;
|
|
176
|
+
const scope = newScope();
|
|
177
|
+
scope.noteKind = cfg.noteKind;
|
|
178
|
+
const stub = (type, id, mark) => `<${cfg.noteTag} w:type="${type}" w:id="${id}"><w:p><w:r>${mark}</w:r></w:p></${cfg.noteTag}>`;
|
|
179
|
+
const noteXmls = [];
|
|
180
|
+
for (const [id, content] of notes) {
|
|
181
|
+
const inner = [];
|
|
182
|
+
for (const el of content) emitBlock(inner, el, losses, state, scope);
|
|
183
|
+
noteXmls.push(`<${cfg.noteTag} w:id="${escapeAttr(id)}">${inner.join("") || "<w:p/>"}</${cfg.noteTag}>`);
|
|
184
|
+
}
|
|
185
|
+
const xml = `<?xml version="1.0" encoding="UTF-8" standalone="yes"?><${cfg.rootTag} xmlns:w="http://schemas.openxmlformats.org/wordprocessingml/2006/main" xmlns:r="http://schemas.openxmlformats.org/officeDocument/2006/relationships">` + stub("separator", -1, "<w:separator/>") + stub("continuationSeparator", 0, "<w:continuationSeparator/>") + noteXmls.join("") + `</${cfg.rootTag}>`;
|
|
186
|
+
extraParts.push({
|
|
187
|
+
path: cfg.partPath,
|
|
188
|
+
data: encoder.encode(xml),
|
|
189
|
+
contentType: cfg.contentType
|
|
190
|
+
});
|
|
191
|
+
if (scope.rels.length > 0) extraPartRels.push({
|
|
192
|
+
sourcePart: cfg.partPath,
|
|
193
|
+
relationships: scope.rels
|
|
194
|
+
});
|
|
195
|
+
docScope.rels.push({
|
|
196
|
+
id: `rId${++docScope.relSeq}`,
|
|
197
|
+
type: cfg.relType,
|
|
198
|
+
target: cfg.target,
|
|
199
|
+
targetMode: "Internal"
|
|
200
|
+
});
|
|
201
|
+
}
|
|
142
202
|
function emitHeadersFooters(flow, section, state, docScope, extraParts, extraPartRels, hfCache, losses) {
|
|
143
203
|
const refs = {
|
|
144
204
|
headers: [],
|
|
@@ -226,12 +286,37 @@ function emitBlock(out, el, losses, state, scope, closingSectPr) {
|
|
|
226
286
|
out.push(`<w:p>${pPrWithSect(el.shape.paragraphProperties, closingSectPr)}<w:r>${drawing}</w:r></w:p>`);
|
|
227
287
|
return;
|
|
228
288
|
}
|
|
229
|
-
|
|
230
|
-
|
|
231
|
-
|
|
232
|
-
|
|
233
|
-
|
|
289
|
+
const chartDrawing = chartBlockXml(el.chart, state, scope, losses);
|
|
290
|
+
if (chartDrawing) out.push(`<w:p>${pPrWithSect(el.chart.paragraphProperties, closingSectPr)}<w:r>${chartDrawing}</w:r></w:p>`);
|
|
291
|
+
else if (closingSectPr) out.push(`<w:p><w:pPr>${closingSectPr}</w:pPr></w:p>`);
|
|
292
|
+
}
|
|
293
|
+
function chartBlockXml(chart, state, scope, losses) {
|
|
294
|
+
const data = state.charts?.get(chart.chartRelId);
|
|
295
|
+
if (!data) {
|
|
296
|
+
losses.push({
|
|
297
|
+
severity: "dropped",
|
|
298
|
+
feature: FEATURES.charts,
|
|
299
|
+
detail: "chart data missing"
|
|
300
|
+
});
|
|
301
|
+
return "";
|
|
302
|
+
}
|
|
303
|
+
const cid = ++state.chartSeq;
|
|
304
|
+
state.chartParts.push({
|
|
305
|
+
path: `word/charts/chart${cid}.xml`,
|
|
306
|
+
data: encoder.encode(chartSpaceXml(data)),
|
|
307
|
+
contentType: CHART_CONTENT_TYPE
|
|
308
|
+
});
|
|
309
|
+
const relId = `rId${++scope.relSeq}`;
|
|
310
|
+
scope.rels.push({
|
|
311
|
+
id: relId,
|
|
312
|
+
type: REL_CHART,
|
|
313
|
+
target: `charts/chart${cid}.xml`,
|
|
314
|
+
targetMode: "Internal"
|
|
234
315
|
});
|
|
316
|
+
const cx = Math.round(chart.width * EMU_PER_PT);
|
|
317
|
+
const cy = Math.round(chart.height * EMU_PER_PT);
|
|
318
|
+
const id = ++state.drawingSeq;
|
|
319
|
+
return `<w:drawing><wp:inline xmlns:wp="http://schemas.openxmlformats.org/drawingml/2006/wordprocessingDrawing"><wp:extent cx="${cx}" cy="${cy}"/><wp:docPr id="${id}" name="Chart ${id}"${chart.altText ? ` descr="${escapeAttr(chart.altText)}"` : ""}/><a:graphic xmlns:a="http://schemas.openxmlformats.org/drawingml/2006/main"><a:graphicData uri="http://schemas.openxmlformats.org/drawingml/2006/chart"><c:chart xmlns:c="http://schemas.openxmlformats.org/drawingml/2006/chart" xmlns:r="http://schemas.openxmlformats.org/officeDocument/2006/relationships" r:id="${relId}"/></a:graphicData></a:graphic></wp:inline></w:drawing>`;
|
|
235
320
|
}
|
|
236
321
|
function drawingXml(resource, widthPt, heightPt, altText, state, scope) {
|
|
237
322
|
if (resource === void 0) return "";
|
|
@@ -402,7 +487,7 @@ function cellMarginsXml(tag, margins) {
|
|
|
402
487
|
return inner ? `<${tag}>${inner}</${tag}>` : "";
|
|
403
488
|
}
|
|
404
489
|
function paragraphXml(p, state, scope, closingSectPr) {
|
|
405
|
-
const visible = p.runs.filter((run) => !run.listMarker && run.math
|
|
490
|
+
const visible = p.runs.filter((run) => !run.listMarker && (run.math !== void 0 || run.text !== "" || run.inlineImage !== void 0 || run.pageBreak || run.footnoteRef !== void 0 || run.endnoteRef !== void 0 || run.noteNumber === true || run.href !== void 0 || run.anchor !== void 0));
|
|
406
491
|
const inner = [];
|
|
407
492
|
let i = 0;
|
|
408
493
|
while (i < visible.length) {
|
|
@@ -440,6 +525,7 @@ function hyperlinkXml(run, inner, _state, scope) {
|
|
|
440
525
|
return `<w:hyperlink w:anchor="${escapeAttr(run.anchor)}">${inner}</w:hyperlink>`;
|
|
441
526
|
}
|
|
442
527
|
function runXml(run, state, scope) {
|
|
528
|
+
if (run.math !== void 0) return `<m:oMath xmlns:m="http://schemas.openxmlformats.org/officeDocument/2006/math">${omathXml(run.math)}</m:oMath>`;
|
|
443
529
|
const rPr = rPrXml(run.properties);
|
|
444
530
|
if (run.inlineImage !== void 0) {
|
|
445
531
|
const img = run.inlineImage;
|
|
@@ -447,6 +533,9 @@ function runXml(run, state, scope) {
|
|
|
447
533
|
if (drawing) return `<w:r>${rPr}${drawing}</w:r>`;
|
|
448
534
|
if (run.text === "" && !run.pageBreak) return "";
|
|
449
535
|
}
|
|
536
|
+
if (run.footnoteRef !== void 0) return `<w:r>${rPr}<w:footnoteReference w:id="${escapeAttr(run.footnoteRef)}"/></w:r>`;
|
|
537
|
+
if (run.endnoteRef !== void 0) return `<w:r>${rPr}<w:endnoteReference w:id="${escapeAttr(run.endnoteRef)}"/></w:r>`;
|
|
538
|
+
if (run.noteNumber) return `<w:r>${rPr}<${scope.noteKind === "endnote" ? "w:endnoteRef" : "w:footnoteRef"}/></w:r>`;
|
|
450
539
|
const brk = run.pageBreak ? "<w:br w:type=\"page\"/>" : "";
|
|
451
540
|
if (run.text === "") return brk ? `<w:r>${rPr}${brk}</w:r>` : "";
|
|
452
541
|
return `<w:r>${rPr}<w:t xml:space="preserve">${escapeXml(run.text)}</w:t>${brk}</w:r>`;
|