reamkit 1.15.0 → 1.15.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/esm/core/ole/cfb.js +34 -4
- package/dist/esm/pdf-reader/flow-build.d.ts +4 -2
- package/dist/esm/pdf-reader/flow-build.js +25 -2
- package/dist/esm/pdf-reader/layout.js +2 -2
- package/dist/esm/pdf-reader/tagged.js +2 -2
- package/dist/esm/pptx/pptx-reader.js +16 -1
- package/dist/esm/word/doc/doc-reader.js +10 -4
- package/dist/esm/word/doc/doc-text.d.ts +4 -0
- package/dist/esm/word/doc/doc-text.js +25 -0
- package/package.json +1 -1
package/dist/esm/core/ole/cfb.js
CHANGED
|
@@ -30,6 +30,7 @@ function openCfb(bytes) {
|
|
|
30
30
|
const sectorSize = 1 << sectorShift;
|
|
31
31
|
const miniSectorSize = 1 << miniSectorShift;
|
|
32
32
|
const miniCutoff = view.getUint32(56, true);
|
|
33
|
+
const majorVersion = view.getUint16(26, true);
|
|
33
34
|
const totalSectors = Math.floor((bytes.length - HEADER_SIZE) / sectorSize);
|
|
34
35
|
const sectorOffset = (sector) => {
|
|
35
36
|
if (sector < 0 || sector >= totalSectors) throw new CfbError(`sector ${sector} out of range`);
|
|
@@ -86,19 +87,29 @@ function openCfb(bytes) {
|
|
|
86
87
|
const dirView = new DataView(dirBytes.buffer, dirBytes.byteOffset, dirBytes.byteLength);
|
|
87
88
|
const entries = [];
|
|
88
89
|
const entryCount = Math.floor(dirBytes.length / DIR_ENTRY_SIZE);
|
|
90
|
+
const rawByIndex = new Array(entryCount);
|
|
89
91
|
for (let i = 0; i < entryCount; i++) {
|
|
90
92
|
const base = i * DIR_ENTRY_SIZE;
|
|
91
93
|
const objType = dirView.getUint8(base + 66);
|
|
92
94
|
if (objType !== 1 && objType !== 2 && objType !== 5) continue;
|
|
93
95
|
const name = decodeName(dirBytes, base, dirView.getUint16(base + 64, true));
|
|
94
96
|
const sizeLow = dirView.getUint32(base + 120, true);
|
|
95
|
-
const
|
|
96
|
-
|
|
97
|
+
const sizeHigh = dirView.getUint32(base + 124, true);
|
|
98
|
+
const size = majorVersion === 3 ? sizeLow : sizeHigh * 4294967296 + sizeLow;
|
|
99
|
+
const entry = {
|
|
97
100
|
name,
|
|
98
101
|
type: objType === 5 ? "root" : objType === 1 ? "storage" : "stream",
|
|
99
102
|
startSector: dirView.getUint32(base + 116, true),
|
|
100
103
|
size
|
|
101
|
-
}
|
|
104
|
+
};
|
|
105
|
+
entries.push(entry);
|
|
106
|
+
rawByIndex[i] = {
|
|
107
|
+
entry,
|
|
108
|
+
left: dirView.getUint32(base + 68, true),
|
|
109
|
+
right: dirView.getUint32(base + 72, true),
|
|
110
|
+
child: dirView.getUint32(base + 76, true),
|
|
111
|
+
isRoot: objType === 5
|
|
112
|
+
};
|
|
102
113
|
}
|
|
103
114
|
const root = entries.find((e) => e.type === "root");
|
|
104
115
|
if (!root) throw new CfbError("compound file has no root entry");
|
|
@@ -137,8 +148,27 @@ function openCfb(bytes) {
|
|
|
137
148
|
if (entry.size >= miniCutoff) return readChainBytes(entry.startSector, MAX_TOTAL_BYTES).subarray(0, entry.size);
|
|
138
149
|
return readMiniStream(entry);
|
|
139
150
|
};
|
|
151
|
+
const topLevel = /* @__PURE__ */ new Set();
|
|
152
|
+
const rootRaw = rawByIndex.find((r) => r?.isRoot);
|
|
153
|
+
if (rootRaw) {
|
|
154
|
+
const stack = [rootRaw.child];
|
|
155
|
+
const seen = /* @__PURE__ */ new Set();
|
|
156
|
+
while (stack.length > 0) {
|
|
157
|
+
const idx = stack.pop();
|
|
158
|
+
if (idx >= entryCount || seen.has(idx)) continue;
|
|
159
|
+
seen.add(idx);
|
|
160
|
+
const node = rawByIndex[idx];
|
|
161
|
+
if (!node) continue;
|
|
162
|
+
topLevel.add(node.entry);
|
|
163
|
+
stack.push(node.left, node.right);
|
|
164
|
+
}
|
|
165
|
+
}
|
|
140
166
|
const byName = /* @__PURE__ */ new Map();
|
|
141
|
-
|
|
167
|
+
const addStreams = (list) => {
|
|
168
|
+
for (const e of list) if (e.type === "stream" && !byName.has(e.name.toLowerCase())) byName.set(e.name.toLowerCase(), e);
|
|
169
|
+
};
|
|
170
|
+
addStreams(topLevel);
|
|
171
|
+
addStreams(entries);
|
|
142
172
|
return {
|
|
143
173
|
entries,
|
|
144
174
|
hasStream: (name) => byName.has(name.toLowerCase()),
|
|
@@ -1,7 +1,8 @@
|
|
|
1
|
-
import { BodyElement } from '../core/document-model/index.js';
|
|
1
|
+
import { BodyElement, SectionProperties } from '../core/document-model/index.js';
|
|
2
2
|
import { FlowDoc } from '../core/ir/flow.js';
|
|
3
3
|
import { Loss, ResourceStore } from '../core/ir/index.js';
|
|
4
4
|
import { PdfImage } from './images.js';
|
|
5
|
+
import { PdfPage } from './document.js';
|
|
5
6
|
import { PdfVector } from './vector.js';
|
|
6
7
|
export interface Reconstruction {
|
|
7
8
|
readonly doc: FlowDoc;
|
|
@@ -16,4 +17,5 @@ export declare function paragraphFromRuns(spans: ReadonlyArray<TextSpan>, outlin
|
|
|
16
17
|
export declare function imageBlock(image: PdfImage, resources: ResourceStore, alt?: string): BodyElement;
|
|
17
18
|
export declare function dedupeLosses(losses: ReadonlyArray<Loss>): Array<Loss>;
|
|
18
19
|
export declare function shapeBlock(v: PdfVector): BodyElement;
|
|
19
|
-
export declare function
|
|
20
|
+
export declare function sectionFromPdfPages(pages: ReadonlyArray<PdfPage>): SectionProperties | undefined;
|
|
21
|
+
export declare function buildFlowDoc(body: ReadonlyArray<BodyElement>, resources?: ResourceStore, section?: SectionProperties): FlowDoc;
|
|
@@ -124,14 +124,37 @@ function shapeBlock(v) {
|
|
|
124
124
|
}
|
|
125
125
|
};
|
|
126
126
|
}
|
|
127
|
-
function
|
|
127
|
+
function sectionFromPdfPages(pages) {
|
|
128
|
+
const box = pages[0]?.mediaBox;
|
|
129
|
+
if (!box) return void 0;
|
|
130
|
+
const width = Math.abs(box[2] - box[0]);
|
|
131
|
+
const height = Math.abs(box[3] - box[1]);
|
|
132
|
+
if (!(width > 0 && height > 0)) return void 0;
|
|
133
|
+
return {
|
|
134
|
+
pageSize: {
|
|
135
|
+
width: pt(width),
|
|
136
|
+
height: pt(height),
|
|
137
|
+
orientation: width > height ? "landscape" : "portrait"
|
|
138
|
+
},
|
|
139
|
+
margins: {
|
|
140
|
+
top: pt(0),
|
|
141
|
+
right: pt(0),
|
|
142
|
+
bottom: pt(0),
|
|
143
|
+
left: pt(0)
|
|
144
|
+
},
|
|
145
|
+
headers: [],
|
|
146
|
+
footers: []
|
|
147
|
+
};
|
|
148
|
+
}
|
|
149
|
+
function buildFlowDoc(body, resources = new ResourceStore(), section) {
|
|
128
150
|
return {
|
|
129
151
|
kind: "flow",
|
|
130
152
|
body: resolveBodyStyles([...body], EMPTY_STYLE_SHEET),
|
|
131
153
|
sections: [],
|
|
154
|
+
...section ? { section } : {},
|
|
132
155
|
styles: EMPTY_STYLE_SHEET,
|
|
133
156
|
resources
|
|
134
157
|
};
|
|
135
158
|
}
|
|
136
159
|
//#endregion
|
|
137
|
-
export { buildFlowDoc, dedupeLosses, imageBlock, paragraphBlock, paragraphFromRuns, shapeBlock };
|
|
160
|
+
export { buildFlowDoc, dedupeLosses, imageBlock, paragraphBlock, paragraphFromRuns, sectionFromPdfPages, shapeBlock };
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import { ResourceStore } from "../core/ir/resources.js";
|
|
2
|
-
import { buildFlowDoc, dedupeLosses, imageBlock, paragraphFromRuns, shapeBlock } from "./flow-build.js";
|
|
2
|
+
import { buildFlowDoc, dedupeLosses, imageBlock, paragraphFromRuns, sectionFromPdfPages, shapeBlock } from "./flow-build.js";
|
|
3
3
|
import { collectPageImages } from "./images.js";
|
|
4
4
|
import { extractPageText } from "./text.js";
|
|
5
5
|
import { collectPageVectors } from "./vector.js";
|
|
@@ -45,7 +45,7 @@ function reconstructByLayout(file) {
|
|
|
45
45
|
for (const block of blocks) body.push(block.el);
|
|
46
46
|
});
|
|
47
47
|
return {
|
|
48
|
-
doc: buildFlowDoc(body, resources),
|
|
48
|
+
doc: buildFlowDoc(body, resources, sectionFromPdfPages(pages)),
|
|
49
49
|
losses: dedupeLosses(losses)
|
|
50
50
|
};
|
|
51
51
|
}
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
import { pt } from "../core/ir/units.js";
|
|
2
2
|
import { ResourceStore } from "../core/ir/resources.js";
|
|
3
|
-
import { buildFlowDoc, dedupeLosses, imageBlock, paragraphBlock, paragraphFromRuns } from "./flow-build.js";
|
|
3
|
+
import { buildFlowDoc, dedupeLosses, imageBlock, paragraphBlock, paragraphFromRuns, sectionFromPdfPages } from "./flow-build.js";
|
|
4
4
|
import { collectPageImages } from "./images.js";
|
|
5
5
|
import { extractPageText } from "./text.js";
|
|
6
6
|
import { readStructTree } from "./struct-tree.js";
|
|
@@ -126,7 +126,7 @@ function reconstructTaggedPdf(file) {
|
|
|
126
126
|
for (const { img } of orphans) body.push(imageBlock(img, resources));
|
|
127
127
|
if (body.length === 0) return void 0;
|
|
128
128
|
return {
|
|
129
|
-
doc: buildFlowDoc(body, resources),
|
|
129
|
+
doc: buildFlowDoc(body, resources, sectionFromPdfPages(pages)),
|
|
130
130
|
losses: imageLosses
|
|
131
131
|
};
|
|
132
132
|
}
|
|
@@ -26,6 +26,10 @@ var parser = new XMLParser({
|
|
|
26
26
|
parseTagValue: false,
|
|
27
27
|
trimValues: false
|
|
28
28
|
});
|
|
29
|
+
function isSlideHidden(data) {
|
|
30
|
+
const root = decoder.decode(data.subarray(0, 4096)).match(/<p:sld\b[^>]*>/);
|
|
31
|
+
return root ? /\bshow\s*=\s*["']0["']/.test(root[0]) : false;
|
|
32
|
+
}
|
|
29
33
|
function readPptx(bytes) {
|
|
30
34
|
const losses = [];
|
|
31
35
|
const pkg = OpcPackage.open(bytes);
|
|
@@ -45,12 +49,23 @@ function readPptx(bytes) {
|
|
|
45
49
|
const slideRelById = new Map(pkg.getPartRelationships(presPath).filter((r) => r.type.endsWith("/slide")).map((r) => [r.id, r]));
|
|
46
50
|
const lst = kids.find((c) => poIs(c, "p:sldIdLst"));
|
|
47
51
|
const ids = lst ? poChildren(lst).filter((c) => poIs(c, "p:sldId")) : [];
|
|
52
|
+
let hidden = 0;
|
|
48
53
|
for (const sldId of ids) {
|
|
49
54
|
const rid = poAttr(sldId, "r:id");
|
|
50
55
|
const rel = rid !== void 0 ? slideRelById.get(rid) : void 0;
|
|
51
56
|
const part = rel ? pkg.resolveRelatedPart(presPath, rel) : void 0;
|
|
52
|
-
if (part)
|
|
57
|
+
if (!part) continue;
|
|
58
|
+
if (isSlideHidden(part.data)) {
|
|
59
|
+
hidden++;
|
|
60
|
+
continue;
|
|
61
|
+
}
|
|
62
|
+
slideParts.push(part);
|
|
53
63
|
}
|
|
64
|
+
if (hidden > 0) losses.push({
|
|
65
|
+
severity: "dropped",
|
|
66
|
+
feature: FEATURES.text,
|
|
67
|
+
detail: `${hidden} hidden slide(s) omitted (p:sld@show="0")`
|
|
68
|
+
});
|
|
54
69
|
}
|
|
55
70
|
if (slideParts.length === 0) losses.push({
|
|
56
71
|
severity: "dropped",
|
|
@@ -140,16 +140,22 @@ function readDoc(bytes) {
|
|
|
140
140
|
addStory(hf.firstFooter, "footer", "first");
|
|
141
141
|
addStory(hf.evenFooter, "footer", "even");
|
|
142
142
|
}
|
|
143
|
+
const ps = content.pageSize;
|
|
144
|
+
const pageSize = ps ? {
|
|
145
|
+
width: pt(ps.widthPt),
|
|
146
|
+
height: pt(ps.heightPt),
|
|
147
|
+
orientation: ps.widthPt > ps.heightPt ? "landscape" : "portrait"
|
|
148
|
+
} : {
|
|
149
|
+
width: pt(612),
|
|
150
|
+
height: pt(792)
|
|
151
|
+
};
|
|
143
152
|
return {
|
|
144
153
|
doc: {
|
|
145
154
|
kind: "flow",
|
|
146
155
|
body: resolveBodyStyles(body, EMPTY_STYLE_SHEET),
|
|
147
156
|
sections: [],
|
|
148
157
|
section: {
|
|
149
|
-
pageSize
|
|
150
|
-
width: pt(612),
|
|
151
|
-
height: pt(792)
|
|
152
|
-
},
|
|
158
|
+
pageSize,
|
|
153
159
|
margins: {
|
|
154
160
|
top: pt(72),
|
|
155
161
|
right: pt(72),
|
|
@@ -74,6 +74,10 @@ export interface DocContent {
|
|
|
74
74
|
readonly paragraphs: ReadonlyArray<DocParagraph>;
|
|
75
75
|
readonly headerFooters?: DocHeaderFooters;
|
|
76
76
|
readonly listTables?: DocListTables;
|
|
77
|
+
readonly pageSize?: {
|
|
78
|
+
readonly widthPt: number;
|
|
79
|
+
readonly heightPt: number;
|
|
80
|
+
};
|
|
77
81
|
readonly encrypted: boolean;
|
|
78
82
|
}
|
|
79
83
|
export declare function extractDocContent(bytes: Uint8Array): DocContent;
|
|
@@ -13,6 +13,8 @@ var OFF_FCPLCFBTEPAPX = 258;
|
|
|
13
13
|
var OFF_LCBPLCFBTEPAPX = 262;
|
|
14
14
|
var OFF_FCCLX = 418;
|
|
15
15
|
var OFF_LCBCLX = 422;
|
|
16
|
+
var OFF_FCPLCFSED = 202;
|
|
17
|
+
var OFF_LCBPLCFSED = 206;
|
|
16
18
|
var OFF_FCPLFLST = 738;
|
|
17
19
|
var OFF_LCBPLFLST = 742;
|
|
18
20
|
var OFF_FCPLFLFO = 746;
|
|
@@ -43,6 +45,9 @@ var SPRM_T_DEF_TABLE_SHD = 54802;
|
|
|
43
45
|
var MAX_ROW_CELLS = 256;
|
|
44
46
|
var SPRM_P_ILVL = 9738;
|
|
45
47
|
var SPRM_P_ILFO = 17931;
|
|
48
|
+
var SPRM_S_XA_PAGE = 45087;
|
|
49
|
+
var SPRM_S_YA_PAGE = 45088;
|
|
50
|
+
var MAX_PAGE_TWIPS = 31680;
|
|
46
51
|
var SPRA_LEN = [
|
|
47
52
|
1,
|
|
48
53
|
1,
|
|
@@ -63,6 +68,24 @@ var EMPTY = {
|
|
|
63
68
|
paragraphs: [],
|
|
64
69
|
encrypted: false
|
|
65
70
|
};
|
|
71
|
+
function readSectionPageSize(wd, table) {
|
|
72
|
+
const fc = u32(wd, OFF_FCPLCFSED);
|
|
73
|
+
const lcb = u32(wd, OFF_LCBPLCFSED);
|
|
74
|
+
if (lcb < 20 || fc + lcb > table.length) return void 0;
|
|
75
|
+
const fcSepx = u32(table.subarray(fc, fc + lcb), (Math.floor((lcb - 4) / 16) + 1) * 4 + 2);
|
|
76
|
+
if (fcSepx === 4294967295 || fcSepx + 2 > wd.length) return void 0;
|
|
77
|
+
const cb = u16(wd, fcSepx);
|
|
78
|
+
const grpprl = wd.subarray(fcSepx + 2, Math.min(wd.length, fcSepx + 2 + cb));
|
|
79
|
+
let xa = 0;
|
|
80
|
+
let ya = 0;
|
|
81
|
+
for (const { sprm, op } of sprms(grpprl)) if (sprm === SPRM_S_XA_PAGE) xa = u16(grpprl, op);
|
|
82
|
+
else if (sprm === SPRM_S_YA_PAGE) ya = u16(grpprl, op);
|
|
83
|
+
if (xa < 1 || ya < 1 || xa > MAX_PAGE_TWIPS || ya > MAX_PAGE_TWIPS) return void 0;
|
|
84
|
+
return {
|
|
85
|
+
widthPt: xa / 20,
|
|
86
|
+
heightPt: ya / 20
|
|
87
|
+
};
|
|
88
|
+
}
|
|
66
89
|
function extractDocContent(bytes) {
|
|
67
90
|
const cfb = openCfb(bytes);
|
|
68
91
|
const wd = cfb.readStream("WordDocument");
|
|
@@ -89,10 +112,12 @@ function extractDocContent(bytes) {
|
|
|
89
112
|
const main = buildParagraphs(wd, pieces, chpx, papx, 0, ccpText, data);
|
|
90
113
|
const headerFooters = extractHeaderFooters(wd, table, pieces, chpx, papx, data, ccpText);
|
|
91
114
|
const listTables = parseListTables(wd, table);
|
|
115
|
+
const pageSize = readSectionPageSize(wd, table);
|
|
92
116
|
return {
|
|
93
117
|
paragraphs: main,
|
|
94
118
|
...headerFooters ? { headerFooters } : {},
|
|
95
119
|
...listTables ? { listTables } : {},
|
|
120
|
+
...pageSize ? { pageSize } : {},
|
|
96
121
|
encrypted: false
|
|
97
122
|
};
|
|
98
123
|
}
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "reamkit",
|
|
3
|
-
"version": "1.15.
|
|
3
|
+
"version": "1.15.1",
|
|
4
4
|
"description": "Ream — convert DOCX, XLSX, PPTX and PDF to PDF, SVG, HTML, DOCX and XLSX, built from scratch on the ECMA-376 and ISO 32000 specifications. Parse once, convert anywhere.",
|
|
5
5
|
"license": "MIT",
|
|
6
6
|
"author": "Alex Krassavin <info@reamkit.dev>",
|