reamkit 1.15.0 → 1.15.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -30,6 +30,7 @@ function openCfb(bytes) {
30
30
  const sectorSize = 1 << sectorShift;
31
31
  const miniSectorSize = 1 << miniSectorShift;
32
32
  const miniCutoff = view.getUint32(56, true);
33
+ const majorVersion = view.getUint16(26, true);
33
34
  const totalSectors = Math.floor((bytes.length - HEADER_SIZE) / sectorSize);
34
35
  const sectorOffset = (sector) => {
35
36
  if (sector < 0 || sector >= totalSectors) throw new CfbError(`sector ${sector} out of range`);
@@ -86,19 +87,29 @@ function openCfb(bytes) {
86
87
  const dirView = new DataView(dirBytes.buffer, dirBytes.byteOffset, dirBytes.byteLength);
87
88
  const entries = [];
88
89
  const entryCount = Math.floor(dirBytes.length / DIR_ENTRY_SIZE);
90
+ const rawByIndex = new Array(entryCount);
89
91
  for (let i = 0; i < entryCount; i++) {
90
92
  const base = i * DIR_ENTRY_SIZE;
91
93
  const objType = dirView.getUint8(base + 66);
92
94
  if (objType !== 1 && objType !== 2 && objType !== 5) continue;
93
95
  const name = decodeName(dirBytes, base, dirView.getUint16(base + 64, true));
94
96
  const sizeLow = dirView.getUint32(base + 120, true);
95
- const size = dirView.getUint32(base + 124, true) * 4294967296 + sizeLow;
96
- entries.push({
97
+ const sizeHigh = dirView.getUint32(base + 124, true);
98
+ const size = majorVersion === 3 ? sizeLow : sizeHigh * 4294967296 + sizeLow;
99
+ const entry = {
97
100
  name,
98
101
  type: objType === 5 ? "root" : objType === 1 ? "storage" : "stream",
99
102
  startSector: dirView.getUint32(base + 116, true),
100
103
  size
101
- });
104
+ };
105
+ entries.push(entry);
106
+ rawByIndex[i] = {
107
+ entry,
108
+ left: dirView.getUint32(base + 68, true),
109
+ right: dirView.getUint32(base + 72, true),
110
+ child: dirView.getUint32(base + 76, true),
111
+ isRoot: objType === 5
112
+ };
102
113
  }
103
114
  const root = entries.find((e) => e.type === "root");
104
115
  if (!root) throw new CfbError("compound file has no root entry");
@@ -137,8 +148,27 @@ function openCfb(bytes) {
137
148
  if (entry.size >= miniCutoff) return readChainBytes(entry.startSector, MAX_TOTAL_BYTES).subarray(0, entry.size);
138
149
  return readMiniStream(entry);
139
150
  };
151
+ const topLevel = /* @__PURE__ */ new Set();
152
+ const rootRaw = rawByIndex.find((r) => r?.isRoot);
153
+ if (rootRaw) {
154
+ const stack = [rootRaw.child];
155
+ const seen = /* @__PURE__ */ new Set();
156
+ while (stack.length > 0) {
157
+ const idx = stack.pop();
158
+ if (idx >= entryCount || seen.has(idx)) continue;
159
+ seen.add(idx);
160
+ const node = rawByIndex[idx];
161
+ if (!node) continue;
162
+ topLevel.add(node.entry);
163
+ stack.push(node.left, node.right);
164
+ }
165
+ }
140
166
  const byName = /* @__PURE__ */ new Map();
141
- for (const e of entries) if (e.type === "stream" && !byName.has(e.name.toLowerCase())) byName.set(e.name.toLowerCase(), e);
167
+ const addStreams = (list) => {
168
+ for (const e of list) if (e.type === "stream" && !byName.has(e.name.toLowerCase())) byName.set(e.name.toLowerCase(), e);
169
+ };
170
+ addStreams(topLevel);
171
+ addStreams(entries);
142
172
  return {
143
173
  entries,
144
174
  hasStream: (name) => byName.has(name.toLowerCase()),
@@ -1,7 +1,8 @@
1
- import { BodyElement } from '../core/document-model/index.js';
1
+ import { BodyElement, SectionProperties } from '../core/document-model/index.js';
2
2
  import { FlowDoc } from '../core/ir/flow.js';
3
3
  import { Loss, ResourceStore } from '../core/ir/index.js';
4
4
  import { PdfImage } from './images.js';
5
+ import { PdfPage } from './document.js';
5
6
  import { PdfVector } from './vector.js';
6
7
  export interface Reconstruction {
7
8
  readonly doc: FlowDoc;
@@ -16,4 +17,5 @@ export declare function paragraphFromRuns(spans: ReadonlyArray<TextSpan>, outlin
16
17
  export declare function imageBlock(image: PdfImage, resources: ResourceStore, alt?: string): BodyElement;
17
18
  export declare function dedupeLosses(losses: ReadonlyArray<Loss>): Array<Loss>;
18
19
  export declare function shapeBlock(v: PdfVector): BodyElement;
19
- export declare function buildFlowDoc(body: ReadonlyArray<BodyElement>, resources?: ResourceStore): FlowDoc;
20
+ export declare function sectionFromPdfPages(pages: ReadonlyArray<PdfPage>): SectionProperties | undefined;
21
+ export declare function buildFlowDoc(body: ReadonlyArray<BodyElement>, resources?: ResourceStore, section?: SectionProperties): FlowDoc;
@@ -124,14 +124,37 @@ function shapeBlock(v) {
124
124
  }
125
125
  };
126
126
  }
127
- function buildFlowDoc(body, resources = new ResourceStore()) {
127
+ function sectionFromPdfPages(pages) {
128
+ const box = pages[0]?.mediaBox;
129
+ if (!box) return void 0;
130
+ const width = Math.abs(box[2] - box[0]);
131
+ const height = Math.abs(box[3] - box[1]);
132
+ if (!(width > 0 && height > 0)) return void 0;
133
+ return {
134
+ pageSize: {
135
+ width: pt(width),
136
+ height: pt(height),
137
+ orientation: width > height ? "landscape" : "portrait"
138
+ },
139
+ margins: {
140
+ top: pt(0),
141
+ right: pt(0),
142
+ bottom: pt(0),
143
+ left: pt(0)
144
+ },
145
+ headers: [],
146
+ footers: []
147
+ };
148
+ }
149
+ function buildFlowDoc(body, resources = new ResourceStore(), section) {
128
150
  return {
129
151
  kind: "flow",
130
152
  body: resolveBodyStyles([...body], EMPTY_STYLE_SHEET),
131
153
  sections: [],
154
+ ...section ? { section } : {},
132
155
  styles: EMPTY_STYLE_SHEET,
133
156
  resources
134
157
  };
135
158
  }
136
159
  //#endregion
137
- export { buildFlowDoc, dedupeLosses, imageBlock, paragraphBlock, paragraphFromRuns, shapeBlock };
160
+ export { buildFlowDoc, dedupeLosses, imageBlock, paragraphBlock, paragraphFromRuns, sectionFromPdfPages, shapeBlock };
@@ -1,5 +1,5 @@
1
1
  import { ResourceStore } from "../core/ir/resources.js";
2
- import { buildFlowDoc, dedupeLosses, imageBlock, paragraphFromRuns, shapeBlock } from "./flow-build.js";
2
+ import { buildFlowDoc, dedupeLosses, imageBlock, paragraphFromRuns, sectionFromPdfPages, shapeBlock } from "./flow-build.js";
3
3
  import { collectPageImages } from "./images.js";
4
4
  import { extractPageText } from "./text.js";
5
5
  import { collectPageVectors } from "./vector.js";
@@ -45,7 +45,7 @@ function reconstructByLayout(file) {
45
45
  for (const block of blocks) body.push(block.el);
46
46
  });
47
47
  return {
48
- doc: buildFlowDoc(body, resources),
48
+ doc: buildFlowDoc(body, resources, sectionFromPdfPages(pages)),
49
49
  losses: dedupeLosses(losses)
50
50
  };
51
51
  }
@@ -1,6 +1,6 @@
1
1
  import { pt } from "../core/ir/units.js";
2
2
  import { ResourceStore } from "../core/ir/resources.js";
3
- import { buildFlowDoc, dedupeLosses, imageBlock, paragraphBlock, paragraphFromRuns } from "./flow-build.js";
3
+ import { buildFlowDoc, dedupeLosses, imageBlock, paragraphBlock, paragraphFromRuns, sectionFromPdfPages } from "./flow-build.js";
4
4
  import { collectPageImages } from "./images.js";
5
5
  import { extractPageText } from "./text.js";
6
6
  import { readStructTree } from "./struct-tree.js";
@@ -126,7 +126,7 @@ function reconstructTaggedPdf(file) {
126
126
  for (const { img } of orphans) body.push(imageBlock(img, resources));
127
127
  if (body.length === 0) return void 0;
128
128
  return {
129
- doc: buildFlowDoc(body, resources),
129
+ doc: buildFlowDoc(body, resources, sectionFromPdfPages(pages)),
130
130
  losses: imageLosses
131
131
  };
132
132
  }
@@ -26,6 +26,10 @@ var parser = new XMLParser({
26
26
  parseTagValue: false,
27
27
  trimValues: false
28
28
  });
29
+ function isSlideHidden(data) {
30
+ const root = decoder.decode(data.subarray(0, 4096)).match(/<p:sld\b[^>]*>/);
31
+ return root ? /\bshow\s*=\s*["']0["']/.test(root[0]) : false;
32
+ }
29
33
  function readPptx(bytes) {
30
34
  const losses = [];
31
35
  const pkg = OpcPackage.open(bytes);
@@ -45,12 +49,23 @@ function readPptx(bytes) {
45
49
  const slideRelById = new Map(pkg.getPartRelationships(presPath).filter((r) => r.type.endsWith("/slide")).map((r) => [r.id, r]));
46
50
  const lst = kids.find((c) => poIs(c, "p:sldIdLst"));
47
51
  const ids = lst ? poChildren(lst).filter((c) => poIs(c, "p:sldId")) : [];
52
+ let hidden = 0;
48
53
  for (const sldId of ids) {
49
54
  const rid = poAttr(sldId, "r:id");
50
55
  const rel = rid !== void 0 ? slideRelById.get(rid) : void 0;
51
56
  const part = rel ? pkg.resolveRelatedPart(presPath, rel) : void 0;
52
- if (part) slideParts.push(part);
57
+ if (!part) continue;
58
+ if (isSlideHidden(part.data)) {
59
+ hidden++;
60
+ continue;
61
+ }
62
+ slideParts.push(part);
53
63
  }
64
+ if (hidden > 0) losses.push({
65
+ severity: "dropped",
66
+ feature: FEATURES.text,
67
+ detail: `${hidden} hidden slide(s) omitted (p:sld@show="0")`
68
+ });
54
69
  }
55
70
  if (slideParts.length === 0) losses.push({
56
71
  severity: "dropped",
@@ -140,16 +140,22 @@ function readDoc(bytes) {
140
140
  addStory(hf.firstFooter, "footer", "first");
141
141
  addStory(hf.evenFooter, "footer", "even");
142
142
  }
143
+ const ps = content.pageSize;
144
+ const pageSize = ps ? {
145
+ width: pt(ps.widthPt),
146
+ height: pt(ps.heightPt),
147
+ orientation: ps.widthPt > ps.heightPt ? "landscape" : "portrait"
148
+ } : {
149
+ width: pt(612),
150
+ height: pt(792)
151
+ };
143
152
  return {
144
153
  doc: {
145
154
  kind: "flow",
146
155
  body: resolveBodyStyles(body, EMPTY_STYLE_SHEET),
147
156
  sections: [],
148
157
  section: {
149
- pageSize: {
150
- width: pt(612),
151
- height: pt(792)
152
- },
158
+ pageSize,
153
159
  margins: {
154
160
  top: pt(72),
155
161
  right: pt(72),
@@ -74,6 +74,10 @@ export interface DocContent {
74
74
  readonly paragraphs: ReadonlyArray<DocParagraph>;
75
75
  readonly headerFooters?: DocHeaderFooters;
76
76
  readonly listTables?: DocListTables;
77
+ readonly pageSize?: {
78
+ readonly widthPt: number;
79
+ readonly heightPt: number;
80
+ };
77
81
  readonly encrypted: boolean;
78
82
  }
79
83
  export declare function extractDocContent(bytes: Uint8Array): DocContent;
@@ -13,6 +13,8 @@ var OFF_FCPLCFBTEPAPX = 258;
13
13
  var OFF_LCBPLCFBTEPAPX = 262;
14
14
  var OFF_FCCLX = 418;
15
15
  var OFF_LCBCLX = 422;
16
+ var OFF_FCPLCFSED = 202;
17
+ var OFF_LCBPLCFSED = 206;
16
18
  var OFF_FCPLFLST = 738;
17
19
  var OFF_LCBPLFLST = 742;
18
20
  var OFF_FCPLFLFO = 746;
@@ -43,6 +45,9 @@ var SPRM_T_DEF_TABLE_SHD = 54802;
43
45
  var MAX_ROW_CELLS = 256;
44
46
  var SPRM_P_ILVL = 9738;
45
47
  var SPRM_P_ILFO = 17931;
48
+ var SPRM_S_XA_PAGE = 45087;
49
+ var SPRM_S_YA_PAGE = 45088;
50
+ var MAX_PAGE_TWIPS = 31680;
46
51
  var SPRA_LEN = [
47
52
  1,
48
53
  1,
@@ -63,6 +68,24 @@ var EMPTY = {
63
68
  paragraphs: [],
64
69
  encrypted: false
65
70
  };
71
+ function readSectionPageSize(wd, table) {
72
+ const fc = u32(wd, OFF_FCPLCFSED);
73
+ const lcb = u32(wd, OFF_LCBPLCFSED);
74
+ if (lcb < 20 || fc + lcb > table.length) return void 0;
75
+ const fcSepx = u32(table.subarray(fc, fc + lcb), (Math.floor((lcb - 4) / 16) + 1) * 4 + 2);
76
+ if (fcSepx === 4294967295 || fcSepx + 2 > wd.length) return void 0;
77
+ const cb = u16(wd, fcSepx);
78
+ const grpprl = wd.subarray(fcSepx + 2, Math.min(wd.length, fcSepx + 2 + cb));
79
+ let xa = 0;
80
+ let ya = 0;
81
+ for (const { sprm, op } of sprms(grpprl)) if (sprm === SPRM_S_XA_PAGE) xa = u16(grpprl, op);
82
+ else if (sprm === SPRM_S_YA_PAGE) ya = u16(grpprl, op);
83
+ if (xa < 1 || ya < 1 || xa > MAX_PAGE_TWIPS || ya > MAX_PAGE_TWIPS) return void 0;
84
+ return {
85
+ widthPt: xa / 20,
86
+ heightPt: ya / 20
87
+ };
88
+ }
66
89
  function extractDocContent(bytes) {
67
90
  const cfb = openCfb(bytes);
68
91
  const wd = cfb.readStream("WordDocument");
@@ -89,10 +112,12 @@ function extractDocContent(bytes) {
89
112
  const main = buildParagraphs(wd, pieces, chpx, papx, 0, ccpText, data);
90
113
  const headerFooters = extractHeaderFooters(wd, table, pieces, chpx, papx, data, ccpText);
91
114
  const listTables = parseListTables(wd, table);
115
+ const pageSize = readSectionPageSize(wd, table);
92
116
  return {
93
117
  paragraphs: main,
94
118
  ...headerFooters ? { headerFooters } : {},
95
119
  ...listTables ? { listTables } : {},
120
+ ...pageSize ? { pageSize } : {},
96
121
  encrypted: false
97
122
  };
98
123
  }
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "reamkit",
3
- "version": "1.15.0",
3
+ "version": "1.15.1",
4
4
  "description": "Ream — convert DOCX, XLSX, PPTX and PDF to PDF, SVG, HTML, DOCX and XLSX, built from scratch on the ECMA-376 and ISO 32000 specifications. Parse once, convert anywhere.",
5
5
  "license": "MIT",
6
6
  "author": "Alex Krassavin <info@reamkit.dev>",