reamkit 1.5.0 → 1.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (39) hide show
  1. package/README.md +2 -2
  2. package/dist/esm/core/converter/facade.js +6 -1
  3. package/dist/esm/core/converter/ream.js +2 -1
  4. package/dist/esm/core/document-model/types.d.ts +9 -1
  5. package/dist/esm/core/spreadsheet-model/index.d.ts +1 -1
  6. package/dist/esm/core/spreadsheet-model/types.d.ts +5 -0
  7. package/dist/esm/excel/column-bands.d.ts +7 -0
  8. package/dist/esm/excel/column-bands.js +87 -0
  9. package/dist/esm/excel/conditional-format.js +22 -0
  10. package/dist/esm/excel/print-model.js +30 -11
  11. package/dist/esm/excel/worksheet-parser.js +26 -0
  12. package/dist/esm/excel/xlsx-writer.js +8 -1
  13. package/dist/esm/html/html-writer.js +100 -5
  14. package/dist/esm/layout/styled-layout.js +88 -19
  15. package/dist/esm/pdf-reader/cmap.d.ts +5 -0
  16. package/dist/esm/pdf-reader/cmap.js +72 -0
  17. package/dist/esm/pdf-reader/content.d.ts +14 -0
  18. package/dist/esm/pdf-reader/content.js +292 -0
  19. package/dist/esm/pdf-reader/document.d.ts +23 -0
  20. package/dist/esm/pdf-reader/document.js +230 -0
  21. package/dist/esm/pdf-reader/flow-build.d.ts +4 -0
  22. package/dist/esm/pdf-reader/flow-build.js +27 -0
  23. package/dist/esm/pdf-reader/font.d.ts +4 -0
  24. package/dist/esm/pdf-reader/font.js +69 -0
  25. package/dist/esm/pdf-reader/layout.d.ts +3 -0
  26. package/dist/esm/pdf-reader/layout.js +69 -0
  27. package/dist/esm/pdf-reader/lexer.d.ts +43 -0
  28. package/dist/esm/pdf-reader/lexer.js +250 -0
  29. package/dist/esm/pdf-reader/parser.d.ts +10 -0
  30. package/dist/esm/pdf-reader/parser.js +86 -0
  31. package/dist/esm/pdf-reader/reader.d.ts +4 -0
  32. package/dist/esm/pdf-reader/reader.js +43 -0
  33. package/dist/esm/pdf-reader/struct-tree.d.ts +14 -0
  34. package/dist/esm/pdf-reader/struct-tree.js +92 -0
  35. package/dist/esm/pdf-reader/tagged.d.ts +3 -0
  36. package/dist/esm/pdf-reader/tagged.js +88 -0
  37. package/dist/esm/pdf-reader/text.d.ts +3 -0
  38. package/dist/esm/pdf-reader/text.js +18 -0
  39. package/package.json +2 -2
@@ -0,0 +1,69 @@
1
+ import { PDF_NULL, PdfName, PdfStream } from "../pdf/objects.js";
2
+ import { parseToUnicodeCMap } from "./cmap.js";
3
+ //#region src/pdf-reader/font.ts
4
+ function buildContentFont(file, fontDict) {
5
+ const isType0 = asName(file.resolve(fontDict.get("Subtype") ?? PDF_NULL)) === "Type0";
6
+ let toUnicode = /* @__PURE__ */ new Map();
7
+ let codeBytes = isType0 ? 2 : 1;
8
+ const tu = file.resolve(fontDict.get("ToUnicode") ?? PDF_NULL);
9
+ if (tu instanceof PdfStream) {
10
+ const parsed = parseToUnicodeCMap(file.streamData(tu));
11
+ toUnicode = parsed.map;
12
+ codeBytes = parsed.codeBytes;
13
+ }
14
+ const width = isType0 ? cidWidths(file, fontDict) : simpleWidths(file, fontDict);
15
+ const bytesPerCode = codeBytes;
16
+ return {
17
+ bytesPerCode,
18
+ decode: (codes) => codes.map((c) => toUnicode.get(c) ?? (bytesPerCode === 1 ? String.fromCharCode(c) : "")).join(""),
19
+ width
20
+ };
21
+ }
22
+ function simpleWidths(file, fontDict) {
23
+ const first = asNumber(file.resolve(fontDict.get("FirstChar") ?? PDF_NULL), 0);
24
+ const widthsVal = file.resolve(fontDict.get("Widths") ?? PDF_NULL);
25
+ const widths = Array.isArray(widthsVal) ? widthsVal : [];
26
+ const descriptor = file.resolve(fontDict.get("FontDescriptor") ?? PDF_NULL);
27
+ const missing = descriptor instanceof Map ? asNumber(file.resolve(descriptor.get("MissingWidth") ?? PDF_NULL), 0) : 0;
28
+ return (code) => {
29
+ const w = widths[code - first];
30
+ return typeof w === "number" ? w : missing > 0 ? missing : 500;
31
+ };
32
+ }
33
+ function cidWidths(file, fontDict) {
34
+ const descFonts = file.resolve(fontDict.get("DescendantFonts") ?? PDF_NULL);
35
+ const desc0 = Array.isArray(descFonts) ? file.resolve(descFonts[0] ?? PDF_NULL) : PDF_NULL;
36
+ const cidFont = desc0 instanceof Map ? desc0 : /* @__PURE__ */ new Map();
37
+ const dw = asNumber(file.resolve(cidFont.get("DW") ?? PDF_NULL), 1e3);
38
+ const wMap = parseCidW(file, file.resolve(cidFont.get("W") ?? PDF_NULL));
39
+ return (cid) => wMap.get(cid) ?? (dw || 1e3);
40
+ }
41
+ function parseCidW(file, wVal) {
42
+ const out = /* @__PURE__ */ new Map();
43
+ if (!Array.isArray(wVal)) return out;
44
+ let i = 0;
45
+ while (i < wVal.length) {
46
+ const c = file.resolve(wVal[i++]);
47
+ if (typeof c !== "number") break;
48
+ const next = file.resolve(wVal[i] ?? PDF_NULL);
49
+ if (Array.isArray(next)) {
50
+ i++;
51
+ next.forEach((w, k) => {
52
+ if (typeof w === "number") out.set(c + k, w);
53
+ });
54
+ } else if (typeof next === "number") {
55
+ i++;
56
+ const w = file.resolve(wVal[i++] ?? PDF_NULL);
57
+ if (typeof w === "number") for (let cc = c; cc <= next && cc - c < 65536; cc++) out.set(cc, w);
58
+ } else break;
59
+ }
60
+ return out;
61
+ }
62
+ function asName(v) {
63
+ return v instanceof PdfName ? v.value : "";
64
+ }
65
+ function asNumber(v, dflt) {
66
+ return typeof v === "number" ? v : dflt;
67
+ }
68
+ //#endregion
69
+ export { buildContentFont };
@@ -0,0 +1,3 @@
1
+ import { FlowDoc } from '../core/ir/flow.js';
2
+ import { PdfFile } from './document.js';
3
+ export declare function reconstructByLayout(file: PdfFile): FlowDoc;
@@ -0,0 +1,69 @@
1
+ import { buildFlowDoc, paragraphBlock } from "./flow-build.js";
2
+ import { extractPageText } from "./text.js";
3
+ //#region src/pdf-reader/layout.ts
4
+ function reconstructByLayout(file) {
5
+ const pages = file.pages().map((page) => groupIntoLines(extractPageText(file, page)).filter((l) => l.text.length > 0));
6
+ const medianFont = median(pages.flat().map((l) => l.fontSize)) || 12;
7
+ const body = [];
8
+ for (const lines of pages) for (const para of groupIntoParagraphs(lines)) body.push(paragraphBlock(para.text, headingLevel(para.fontSize, medianFont)));
9
+ return buildFlowDoc(body);
10
+ }
11
+ function groupIntoLines(runs) {
12
+ const sorted = [...runs].sort((a, b) => b.y - a.y || a.x - b.x);
13
+ const clusters = [];
14
+ for (const run of sorted) {
15
+ const last = clusters[clusters.length - 1];
16
+ const tol = Math.max(1, (run.fontSizePt || 10) * .5);
17
+ if (last && Math.abs(last.y - run.y) <= tol) {
18
+ last.runs.push(run);
19
+ last.fontSize = Math.max(last.fontSize, run.fontSizePt || 0);
20
+ } else clusters.push({
21
+ y: run.y,
22
+ fontSize: run.fontSizePt || 10,
23
+ runs: [run]
24
+ });
25
+ }
26
+ return clusters.map((c) => {
27
+ const ordered = c.runs.sort((a, b) => a.x - b.x);
28
+ return {
29
+ y: c.y,
30
+ fontSize: c.fontSize || 10,
31
+ text: joinLine(ordered, c.fontSize || 10)
32
+ };
33
+ });
34
+ }
35
+ function joinLine(runs, fontSize) {
36
+ let text = "";
37
+ let prevEnd;
38
+ for (const run of runs) {
39
+ if (prevEnd !== void 0 && run.x - prevEnd > fontSize * .25) text += " ";
40
+ text += run.text;
41
+ prevEnd = run.x + run.text.length * (run.fontSizePt || fontSize) * .5;
42
+ }
43
+ return text.replace(/\s+/g, " ").trim();
44
+ }
45
+ function groupIntoParagraphs(lines) {
46
+ const groups = [];
47
+ let prevY;
48
+ for (const line of lines) {
49
+ const gap = prevY !== void 0 ? prevY - line.y : 0;
50
+ if (groups.length === 0 || prevY !== void 0 && gap > line.fontSize * 1.5) groups.push([]);
51
+ groups[groups.length - 1].push(line);
52
+ prevY = line.y;
53
+ }
54
+ return groups.map((g) => ({
55
+ text: g.map((l) => l.text).join(" ").replace(/\s+/g, " ").trim(),
56
+ fontSize: Math.max(...g.map((l) => l.fontSize))
57
+ }));
58
+ }
59
+ function headingLevel(fontSize, medianFont) {
60
+ if (fontSize >= medianFont * 1.5) return 0;
61
+ if (fontSize >= medianFont * 1.25) return 1;
62
+ }
63
+ function median(values) {
64
+ if (values.length === 0) return 0;
65
+ const sorted = [...values].sort((a, b) => a - b);
66
+ return sorted[Math.floor(sorted.length / 2)];
67
+ }
68
+ //#endregion
69
+ export { reconstructByLayout };
@@ -0,0 +1,43 @@
1
+ export type Token = {
2
+ readonly kind: 'num';
3
+ readonly value: number;
4
+ } | {
5
+ readonly kind: 'name';
6
+ readonly value: string;
7
+ } | {
8
+ readonly kind: 'str';
9
+ readonly value: string;
10
+ } | {
11
+ readonly kind: 'hexstr';
12
+ readonly bytes: Uint8Array;
13
+ } | {
14
+ readonly kind: 'arrayOpen';
15
+ } | {
16
+ readonly kind: 'arrayClose';
17
+ } | {
18
+ readonly kind: 'dictOpen';
19
+ } | {
20
+ readonly kind: 'dictClose';
21
+ } | {
22
+ readonly kind: 'keyword';
23
+ readonly value: string;
24
+ } | {
25
+ readonly kind: 'eof';
26
+ };
27
+ export declare class Lexer {
28
+ private readonly buf;
29
+ pos: number;
30
+ constructor(buf: Uint8Array, pos?: number);
31
+ get length(): number;
32
+ byteAt(i: number): number;
33
+ skipWhitespace(): void;
34
+ nextToken(): Token;
35
+ private readNumber;
36
+ private readName;
37
+ private readKeyword;
38
+ private readHexString;
39
+ private readLiteralString;
40
+ indexOfAscii(needle: string, from: number): number;
41
+ readStreamBody(length: number | undefined): Uint8Array;
42
+ }
43
+ export declare function latin1(bytes: Uint8Array): string;
@@ -0,0 +1,250 @@
1
+ //#region src/pdf-reader/lexer.ts
2
+ function isWhitespace(b) {
3
+ return b === 0 || b === 9 || b === 10 || b === 12 || b === 13 || b === 32;
4
+ }
5
+ function isDelimiter(b) {
6
+ return b === 40 || b === 41 || b === 60 || b === 62 || b === 91 || b === 93 || b === 123 || b === 125 || b === 47 || b === 37;
7
+ }
8
+ function isRegular(b) {
9
+ return !isWhitespace(b) && !isDelimiter(b);
10
+ }
11
+ function hexVal(b) {
12
+ if (b >= 48 && b <= 57) return b - 48;
13
+ if (b >= 65 && b <= 70) return b - 65 + 10;
14
+ if (b >= 97 && b <= 102) return b - 97 + 10;
15
+ return -1;
16
+ }
17
+ var Lexer = class {
18
+ pos;
19
+ constructor(buf, pos = 0) {
20
+ this.buf = buf;
21
+ this.pos = pos;
22
+ }
23
+ get length() {
24
+ return this.buf.length;
25
+ }
26
+ byteAt(i) {
27
+ return i >= 0 && i < this.buf.length ? this.buf[i] : -1;
28
+ }
29
+ skipWhitespace() {
30
+ const buf = this.buf;
31
+ while (this.pos < buf.length) {
32
+ const b = buf[this.pos];
33
+ if (isWhitespace(b)) this.pos++;
34
+ else if (b === 37) {
35
+ this.pos++;
36
+ while (this.pos < buf.length && buf[this.pos] !== 10 && buf[this.pos] !== 13) this.pos++;
37
+ } else break;
38
+ }
39
+ }
40
+ nextToken() {
41
+ this.skipWhitespace();
42
+ const buf = this.buf;
43
+ if (this.pos >= buf.length) return { kind: "eof" };
44
+ const b = buf[this.pos];
45
+ switch (b) {
46
+ case 91:
47
+ this.pos++;
48
+ return { kind: "arrayOpen" };
49
+ case 93:
50
+ this.pos++;
51
+ return { kind: "arrayClose" };
52
+ case 60:
53
+ if (buf[this.pos + 1] === 60) {
54
+ this.pos += 2;
55
+ return { kind: "dictOpen" };
56
+ }
57
+ return this.readHexString();
58
+ case 62:
59
+ if (buf[this.pos + 1] === 62) {
60
+ this.pos += 2;
61
+ return { kind: "dictClose" };
62
+ }
63
+ this.pos++;
64
+ return this.nextToken();
65
+ case 40: return this.readLiteralString();
66
+ case 47: return this.readName();
67
+ case 123:
68
+ case 125:
69
+ this.pos++;
70
+ return {
71
+ kind: "keyword",
72
+ value: String.fromCharCode(b)
73
+ };
74
+ }
75
+ if (b === 43 || b === 45 || b === 46 || b >= 48 && b <= 57) return this.readNumber();
76
+ if (isRegular(b)) return this.readKeyword();
77
+ this.pos++;
78
+ return this.nextToken();
79
+ }
80
+ readNumber() {
81
+ const buf = this.buf;
82
+ const start = this.pos;
83
+ if (buf[this.pos] === 43 || buf[this.pos] === 45) this.pos++;
84
+ while (this.pos < buf.length) {
85
+ const b = buf[this.pos];
86
+ if (b >= 48 && b <= 57 || b === 46) this.pos++;
87
+ else break;
88
+ }
89
+ const text = latin1(buf.subarray(start, this.pos));
90
+ const value = Number(text);
91
+ return {
92
+ kind: "num",
93
+ value: Number.isFinite(value) ? value : 0
94
+ };
95
+ }
96
+ readName() {
97
+ const buf = this.buf;
98
+ this.pos++;
99
+ const out = [];
100
+ while (this.pos < buf.length) {
101
+ const b = buf[this.pos];
102
+ if (!isRegular(b)) break;
103
+ if (b === 35 && this.pos + 2 < buf.length) {
104
+ const hi = hexVal(buf[this.pos + 1]);
105
+ const lo = hexVal(buf[this.pos + 2]);
106
+ if (hi >= 0 && lo >= 0) {
107
+ out.push(hi * 16 + lo);
108
+ this.pos += 3;
109
+ continue;
110
+ }
111
+ }
112
+ out.push(b);
113
+ this.pos++;
114
+ }
115
+ return {
116
+ kind: "name",
117
+ value: latin1(Uint8Array.from(out))
118
+ };
119
+ }
120
+ readKeyword() {
121
+ const buf = this.buf;
122
+ const start = this.pos;
123
+ while (this.pos < buf.length && isRegular(buf[this.pos])) this.pos++;
124
+ return {
125
+ kind: "keyword",
126
+ value: latin1(buf.subarray(start, this.pos))
127
+ };
128
+ }
129
+ readHexString() {
130
+ const buf = this.buf;
131
+ this.pos++;
132
+ const out = [];
133
+ let hi = -1;
134
+ while (this.pos < buf.length) {
135
+ const b = buf[this.pos];
136
+ this.pos++;
137
+ if (b === 62) break;
138
+ const v = hexVal(b);
139
+ if (v < 0) continue;
140
+ if (hi < 0) hi = v;
141
+ else {
142
+ out.push(hi * 16 + v);
143
+ hi = -1;
144
+ }
145
+ }
146
+ if (hi >= 0) out.push(hi * 16);
147
+ return {
148
+ kind: "hexstr",
149
+ bytes: Uint8Array.from(out)
150
+ };
151
+ }
152
+ readLiteralString() {
153
+ const buf = this.buf;
154
+ this.pos++;
155
+ const out = [];
156
+ let depth = 1;
157
+ while (this.pos < buf.length) {
158
+ const b = buf[this.pos];
159
+ this.pos++;
160
+ if (b === 92) {
161
+ if (this.pos >= buf.length) break;
162
+ const e = buf[this.pos];
163
+ this.pos++;
164
+ switch (e) {
165
+ case 110:
166
+ out.push(10);
167
+ break;
168
+ case 114:
169
+ out.push(13);
170
+ break;
171
+ case 116:
172
+ out.push(9);
173
+ break;
174
+ case 98:
175
+ out.push(8);
176
+ break;
177
+ case 102:
178
+ out.push(12);
179
+ break;
180
+ case 10: break;
181
+ case 13:
182
+ if (buf[this.pos] === 10) this.pos++;
183
+ break;
184
+ default: if (e >= 48 && e <= 55) {
185
+ let oct = e - 48;
186
+ for (let k = 0; k < 2 && this.pos < buf.length; k++) {
187
+ const d = buf[this.pos];
188
+ if (d < 48 || d > 55) break;
189
+ oct = oct * 8 + (d - 48);
190
+ this.pos++;
191
+ }
192
+ out.push(oct & 255);
193
+ } else out.push(e);
194
+ }
195
+ continue;
196
+ }
197
+ if (b === 40) {
198
+ depth++;
199
+ out.push(b);
200
+ continue;
201
+ }
202
+ if (b === 41) {
203
+ depth--;
204
+ if (depth === 0) break;
205
+ out.push(b);
206
+ continue;
207
+ }
208
+ out.push(b);
209
+ }
210
+ return {
211
+ kind: "str",
212
+ value: latin1(Uint8Array.from(out))
213
+ };
214
+ }
215
+ indexOfAscii(needle, from) {
216
+ const buf = this.buf;
217
+ const n = needle.length;
218
+ outer: for (let i = from; i <= buf.length - n; i++) {
219
+ for (let j = 0; j < n; j++) if (buf[i + j] !== needle.charCodeAt(j)) continue outer;
220
+ return i;
221
+ }
222
+ return -1;
223
+ }
224
+ readStreamBody(length) {
225
+ const buf = this.buf;
226
+ if (buf[this.pos] === 13 && buf[this.pos + 1] === 10) this.pos += 2;
227
+ else if (buf[this.pos] === 10 || buf[this.pos] === 13) this.pos += 1;
228
+ const start = this.pos;
229
+ if (length !== void 0 && length >= 0 && start + length <= buf.length) {
230
+ this.pos = start + length;
231
+ return buf.subarray(start, start + length);
232
+ }
233
+ const es = this.indexOfAscii("endstream", start);
234
+ const end = es < 0 ? buf.length : es;
235
+ let dataEnd = end;
236
+ if (dataEnd > start && buf[dataEnd - 1] === 10) {
237
+ dataEnd--;
238
+ if (dataEnd > start && buf[dataEnd - 1] === 13) dataEnd--;
239
+ } else if (dataEnd > start && buf[dataEnd - 1] === 13) dataEnd--;
240
+ this.pos = end;
241
+ return buf.subarray(start, dataEnd);
242
+ }
243
+ };
244
+ function latin1(bytes) {
245
+ let s = "";
246
+ for (const b of bytes) s += String.fromCharCode(b);
247
+ return s;
248
+ }
249
+ //#endregion
250
+ export { Lexer };
@@ -0,0 +1,10 @@
1
+ import { Lexer } from './lexer.js';
2
+ import { PdfValue, PdfRef } from '../pdf/objects.js';
3
+ export type LengthResolver = (ref: PdfRef) => number | undefined;
4
+ export interface IndirectObject {
5
+ readonly id: number;
6
+ readonly generation: number;
7
+ readonly value: PdfValue;
8
+ }
9
+ export declare function parseObject(lexer: Lexer, resolveLength?: LengthResolver): PdfValue;
10
+ export declare function parseIndirectObject(lexer: Lexer, resolveLength?: LengthResolver): IndirectObject | undefined;
@@ -0,0 +1,86 @@
1
+ import { PDF_NULL, PdfHexString, PdfName, PdfRef, PdfStream } from "../pdf/objects.js";
2
+ //#region src/pdf-reader/parser.ts
3
+ function parseObject(lexer, resolveLength) {
4
+ return parseValue(lexer, lexer.nextToken(), resolveLength);
5
+ }
6
+ function parseIndirectObject(lexer, resolveLength) {
7
+ const idTok = lexer.nextToken();
8
+ if (idTok.kind !== "num") return void 0;
9
+ const genTok = lexer.nextToken();
10
+ if (genTok.kind !== "num") return void 0;
11
+ const objTok = lexer.nextToken();
12
+ if (objTok.kind !== "keyword" || objTok.value !== "obj") return void 0;
13
+ const value = parseObject(lexer, resolveLength);
14
+ const save = lexer.pos;
15
+ const end = lexer.nextToken();
16
+ if (!(end.kind === "keyword" && end.value === "endobj")) lexer.pos = save;
17
+ return {
18
+ id: idTok.value,
19
+ generation: genTok.value,
20
+ value
21
+ };
22
+ }
23
+ function parseValue(lexer, tok, resolveLength) {
24
+ switch (tok.kind) {
25
+ case "num": return parseNumberOrRef(lexer, tok.value);
26
+ case "name": return new PdfName(tok.value);
27
+ case "str": return tok.value;
28
+ case "hexstr": return new PdfHexString(tok.bytes);
29
+ case "arrayOpen": return parseArray(lexer, resolveLength);
30
+ case "dictOpen": return parseDictOrStream(lexer, resolveLength);
31
+ case "keyword":
32
+ if (tok.value === "true") return true;
33
+ if (tok.value === "false") return false;
34
+ return PDF_NULL;
35
+ default: return PDF_NULL;
36
+ }
37
+ }
38
+ function parseNumberOrRef(lexer, first) {
39
+ if (!Number.isInteger(first) || first < 0) return first;
40
+ const save = lexer.pos;
41
+ const gen = lexer.nextToken();
42
+ if (gen.kind === "num" && Number.isInteger(gen.value)) {
43
+ const r = lexer.nextToken();
44
+ if (r.kind === "keyword" && r.value === "R") return new PdfRef(first, gen.value);
45
+ }
46
+ lexer.pos = save;
47
+ return first;
48
+ }
49
+ function parseArray(lexer, resolveLength) {
50
+ const out = [];
51
+ for (;;) {
52
+ const tok = lexer.nextToken();
53
+ if (tok.kind === "arrayClose" || tok.kind === "eof") break;
54
+ out.push(parseValue(lexer, tok, resolveLength));
55
+ }
56
+ return out;
57
+ }
58
+ function parseDictOrStream(lexer, resolveLength) {
59
+ const map = /* @__PURE__ */ new Map();
60
+ for (;;) {
61
+ const keyTok = lexer.nextToken();
62
+ if (keyTok.kind === "dictClose" || keyTok.kind === "eof") break;
63
+ if (keyTok.kind !== "name") {
64
+ if (keyTok.kind === "arrayOpen" || keyTok.kind === "dictOpen") parseValue(lexer, keyTok, resolveLength);
65
+ continue;
66
+ }
67
+ const value = parseObject(lexer, resolveLength);
68
+ map.set(keyTok.value, value);
69
+ }
70
+ const save = lexer.pos;
71
+ const next = lexer.nextToken();
72
+ if (next.kind === "keyword" && next.value === "stream") {
73
+ const lengthVal = map.get("Length");
74
+ let length;
75
+ if (typeof lengthVal === "number") length = lengthVal;
76
+ else if (lengthVal instanceof PdfRef && resolveLength) length = resolveLength(lengthVal);
77
+ const data = lexer.readStreamBody(length);
78
+ const endTok = lexer.nextToken();
79
+ if (!(endTok.kind === "keyword" && endTok.value === "endstream")) {}
80
+ return new PdfStream(map, data);
81
+ }
82
+ lexer.pos = save;
83
+ return map;
84
+ }
85
+ //#endregion
86
+ export { parseIndirectObject, parseObject };
@@ -0,0 +1,4 @@
1
+ import { DocumentReader, ReadResult } from '../core/ir/adapters.js';
2
+ import { FlowDoc } from '../core/ir/flow.js';
3
+ export declare function readPdf(bytes: Uint8Array): ReadResult<FlowDoc>;
4
+ export declare const pdfReader: DocumentReader<FlowDoc>;
@@ -0,0 +1,43 @@
1
+ import { FEATURES } from "../core/ir/features.js";
2
+ import { PdfFile } from "./document.js";
3
+ import { reconstructByLayout } from "./layout.js";
4
+ import { reconstructTaggedPdf } from "./tagged.js";
5
+ //#region src/pdf-reader/reader.ts
6
+ function sniffPdf(bytes) {
7
+ const limit = Math.min(bytes.length - 5, 1024);
8
+ for (let i = 0; i <= limit; i++) if (bytes[i] === 37 && bytes[i + 1] === 80 && bytes[i + 2] === 68 && bytes[i + 3] === 70 && bytes[i + 4] === 45) return true;
9
+ return false;
10
+ }
11
+ function readPdf(bytes) {
12
+ const file = PdfFile.parse(bytes);
13
+ const losses = [];
14
+ const tagged = reconstructTaggedPdf(file);
15
+ const doc = tagged ?? reconstructByLayout(file);
16
+ if (!tagged) losses.push({
17
+ severity: "degraded",
18
+ feature: FEATURES.text,
19
+ detail: "untagged PDF — text and headings reconstructed heuristically from glyph positions; structure is approximate"
20
+ });
21
+ losses.push({
22
+ severity: "dropped",
23
+ feature: FEATURES.images,
24
+ detail: "PDF images and vector graphics are not reconstructed"
25
+ });
26
+ return {
27
+ doc,
28
+ losses
29
+ };
30
+ }
31
+ var pdfReader = {
32
+ id: "pdf",
33
+ produces: "flow",
34
+ supports: new Set([
35
+ FEATURES.text,
36
+ FEATURES.tables,
37
+ FEATURES.lists
38
+ ]),
39
+ sniff: sniffPdf,
40
+ read: (bytes) => readPdf(bytes)
41
+ };
42
+ //#endregion
43
+ export { pdfReader };
@@ -0,0 +1,14 @@
1
+ import { PdfFile } from './document.js';
2
+ export interface StructMcid {
3
+ readonly page: number;
4
+ readonly mcid: number;
5
+ }
6
+ export interface StructNode {
7
+ readonly type: string;
8
+ readonly mcids: ReadonlyArray<StructMcid>;
9
+ readonly children: ReadonlyArray<StructNode>;
10
+ readonly alt?: string;
11
+ readonly colSpan?: number;
12
+ readonly rowSpan?: number;
13
+ }
14
+ export declare function readStructTree(file: PdfFile): StructNode | undefined;
@@ -0,0 +1,92 @@
1
+ import { PDF_NULL, PdfName } from "../pdf/objects.js";
2
+ //#region src/pdf-reader/struct-tree.ts
3
+ var MAX_NODES = 2e5;
4
+ function readStructTree(file) {
5
+ const stRoot = file.get(file.catalog, "StructTreeRoot");
6
+ if (!(stRoot instanceof Map)) return void 0;
7
+ const pageMap = /* @__PURE__ */ new Map();
8
+ file.pages().forEach((p, i) => pageMap.set(p.dict, i));
9
+ const pageIndexOf = (pgVal) => {
10
+ const pg = file.resolve(pgVal);
11
+ return pg instanceof Map ? pageMap.get(pg) : void 0;
12
+ };
13
+ const seen = /* @__PURE__ */ new Set();
14
+ const read = (value, parentPage) => {
15
+ const elem = file.resolve(value);
16
+ if (!(elem instanceof Map) || seen.has(elem) || seen.size > MAX_NODES) return void 0;
17
+ seen.add(elem);
18
+ const ownPage = pageIndexOf(elem.get("Pg") ?? PDF_NULL) ?? parentPage;
19
+ const mcids = [];
20
+ const children = [];
21
+ for (const kid of kidList(file, elem.get("K"))) {
22
+ const rk = file.resolve(kid);
23
+ if (typeof rk === "number") {
24
+ if (ownPage >= 0) mcids.push({
25
+ page: ownPage,
26
+ mcid: rk
27
+ });
28
+ } else if (rk instanceof Map) {
29
+ const kind = nameOf(rk.get("Type"));
30
+ if (kind === "MCR") {
31
+ const m = rk.get("MCID");
32
+ const page = pageIndexOf(rk.get("Pg") ?? PDF_NULL) ?? ownPage;
33
+ if (typeof m === "number" && page >= 0) mcids.push({
34
+ page,
35
+ mcid: m
36
+ });
37
+ } else if (kind === "OBJR") {} else {
38
+ const child = read(rk, ownPage);
39
+ if (child) children.push(child);
40
+ }
41
+ }
42
+ }
43
+ const alt = elem.get("Alt");
44
+ const { colSpan, rowSpan } = readSpans(file, elem.get("A") ?? PDF_NULL);
45
+ return {
46
+ type: nameOf(elem.get("S")),
47
+ mcids,
48
+ children,
49
+ ...typeof alt === "string" ? { alt } : {},
50
+ ...colSpan > 1 ? { colSpan } : {},
51
+ ...rowSpan > 1 ? { rowSpan } : {}
52
+ };
53
+ };
54
+ const roots = kidList(file, stRoot.get("K")).map((k) => read(k, -1)).filter((n) => n !== void 0);
55
+ if (roots.length === 1) return roots[0];
56
+ return {
57
+ type: "Document",
58
+ mcids: [],
59
+ children: roots
60
+ };
61
+ }
62
+ function kidList(file, kVal) {
63
+ if (kVal === void 0) return [];
64
+ const k = file.resolve(kVal);
65
+ if (Array.isArray(k)) return k;
66
+ if (k === PDF_NULL) return [];
67
+ return [k];
68
+ }
69
+ function nameOf(v) {
70
+ return v instanceof PdfName ? v.value : "";
71
+ }
72
+ function readSpans(file, aVal) {
73
+ let colSpan = 1;
74
+ let rowSpan = 1;
75
+ const a = file.resolve(aVal);
76
+ const attrs = Array.isArray(a) ? a : [a];
77
+ for (const entry of attrs) {
78
+ const d = file.resolve(entry);
79
+ if (d instanceof Map) {
80
+ const cs = d.get("ColSpan");
81
+ const rs = d.get("RowSpan");
82
+ if (typeof cs === "number") colSpan = cs;
83
+ if (typeof rs === "number") rowSpan = rs;
84
+ }
85
+ }
86
+ return {
87
+ colSpan,
88
+ rowSpan
89
+ };
90
+ }
91
+ //#endregion
92
+ export { readStructTree };
@@ -0,0 +1,3 @@
1
+ import { FlowDoc } from '../core/ir/flow.js';
2
+ import { PdfFile } from './document.js';
3
+ export declare function reconstructTaggedPdf(file: PdfFile): FlowDoc | undefined;