reamkit 1.4.0 → 1.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (62) hide show
  1. package/README.md +10 -8
  2. package/dist/esm/core/converter/facade.d.ts +12 -6
  3. package/dist/esm/core/converter/facade.js +34 -5
  4. package/dist/esm/core/converter/ream.d.ts +5 -3
  5. package/dist/esm/core/converter/ream.js +18 -4
  6. package/dist/esm/core/document-model/index.d.ts +1 -1
  7. package/dist/esm/core/document-model/types.d.ts +26 -0
  8. package/dist/esm/core/drawingml/sparkline-geometry.d.ts +8 -0
  9. package/dist/esm/core/drawingml/sparkline-geometry.js +111 -0
  10. package/dist/esm/core/ir/adapters.d.ts +5 -5
  11. package/dist/esm/core/ir/sheet.d.ts +31 -0
  12. package/dist/esm/core/opc/relationships.js +1 -0
  13. package/dist/esm/core/spreadsheet-model/index.d.ts +1 -0
  14. package/dist/esm/core/spreadsheet-model/types.d.ts +189 -0
  15. package/dist/esm/excel/column-bands.d.ts +7 -0
  16. package/dist/esm/excel/column-bands.js +87 -0
  17. package/dist/esm/excel/conditional-format.d.ts +16 -0
  18. package/dist/esm/excel/conditional-format.js +327 -0
  19. package/dist/esm/excel/index.d.ts +4 -3
  20. package/dist/esm/excel/print-model.d.ts +2 -1
  21. package/dist/esm/excel/print-model.js +144 -13
  22. package/dist/esm/excel/sheet-drawing.d.ts +1 -1
  23. package/dist/esm/excel/sheet-to-flow.d.ts +3 -0
  24. package/dist/esm/excel/sheet-to-flow.js +51 -0
  25. package/dist/esm/excel/styles-parser.d.ts +1 -50
  26. package/dist/esm/excel/styles-parser.js +38 -1
  27. package/dist/esm/excel/table-parser.d.ts +2 -0
  28. package/dist/esm/excel/table-parser.js +51 -0
  29. package/dist/esm/excel/workbook-parser.d.ts +1 -5
  30. package/dist/esm/excel/worksheet-parser.d.ts +1 -60
  31. package/dist/esm/excel/worksheet-parser.js +265 -1
  32. package/dist/esm/excel/xlsx-reader.d.ts +3 -1
  33. package/dist/esm/excel/xlsx-reader.js +86 -51
  34. package/dist/esm/excel/xlsx-writer.d.ts +4 -0
  35. package/dist/esm/excel/xlsx-writer.js +278 -0
  36. package/dist/esm/html/html-writer.js +126 -4
  37. package/dist/esm/layout/styled-layout.js +160 -1
  38. package/dist/esm/pdf-reader/cmap.d.ts +5 -0
  39. package/dist/esm/pdf-reader/cmap.js +72 -0
  40. package/dist/esm/pdf-reader/content.d.ts +14 -0
  41. package/dist/esm/pdf-reader/content.js +292 -0
  42. package/dist/esm/pdf-reader/document.d.ts +23 -0
  43. package/dist/esm/pdf-reader/document.js +230 -0
  44. package/dist/esm/pdf-reader/flow-build.d.ts +4 -0
  45. package/dist/esm/pdf-reader/flow-build.js +27 -0
  46. package/dist/esm/pdf-reader/font.d.ts +4 -0
  47. package/dist/esm/pdf-reader/font.js +69 -0
  48. package/dist/esm/pdf-reader/layout.d.ts +3 -0
  49. package/dist/esm/pdf-reader/layout.js +69 -0
  50. package/dist/esm/pdf-reader/lexer.d.ts +43 -0
  51. package/dist/esm/pdf-reader/lexer.js +250 -0
  52. package/dist/esm/pdf-reader/parser.d.ts +10 -0
  53. package/dist/esm/pdf-reader/parser.js +86 -0
  54. package/dist/esm/pdf-reader/reader.d.ts +4 -0
  55. package/dist/esm/pdf-reader/reader.js +43 -0
  56. package/dist/esm/pdf-reader/struct-tree.d.ts +14 -0
  57. package/dist/esm/pdf-reader/struct-tree.js +92 -0
  58. package/dist/esm/pdf-reader/tagged.d.ts +3 -0
  59. package/dist/esm/pdf-reader/tagged.js +88 -0
  60. package/dist/esm/pdf-reader/text.d.ts +3 -0
  61. package/dist/esm/pdf-reader/text.js +18 -0
  62. package/package.json +4 -3
@@ -0,0 +1,72 @@
1
+ import { Lexer } from "./lexer.js";
2
+ //#region src/pdf-reader/cmap.ts
3
+ var MAX_RANGE = 65536;
4
+ function parseToUnicodeCMap(bytes) {
5
+ const lexer = new Lexer(bytes);
6
+ const map = /* @__PURE__ */ new Map();
7
+ let codeBytes = 1;
8
+ for (;;) {
9
+ const tok = lexer.nextToken();
10
+ if (tok.kind === "eof") break;
11
+ if (tok.kind !== "keyword") continue;
12
+ if (tok.value === "begincodespacerange") for (;;) {
13
+ const t = lexer.nextToken();
14
+ if (t.kind === "eof" || t.kind === "keyword" && t.value === "endcodespacerange") break;
15
+ if (t.kind === "hexstr") {
16
+ if (t.bytes.length >= 2) codeBytes = 2;
17
+ lexer.nextToken();
18
+ }
19
+ }
20
+ else if (tok.value === "beginbfchar") for (;;) {
21
+ const src = lexer.nextToken();
22
+ if (src.kind === "eof" || src.kind === "keyword" && src.value === "endbfchar") break;
23
+ if (src.kind !== "hexstr") continue;
24
+ const dst = lexer.nextToken();
25
+ if (dst.kind !== "hexstr") continue;
26
+ map.set(bytesToInt(src.bytes), utf16be(dst.bytes));
27
+ }
28
+ else if (tok.value === "beginbfrange") for (;;) {
29
+ const lo = lexer.nextToken();
30
+ if (lo.kind === "eof" || lo.kind === "keyword" && lo.value === "endbfrange") break;
31
+ if (lo.kind !== "hexstr") continue;
32
+ const hi = lexer.nextToken();
33
+ if (hi.kind !== "hexstr") continue;
34
+ const loN = bytesToInt(lo.bytes);
35
+ const hiN = bytesToInt(hi.bytes);
36
+ const dst = lexer.nextToken();
37
+ if (dst.kind === "hexstr") {
38
+ const base = utf16be(dst.bytes);
39
+ for (let i = 0; loN + i <= hiN && i < MAX_RANGE; i++) map.set(loN + i, incString(base, i));
40
+ } else if (dst.kind === "arrayOpen") {
41
+ let i = 0;
42
+ for (;;) {
43
+ const el = lexer.nextToken();
44
+ if (el.kind === "arrayClose" || el.kind === "eof") break;
45
+ if (el.kind === "hexstr") map.set(loN + i++, utf16be(el.bytes));
46
+ }
47
+ }
48
+ }
49
+ }
50
+ return {
51
+ map,
52
+ codeBytes
53
+ };
54
+ }
55
+ function bytesToInt(bytes) {
56
+ let n = 0;
57
+ for (const b of bytes) n = n * 256 + b;
58
+ return n;
59
+ }
60
+ function utf16be(bytes) {
61
+ let s = "";
62
+ for (let i = 0; i + 1 < bytes.length; i += 2) s += String.fromCharCode(bytes[i] << 8 | bytes[i + 1]);
63
+ if (bytes.length % 2 === 1) s += String.fromCharCode(bytes[bytes.length - 1]);
64
+ return s;
65
+ }
66
+ function incString(base, i) {
67
+ if (i === 0) return base;
68
+ if (base.length === 1) return String.fromCharCode(base.charCodeAt(0) + i);
69
+ return base;
70
+ }
71
+ //#endregion
72
+ export { parseToUnicodeCMap };
@@ -0,0 +1,14 @@
1
+ export interface ContentFont {
2
+ readonly bytesPerCode: 1 | 2;
3
+ decode: (codes: ReadonlyArray<number>) => string;
4
+ width: (code: number) => number;
5
+ }
6
+ export interface TextRun {
7
+ readonly text: string;
8
+ readonly x: number;
9
+ readonly y: number;
10
+ readonly fontSizePt: number;
11
+ readonly fontKey: string;
12
+ readonly mcid?: number;
13
+ }
14
+ export declare function interpretContent(bytes: Uint8Array, fonts: ReadonlyMap<string, ContentFont>): Array<TextRun>;
@@ -0,0 +1,292 @@
1
+ import { PDF_NULL, PdfHexString, PdfName } from "../pdf/objects.js";
2
+ import { Lexer } from "./lexer.js";
3
+ //#region src/pdf-reader/content.ts
4
+ var IDENTITY = [
5
+ 1,
6
+ 0,
7
+ 0,
8
+ 1,
9
+ 0,
10
+ 0
11
+ ];
12
+ function multiply(a, b) {
13
+ return [
14
+ a[0] * b[0] + a[1] * b[2],
15
+ a[0] * b[1] + a[1] * b[3],
16
+ a[2] * b[0] + a[3] * b[2],
17
+ a[2] * b[1] + a[3] * b[3],
18
+ a[4] * b[0] + a[5] * b[2] + b[4],
19
+ a[4] * b[1] + a[5] * b[3] + b[5]
20
+ ];
21
+ }
22
+ function translation(tx, ty) {
23
+ return [
24
+ 1,
25
+ 0,
26
+ 0,
27
+ 1,
28
+ tx,
29
+ ty
30
+ ];
31
+ }
32
+ var FALLBACK_FONT = {
33
+ bytesPerCode: 1,
34
+ decode: (codes) => codes.map((c) => String.fromCharCode(c)).join(""),
35
+ width: () => 500
36
+ };
37
+ function initialState() {
38
+ return {
39
+ ctm: IDENTITY,
40
+ fontKey: "",
41
+ font: FALLBACK_FONT,
42
+ fontSize: 0,
43
+ charSpacing: 0,
44
+ wordSpacing: 0,
45
+ hScale: 1,
46
+ leading: 0,
47
+ rise: 0
48
+ };
49
+ }
50
+ function interpretContent(bytes, fonts) {
51
+ const runs = [];
52
+ const lexer = new Lexer(bytes);
53
+ const stack = [];
54
+ let state = initialState();
55
+ let tm = IDENTITY;
56
+ let tlm = IDENTITY;
57
+ let operands = [];
58
+ const mcStack = [];
59
+ const num = (i) => {
60
+ const v = operands[i];
61
+ return typeof v === "number" ? v : 0;
62
+ };
63
+ const advanceGlyph = (code) => {
64
+ const w0 = state.font.width(code) / 1e3;
65
+ const isSpace = state.font.bytesPerCode === 1 && code === 32;
66
+ tm = multiply(translation((w0 * state.fontSize + state.charSpacing + (isSpace ? state.wordSpacing : 0)) * state.hScale, 0), tm);
67
+ };
68
+ const consume = (operand) => {
69
+ const codes = splitCodes(toBytes(operand), state.font.bytesPerCode);
70
+ for (const code of codes) advanceGlyph(code);
71
+ return state.font.decode(codes);
72
+ };
73
+ const emitAt = (origin, text) => {
74
+ if (text.length === 0) return;
75
+ const scaleY = Math.hypot(origin[2], origin[3]) || 1;
76
+ const mcid = mcStack.length > 0 ? mcStack[mcStack.length - 1] : void 0;
77
+ runs.push({
78
+ text,
79
+ x: origin[4],
80
+ y: origin[5],
81
+ fontSizePt: state.fontSize * scaleY,
82
+ fontKey: state.fontKey,
83
+ ...mcid !== void 0 ? { mcid } : {}
84
+ });
85
+ };
86
+ const showString = (operand) => {
87
+ emitAt(multiply(tm, state.ctm), consume(operand));
88
+ };
89
+ const showArray = (arr) => {
90
+ const origin = multiply(tm, state.ctm);
91
+ let text = "";
92
+ for (const el of arr) if (typeof el === "number") tm = multiply(translation(-el / 1e3 * state.fontSize * state.hScale, 0), tm);
93
+ else if (typeof el === "string" || el instanceof PdfHexString) text += consume(el);
94
+ emitAt(origin, text);
95
+ };
96
+ const exec = (op) => {
97
+ switch (op) {
98
+ case "q":
99
+ stack.push({ ...state });
100
+ break;
101
+ case "Q":
102
+ state = stack.pop() ?? state;
103
+ break;
104
+ case "cm":
105
+ state.ctm = multiply(matrixFromOperands(operands), state.ctm);
106
+ break;
107
+ case "BT":
108
+ tm = IDENTITY;
109
+ tlm = IDENTITY;
110
+ break;
111
+ case "ET": break;
112
+ case "Tf": {
113
+ const key = operands[0] instanceof PdfName ? operands[0].value : "";
114
+ state.fontKey = key;
115
+ state.font = fonts.get(key) ?? FALLBACK_FONT;
116
+ state.fontSize = num(1);
117
+ break;
118
+ }
119
+ case "Td":
120
+ tlm = multiply(translation(num(0), num(1)), tlm);
121
+ tm = tlm;
122
+ break;
123
+ case "TD":
124
+ state.leading = -num(1);
125
+ tlm = multiply(translation(num(0), num(1)), tlm);
126
+ tm = tlm;
127
+ break;
128
+ case "Tm":
129
+ tlm = matrixFromOperands(operands);
130
+ tm = tlm;
131
+ break;
132
+ case "T*":
133
+ tlm = multiply(translation(0, -state.leading), tlm);
134
+ tm = tlm;
135
+ break;
136
+ case "TL":
137
+ state.leading = num(0);
138
+ break;
139
+ case "Tc":
140
+ state.charSpacing = num(0);
141
+ break;
142
+ case "Tw":
143
+ state.wordSpacing = num(0);
144
+ break;
145
+ case "Tz":
146
+ state.hScale = num(0) / 100;
147
+ break;
148
+ case "Ts":
149
+ state.rise = num(0);
150
+ break;
151
+ case "Tj":
152
+ if (operands.length > 0) showString(operands[operands.length - 1]);
153
+ break;
154
+ case "TJ":
155
+ if (Array.isArray(operands[0])) showArray(operands[0]);
156
+ break;
157
+ case "'":
158
+ tlm = multiply(translation(0, -state.leading), tlm);
159
+ tm = tlm;
160
+ if (operands.length > 0) showString(operands[operands.length - 1]);
161
+ break;
162
+ case "\"":
163
+ state.wordSpacing = num(0);
164
+ state.charSpacing = num(1);
165
+ tlm = multiply(translation(0, -state.leading), tlm);
166
+ tm = tlm;
167
+ if (operands.length > 2) showString(operands[2]);
168
+ break;
169
+ case "BDC": {
170
+ const tag = operands[operands.length - 2];
171
+ const props = operands[operands.length - 1];
172
+ const mcidVal = !(tag instanceof PdfName && tag.value === "Artifact") && props instanceof Map ? props.get("MCID") : void 0;
173
+ mcStack.push(typeof mcidVal === "number" ? mcidVal : void 0);
174
+ break;
175
+ }
176
+ case "BMC":
177
+ mcStack.push(void 0);
178
+ break;
179
+ case "EMC":
180
+ mcStack.pop();
181
+ break;
182
+ default: break;
183
+ }
184
+ };
185
+ for (;;) {
186
+ lexer.skipWhitespace();
187
+ const tok = lexer.nextToken();
188
+ if (tok.kind === "eof") break;
189
+ switch (tok.kind) {
190
+ case "num":
191
+ operands.push(tok.value);
192
+ break;
193
+ case "name":
194
+ operands.push(new PdfName(tok.value));
195
+ break;
196
+ case "str":
197
+ operands.push(tok.value);
198
+ break;
199
+ case "hexstr":
200
+ operands.push(new PdfHexString(tok.bytes));
201
+ break;
202
+ case "arrayOpen":
203
+ operands.push(readArray(lexer));
204
+ break;
205
+ case "dictOpen":
206
+ operands.push(readDict(lexer));
207
+ break;
208
+ case "keyword":
209
+ if (tok.value === "BI") skipInlineImage(lexer);
210
+ else exec(tok.value);
211
+ operands = [];
212
+ break;
213
+ default:
214
+ operands = [];
215
+ break;
216
+ }
217
+ }
218
+ return runs;
219
+ }
220
+ function matrixFromOperands(operands) {
221
+ const n = (i) => typeof operands[i] === "number" ? operands[i] : 0;
222
+ return [
223
+ n(0),
224
+ n(1),
225
+ n(2),
226
+ n(3),
227
+ n(4),
228
+ n(5)
229
+ ];
230
+ }
231
+ function toBytes(operand) {
232
+ if (operand instanceof PdfHexString) return operand.bytes;
233
+ if (typeof operand === "string") {
234
+ const out = new Uint8Array(operand.length);
235
+ for (let i = 0; i < operand.length; i++) out[i] = operand.charCodeAt(i) & 255;
236
+ return out;
237
+ }
238
+ return new Uint8Array(0);
239
+ }
240
+ function splitCodes(bytes, bytesPerCode) {
241
+ const out = [];
242
+ if (bytesPerCode === 2) {
243
+ for (let i = 0; i + 1 < bytes.length; i += 2) out.push(bytes[i] << 8 | bytes[i + 1]);
244
+ if (bytes.length % 2 === 1) out.push(bytes[bytes.length - 1]);
245
+ } else for (const b of bytes) out.push(b);
246
+ return out;
247
+ }
248
+ function readArray(lexer) {
249
+ const out = [];
250
+ for (;;) {
251
+ const tok = lexer.nextToken();
252
+ if (tok.kind === "arrayClose" || tok.kind === "eof") break;
253
+ if (tok.kind === "num") out.push(tok.value);
254
+ else if (tok.kind === "str") out.push(tok.value);
255
+ else if (tok.kind === "hexstr") out.push(new PdfHexString(tok.bytes));
256
+ }
257
+ return out;
258
+ }
259
+ function readDict(lexer) {
260
+ const map = /* @__PURE__ */ new Map();
261
+ for (;;) {
262
+ const key = lexer.nextToken();
263
+ if (key.kind === "dictClose" || key.kind === "eof") break;
264
+ if (key.kind !== "name") continue;
265
+ map.set(key.value, readValue(lexer));
266
+ }
267
+ return map;
268
+ }
269
+ function readValue(lexer) {
270
+ const tok = lexer.nextToken();
271
+ switch (tok.kind) {
272
+ case "num": return tok.value;
273
+ case "name": return new PdfName(tok.value);
274
+ case "str": return tok.value;
275
+ case "hexstr": return new PdfHexString(tok.bytes);
276
+ case "arrayOpen": return readArray(lexer);
277
+ case "dictOpen": return readDict(lexer);
278
+ case "keyword": return tok.value === "true" ? true : tok.value === "false" ? false : PDF_NULL;
279
+ default: return PDF_NULL;
280
+ }
281
+ }
282
+ function skipInlineImage(lexer) {
283
+ for (;;) {
284
+ const tok = lexer.nextToken();
285
+ if (tok.kind === "eof") return;
286
+ if (tok.kind === "keyword" && tok.value === "ID") break;
287
+ }
288
+ const ei = lexer.indexOfAscii("EI", lexer.pos);
289
+ lexer.pos = ei < 0 ? lexer.length : ei + 2;
290
+ }
291
+ //#endregion
292
+ export { interpretContent };
@@ -0,0 +1,23 @@
1
+ import { PdfArray, PdfDict, PdfValue, PdfStream } from '../pdf/objects.js';
2
+ export type Rectangle = readonly [number, number, number, number];
3
+ export interface PdfPage {
4
+ readonly dict: PdfDict;
5
+ readonly mediaBox: Rectangle;
6
+ readonly resources: PdfDict | undefined;
7
+ }
8
+ export declare class PdfFile {
9
+ private readonly buf;
10
+ private readonly xref;
11
+ readonly trailer: PdfDict;
12
+ private readonly cache;
13
+ private constructor();
14
+ static parse(bytes: Uint8Array): PdfFile;
15
+ resolve(value: PdfValue): PdfValue;
16
+ get(dict: PdfDict, key: string): PdfValue;
17
+ get catalog(): PdfDict;
18
+ pages(): Array<PdfPage>;
19
+ private walkPageTree;
20
+ pageContent(page: PdfPage): Uint8Array;
21
+ streamData(stream: PdfStream): Uint8Array;
22
+ }
23
+ export type { PdfArray };
@@ -0,0 +1,230 @@
1
+ import { PDF_NULL, PdfName, PdfRef, PdfStream } from "../pdf/objects.js";
2
+ import { Lexer } from "./lexer.js";
3
+ import { parseIndirectObject, parseObject } from "./parser.js";
4
+ import { unzlibSync } from "fflate";
5
+ //#region src/pdf-reader/document.ts
6
+ var DEFAULT_MEDIA_BOX = [
7
+ 0,
8
+ 0,
9
+ 612,
10
+ 792
11
+ ];
12
+ var MAX_PAGES = 5e4;
13
+ var PdfFile = class PdfFile {
14
+ cache = /* @__PURE__ */ new Map();
15
+ constructor(buf, xref, trailer) {
16
+ this.buf = buf;
17
+ this.xref = xref;
18
+ this.trailer = trailer;
19
+ }
20
+ static parse(bytes) {
21
+ let xref = /* @__PURE__ */ new Map();
22
+ let trailer = /* @__PURE__ */ new Map();
23
+ try {
24
+ const start = findStartXref(bytes);
25
+ if (start >= 0) {
26
+ const built = readXrefChain(bytes, start);
27
+ xref = built.xref;
28
+ trailer = built.trailer;
29
+ }
30
+ } catch {}
31
+ if (xref.size === 0 || !(trailer.get("Root") instanceof PdfRef)) {
32
+ const scanned = bruteForceScan(bytes);
33
+ for (const [id, off] of scanned.xref) if (!xref.has(id)) xref.set(id, off);
34
+ if (!(trailer.get("Root") instanceof PdfRef) && scanned.root) {
35
+ trailer = new Map(trailer);
36
+ trailer.set("Root", scanned.root);
37
+ }
38
+ }
39
+ return new PdfFile(bytes, xref, trailer);
40
+ }
41
+ resolve(value) {
42
+ if (!(value instanceof PdfRef)) return value;
43
+ const cached = this.cache.get(value.id);
44
+ if (cached !== void 0) return cached;
45
+ const offset = this.xref.get(value.id);
46
+ if (offset === void 0 || offset < 0 || offset >= this.buf.length) return PDF_NULL;
47
+ this.cache.set(value.id, PDF_NULL);
48
+ const obj = parseIndirectObject(new Lexer(this.buf, offset), (r) => {
49
+ const n = this.resolve(r);
50
+ return typeof n === "number" ? n : void 0;
51
+ });
52
+ const result = obj ? obj.value : PDF_NULL;
53
+ this.cache.set(value.id, result);
54
+ return result;
55
+ }
56
+ get(dict, key) {
57
+ return this.resolve(dict.get(key) ?? PDF_NULL);
58
+ }
59
+ get catalog() {
60
+ const root = this.resolve(this.trailer.get("Root") ?? PDF_NULL);
61
+ return root instanceof Map ? root : /* @__PURE__ */ new Map();
62
+ }
63
+ pages() {
64
+ const out = [];
65
+ const root = this.get(this.catalog, "Pages");
66
+ if (root instanceof Map) this.walkPageTree(root, {}, out, /* @__PURE__ */ new Set());
67
+ return out;
68
+ }
69
+ walkPageTree(node, inherited, out, seen) {
70
+ if (out.length >= MAX_PAGES || seen.has(node)) return;
71
+ seen.add(node);
72
+ const mediaBox = readRectangle(this.get(node, "MediaBox")) ?? inherited.mediaBox;
73
+ const resourcesVal = this.get(node, "Resources");
74
+ const resources = resourcesVal instanceof Map ? resourcesVal : inherited.resources;
75
+ const type = node.get("Type");
76
+ const kids = this.get(node, "Kids");
77
+ if (type instanceof PdfName && type.value === "Pages" && Array.isArray(kids)) {
78
+ for (const kid of kids) {
79
+ const kidNode = this.resolve(kid);
80
+ if (kidNode instanceof Map) this.walkPageTree(kidNode, {
81
+ mediaBox,
82
+ resources
83
+ }, out, seen);
84
+ if (out.length >= MAX_PAGES) break;
85
+ }
86
+ return;
87
+ }
88
+ out.push({
89
+ dict: node,
90
+ mediaBox: mediaBox ?? DEFAULT_MEDIA_BOX,
91
+ resources
92
+ });
93
+ }
94
+ pageContent(page) {
95
+ const contents = this.get(page.dict, "Contents");
96
+ const streams = [];
97
+ if (contents instanceof PdfStream) streams.push(contents);
98
+ else if (Array.isArray(contents)) for (const c of contents) {
99
+ const s = this.resolve(c);
100
+ if (s instanceof PdfStream) streams.push(s);
101
+ }
102
+ return concatWithSpaces(streams.map((s) => this.streamData(s)));
103
+ }
104
+ streamData(stream) {
105
+ let data = stream.data;
106
+ const filter = this.resolve(stream.dict.get("Filter") ?? PDF_NULL);
107
+ const filters = Array.isArray(filter) ? filter : [filter];
108
+ for (const f of filters) if (f instanceof PdfName && (f.value === "FlateDecode" || f.value === "Fl")) try {
109
+ data = unzlibSync(data);
110
+ } catch {}
111
+ return data;
112
+ }
113
+ };
114
+ function findStartXref(buf) {
115
+ const tail = lastIndexOfAscii(buf, "startxref");
116
+ if (tail < 0) return -1;
117
+ const tok = new Lexer(buf, tail + 9).nextToken();
118
+ return tok.kind === "num" ? tok.value : -1;
119
+ }
120
+ function readXrefChain(buf, offset) {
121
+ const xref = /* @__PURE__ */ new Map();
122
+ let trailer = /* @__PURE__ */ new Map();
123
+ const visited = /* @__PURE__ */ new Set();
124
+ let at = offset;
125
+ while (at !== void 0 && at >= 0 && at < buf.length && !visited.has(at)) {
126
+ visited.add(at);
127
+ const section = readXrefSection(buf, at);
128
+ if (!section) break;
129
+ for (const [id, off] of section.xref) if (!xref.has(id)) xref.set(id, off);
130
+ if (trailer.size === 0) trailer = section.trailer;
131
+ const prev = section.trailer.get("Prev");
132
+ at = typeof prev === "number" ? prev : void 0;
133
+ }
134
+ return {
135
+ xref,
136
+ trailer
137
+ };
138
+ }
139
+ function readXrefSection(buf, offset) {
140
+ const lexer = new Lexer(buf, offset);
141
+ const head = lexer.nextToken();
142
+ if (!(head.kind === "keyword" && head.value === "xref")) return void 0;
143
+ const xref = /* @__PURE__ */ new Map();
144
+ for (;;) {
145
+ const tok = lexer.nextToken();
146
+ if (tok.kind === "keyword" && tok.value === "trailer") break;
147
+ if (tok.kind !== "num") return void 0;
148
+ const first = tok.value;
149
+ const countTok = lexer.nextToken();
150
+ if (countTok.kind !== "num") return void 0;
151
+ const count = countTok.value;
152
+ for (let i = 0; i < count; i++) {
153
+ const off = lexer.nextToken();
154
+ const gen = lexer.nextToken();
155
+ const type = lexer.nextToken();
156
+ if (off.kind !== "num" || gen.kind !== "num" || type.kind !== "keyword") return void 0;
157
+ if (type.value === "n" && !xref.has(first + i)) xref.set(first + i, off.value);
158
+ }
159
+ }
160
+ const trailerVal = parseObject(lexer);
161
+ return {
162
+ xref,
163
+ trailer: trailerVal instanceof Map ? trailerVal : /* @__PURE__ */ new Map()
164
+ };
165
+ }
166
+ function bruteForceScan(buf) {
167
+ const xref = /* @__PURE__ */ new Map();
168
+ let root;
169
+ const lexer = new Lexer(buf);
170
+ let prev2;
171
+ let prev1;
172
+ for (;;) {
173
+ lexer.skipWhitespace();
174
+ const start = lexer.pos;
175
+ const tok = lexer.nextToken();
176
+ if (tok.kind === "eof") break;
177
+ if (tok.kind === "keyword" && tok.value === "obj" && prev2 && prev1) {
178
+ xref.set(prev2.value, prev2.start);
179
+ if (root === void 0) {
180
+ const dict = parseIndirectObject(new Lexer(buf, prev2.start))?.value;
181
+ if (dict instanceof Map) {
182
+ const type = dict.get("Type");
183
+ if (type instanceof PdfName && type.value === "Catalog") root = new PdfRef(prev2.value, prev1.value);
184
+ }
185
+ }
186
+ }
187
+ prev2 = prev1;
188
+ prev1 = tok.kind === "num" ? {
189
+ start,
190
+ value: tok.value
191
+ } : void 0;
192
+ }
193
+ return {
194
+ xref,
195
+ root
196
+ };
197
+ }
198
+ function readRectangle(value) {
199
+ if (!Array.isArray(value) || value.length < 4) return void 0;
200
+ const nums = value.slice(0, 4).map((v) => typeof v === "number" ? v : NaN);
201
+ if (nums.some((n) => !Number.isFinite(n))) return void 0;
202
+ return [
203
+ nums[0],
204
+ nums[1],
205
+ nums[2],
206
+ nums[3]
207
+ ];
208
+ }
209
+ function concatWithSpaces(parts) {
210
+ if (parts.length === 1) return parts[0];
211
+ const total = parts.reduce((n, p) => n + p.length + 1, 0);
212
+ const out = new Uint8Array(Math.max(0, total - 1));
213
+ let at = 0;
214
+ parts.forEach((p, i) => {
215
+ out.set(p, at);
216
+ at += p.length;
217
+ if (i < parts.length - 1) out[at++] = 10;
218
+ });
219
+ return out;
220
+ }
221
+ function lastIndexOfAscii(buf, needle) {
222
+ const n = needle.length;
223
+ outer: for (let i = buf.length - n; i >= 0; i--) {
224
+ for (let j = 0; j < n; j++) if (buf[i + j] !== needle.charCodeAt(j)) continue outer;
225
+ return i;
226
+ }
227
+ return -1;
228
+ }
229
+ //#endregion
230
+ export { PdfFile };
@@ -0,0 +1,4 @@
1
+ import { BodyElement } from '../core/document-model/index.js';
2
+ import { FlowDoc } from '../core/ir/flow.js';
3
+ export declare function paragraphBlock(text: string, outlineLevel?: number): BodyElement;
4
+ export declare function buildFlowDoc(body: ReadonlyArray<BodyElement>): FlowDoc;
@@ -0,0 +1,27 @@
1
+ import { ResourceStore } from "../core/ir/resources.js";
2
+ import { EMPTY_STYLE_SHEET, resolveBodyStyles } from "../core/style-cascade/resolver.js";
3
+ import "../core/style-cascade/index.js";
4
+ //#region src/pdf-reader/flow-build.ts
5
+ function paragraphBlock(text, outlineLevel) {
6
+ return {
7
+ kind: "paragraph",
8
+ paragraph: {
9
+ properties: outlineLevel !== void 0 ? { outlineLevel } : {},
10
+ runs: text.length > 0 ? [{
11
+ text,
12
+ properties: {}
13
+ }] : []
14
+ }
15
+ };
16
+ }
17
+ function buildFlowDoc(body) {
18
+ return {
19
+ kind: "flow",
20
+ body: resolveBodyStyles([...body], EMPTY_STYLE_SHEET),
21
+ sections: [],
22
+ styles: EMPTY_STYLE_SHEET,
23
+ resources: new ResourceStore()
24
+ };
25
+ }
26
+ //#endregion
27
+ export { buildFlowDoc, paragraphBlock };
@@ -0,0 +1,4 @@
1
+ import { PdfDict } from '../pdf/objects.js';
2
+ import { ContentFont } from './content.js';
3
+ import { PdfFile } from './document.js';
4
+ export declare function buildContentFont(file: PdfFile, fontDict: PdfDict): ContentFont;