reamkit 1.6.0 → 1.7.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/esm/core/drawingml/chart-serializer.d.ts +2 -0
- package/dist/esm/core/drawingml/chart-serializer.js +53 -0
- package/dist/esm/excel/print-model.js +6 -3
- package/dist/esm/excel/xlsx-writer.js +81 -17
- package/dist/esm/pdf-reader/content.d.ts +39 -1
- package/dist/esm/pdf-reader/content.js +132 -4
- package/dist/esm/pdf-reader/crypto.d.ts +7 -0
- package/dist/esm/pdf-reader/crypto.js +609 -0
- package/dist/esm/pdf-reader/decrypt.d.ts +5 -0
- package/dist/esm/pdf-reader/decrypt.js +199 -0
- package/dist/esm/pdf-reader/document.d.ts +7 -0
- package/dist/esm/pdf-reader/document.js +209 -26
- package/dist/esm/pdf-reader/flow-build.d.ts +16 -1
- package/dist/esm/pdf-reader/flow-build.js +102 -3
- package/dist/esm/pdf-reader/image-decode.d.ts +15 -0
- package/dist/esm/pdf-reader/image-decode.js +442 -0
- package/dist/esm/pdf-reader/images.d.ts +16 -0
- package/dist/esm/pdf-reader/images.js +90 -0
- package/dist/esm/pdf-reader/layout.d.ts +2 -2
- package/dist/esm/pdf-reader/layout.js +48 -14
- package/dist/esm/pdf-reader/png-encode.d.ts +2 -0
- package/dist/esm/pdf-reader/png-encode.js +98 -0
- package/dist/esm/pdf-reader/predictor.d.ts +7 -0
- package/dist/esm/pdf-reader/predictor.js +61 -0
- package/dist/esm/pdf-reader/reader.js +11 -4
- package/dist/esm/pdf-reader/tagged.d.ts +2 -2
- package/dist/esm/pdf-reader/tagged.js +60 -7
- package/dist/esm/pdf-reader/text.js +48 -1
- package/dist/esm/pdf-reader/vector.d.ts +12 -0
- package/dist/esm/pdf-reader/vector.js +54 -0
- package/dist/esm/word/docx-writer.js +96 -7
- package/dist/esm/word/omml-serializer.d.ts +2 -0
- package/dist/esm/word/omml-serializer.js +48 -0
- package/package.json +1 -1
|
@@ -0,0 +1,199 @@
|
|
|
1
|
+
import { PdfHexString, PdfName, PdfStream } from "../pdf/objects.js";
|
|
2
|
+
import { aesCbcDecrypt, aesCbcEncrypt, md5, rc4, sha256, sha384, sha512 } from "./crypto.js";
|
|
3
|
+
//#region src/pdf-reader/decrypt.ts
|
|
4
|
+
var PAD = Uint8Array.from([
|
|
5
|
+
40,
|
|
6
|
+
191,
|
|
7
|
+
78,
|
|
8
|
+
94,
|
|
9
|
+
78,
|
|
10
|
+
117,
|
|
11
|
+
138,
|
|
12
|
+
65,
|
|
13
|
+
100,
|
|
14
|
+
0,
|
|
15
|
+
78,
|
|
16
|
+
86,
|
|
17
|
+
255,
|
|
18
|
+
250,
|
|
19
|
+
1,
|
|
20
|
+
8,
|
|
21
|
+
46,
|
|
22
|
+
46,
|
|
23
|
+
0,
|
|
24
|
+
182,
|
|
25
|
+
208,
|
|
26
|
+
104,
|
|
27
|
+
62,
|
|
28
|
+
128,
|
|
29
|
+
47,
|
|
30
|
+
12,
|
|
31
|
+
169,
|
|
32
|
+
254,
|
|
33
|
+
100,
|
|
34
|
+
83,
|
|
35
|
+
105,
|
|
36
|
+
122
|
|
37
|
+
]);
|
|
38
|
+
var AES_SALT = Uint8Array.from([
|
|
39
|
+
115,
|
|
40
|
+
65,
|
|
41
|
+
108,
|
|
42
|
+
84
|
|
43
|
+
]);
|
|
44
|
+
var ZERO16 = new Uint8Array(16);
|
|
45
|
+
var EMPTY = new Uint8Array(0);
|
|
46
|
+
function buildDecryptor(encrypt, idArray) {
|
|
47
|
+
const filter = encrypt.get("Filter");
|
|
48
|
+
if (!(filter instanceof PdfName) || filter.value !== "Standard") return void 0;
|
|
49
|
+
const v = numOf(encrypt.get("V"));
|
|
50
|
+
const r = numOf(encrypt.get("R"));
|
|
51
|
+
let fileKey;
|
|
52
|
+
let method;
|
|
53
|
+
if (r >= 5 || v >= 5) {
|
|
54
|
+
fileKey = deriveKeyR6(strBytes(encrypt.get("U")), strBytes(encrypt.get("UE")));
|
|
55
|
+
method = "aesv3";
|
|
56
|
+
} else {
|
|
57
|
+
const keyLen = (numOf(encrypt.get("Length")) || 40) / 8;
|
|
58
|
+
const id0 = Array.isArray(idArray) ? strBytes(idArray[0]) : EMPTY;
|
|
59
|
+
fileKey = deriveKeyLegacy(strBytes(encrypt.get("O")), numOf(encrypt.get("P")) | 0, id0, r, Math.max(5, Math.min(16, keyLen)), encrypt.get("EncryptMetadata") !== false);
|
|
60
|
+
const m = cipherMethod(encrypt, v);
|
|
61
|
+
if (m === void 0) return void 0;
|
|
62
|
+
method = m;
|
|
63
|
+
}
|
|
64
|
+
if (!fileKey) return void 0;
|
|
65
|
+
const key = fileKey;
|
|
66
|
+
return { decrypt: (value, objNum, gen) => decryptValue(value, objNum, gen, key, method) };
|
|
67
|
+
}
|
|
68
|
+
function deriveKeyLegacy(o, p, id0, r, keyLen, encryptMetadata) {
|
|
69
|
+
const tail = r >= 4 && !encryptMetadata ? Uint8Array.from([
|
|
70
|
+
255,
|
|
71
|
+
255,
|
|
72
|
+
255,
|
|
73
|
+
255
|
|
74
|
+
]) : EMPTY;
|
|
75
|
+
let hash = md5(concat(PAD, o.subarray(0, 32), p32le(p), id0, tail));
|
|
76
|
+
if (r >= 3) for (let i = 0; i < 50; i++) hash = md5(hash.subarray(0, keyLen));
|
|
77
|
+
return hash.subarray(0, keyLen);
|
|
78
|
+
}
|
|
79
|
+
function deriveKeyR6(u, ue) {
|
|
80
|
+
if (u.length < 48 || ue.length < 32) return void 0;
|
|
81
|
+
const validationSalt = u.subarray(32, 40);
|
|
82
|
+
const keySalt = u.subarray(40, 48);
|
|
83
|
+
if (!equal(hash2B(EMPTY, validationSalt, EMPTY), u.subarray(0, 32))) return void 0;
|
|
84
|
+
return aesCbcDecrypt(hash2B(EMPTY, keySalt, EMPTY), ZERO16, ue.subarray(0, 32), false);
|
|
85
|
+
}
|
|
86
|
+
function hash2B(pw, salt, udata) {
|
|
87
|
+
let k = sha256(concat(pw, salt, udata));
|
|
88
|
+
for (let round = 1; round <= 256; round++) {
|
|
89
|
+
const block = concat(pw, k, udata);
|
|
90
|
+
const k1 = new Uint8Array(block.length * 64);
|
|
91
|
+
for (let i = 0; i < 64; i++) k1.set(block, i * block.length);
|
|
92
|
+
const e = aesCbcEncrypt(k.subarray(0, 16), k.subarray(16, 32), k1);
|
|
93
|
+
let sum = 0;
|
|
94
|
+
for (let i = 0; i < 16; i++) sum += e[i];
|
|
95
|
+
const mod = sum % 3;
|
|
96
|
+
k = mod === 0 ? sha256(e) : mod === 1 ? sha384(e) : sha512(e);
|
|
97
|
+
if (round >= 64 && e[e.length - 1] <= round - 32) break;
|
|
98
|
+
}
|
|
99
|
+
return k.subarray(0, 32);
|
|
100
|
+
}
|
|
101
|
+
function cipherMethod(encrypt, v) {
|
|
102
|
+
if (v < 4) return "rc4";
|
|
103
|
+
const stmf = encrypt.get("StmF");
|
|
104
|
+
const cfName = stmf instanceof PdfName ? stmf.value : "StdCF";
|
|
105
|
+
if (cfName === "Identity") return void 0;
|
|
106
|
+
const cf = encrypt.get("CF");
|
|
107
|
+
const cfDict = cf instanceof Map ? cf.get(cfName) : void 0;
|
|
108
|
+
const cfm = cfDict instanceof Map ? cfDict.get("CFM") : void 0;
|
|
109
|
+
if (cfm instanceof PdfName) {
|
|
110
|
+
if (cfm.value === "AESV2") return "aesv2";
|
|
111
|
+
if (cfm.value === "AESV3") return "aesv3";
|
|
112
|
+
}
|
|
113
|
+
return "rc4";
|
|
114
|
+
}
|
|
115
|
+
function objectKey(fileKey, objNum, gen, aes) {
|
|
116
|
+
const extra = aes ? AES_SALT : EMPTY;
|
|
117
|
+
const seed = new Uint8Array(fileKey.length + 5 + extra.length);
|
|
118
|
+
seed.set(fileKey, 0);
|
|
119
|
+
seed[fileKey.length] = objNum & 255;
|
|
120
|
+
seed[fileKey.length + 1] = objNum >> 8 & 255;
|
|
121
|
+
seed[fileKey.length + 2] = objNum >> 16 & 255;
|
|
122
|
+
seed[fileKey.length + 3] = gen & 255;
|
|
123
|
+
seed[fileKey.length + 4] = gen >> 8 & 255;
|
|
124
|
+
seed.set(extra, fileKey.length + 5);
|
|
125
|
+
return md5(seed).subarray(0, Math.min(fileKey.length + 5, 16));
|
|
126
|
+
}
|
|
127
|
+
function decryptBytes(data, objNum, gen, fileKey, method) {
|
|
128
|
+
if (method === "aesv3") {
|
|
129
|
+
if (data.length < 16) return data;
|
|
130
|
+
return aesCbcDecrypt(fileKey, data.subarray(0, 16), data.subarray(16), true);
|
|
131
|
+
}
|
|
132
|
+
const key = objectKey(fileKey, objNum, gen, method === "aesv2");
|
|
133
|
+
if (method === "aesv2") {
|
|
134
|
+
if (data.length < 16) return data;
|
|
135
|
+
return aesCbcDecrypt(key, data.subarray(0, 16), data.subarray(16), true);
|
|
136
|
+
}
|
|
137
|
+
return rc4(key, data);
|
|
138
|
+
}
|
|
139
|
+
function decryptValue(value, objNum, gen, fileKey, method) {
|
|
140
|
+
if (typeof value === "string") return latin1(decryptBytes(strToBytes(value), objNum, gen, fileKey, method));
|
|
141
|
+
if (value instanceof PdfHexString) return new PdfHexString(decryptBytes(value.bytes, objNum, gen, fileKey, method));
|
|
142
|
+
if (value instanceof PdfStream) {
|
|
143
|
+
const dict = decryptValue(value.dict, objNum, gen, fileKey, method);
|
|
144
|
+
const data = decryptBytes(value.data, objNum, gen, fileKey, method);
|
|
145
|
+
return new PdfStream(dict instanceof Map ? dict : value.dict, data);
|
|
146
|
+
}
|
|
147
|
+
if (Array.isArray(value)) return value.map((v) => decryptValue(v, objNum, gen, fileKey, method));
|
|
148
|
+
if (value instanceof Map) {
|
|
149
|
+
const out = /* @__PURE__ */ new Map();
|
|
150
|
+
for (const [k, v] of value) out.set(k, decryptValue(v, objNum, gen, fileKey, method));
|
|
151
|
+
return out;
|
|
152
|
+
}
|
|
153
|
+
return value;
|
|
154
|
+
}
|
|
155
|
+
function numOf(v) {
|
|
156
|
+
return typeof v === "number" ? v : 0;
|
|
157
|
+
}
|
|
158
|
+
function strBytes(v) {
|
|
159
|
+
if (v instanceof PdfHexString) return v.bytes;
|
|
160
|
+
if (typeof v === "string") return strToBytes(v);
|
|
161
|
+
return EMPTY;
|
|
162
|
+
}
|
|
163
|
+
function strToBytes(s) {
|
|
164
|
+
const out = new Uint8Array(s.length);
|
|
165
|
+
for (let i = 0; i < s.length; i++) out[i] = s.charCodeAt(i) & 255;
|
|
166
|
+
return out;
|
|
167
|
+
}
|
|
168
|
+
function latin1(b) {
|
|
169
|
+
let s = "";
|
|
170
|
+
for (const x of b) s += String.fromCharCode(x);
|
|
171
|
+
return s;
|
|
172
|
+
}
|
|
173
|
+
function p32le(p) {
|
|
174
|
+
const u = p >>> 0;
|
|
175
|
+
return Uint8Array.from([
|
|
176
|
+
u & 255,
|
|
177
|
+
u >> 8 & 255,
|
|
178
|
+
u >> 16 & 255,
|
|
179
|
+
u >> 24 & 255
|
|
180
|
+
]);
|
|
181
|
+
}
|
|
182
|
+
function concat(...parts) {
|
|
183
|
+
let total = 0;
|
|
184
|
+
for (const p of parts) total += p.length;
|
|
185
|
+
const out = new Uint8Array(total);
|
|
186
|
+
let off = 0;
|
|
187
|
+
for (const p of parts) {
|
|
188
|
+
out.set(p, off);
|
|
189
|
+
off += p.length;
|
|
190
|
+
}
|
|
191
|
+
return out;
|
|
192
|
+
}
|
|
193
|
+
function equal(a, b) {
|
|
194
|
+
if (a.length !== b.length) return false;
|
|
195
|
+
for (let i = 0; i < a.length; i++) if (a[i] !== b[i]) return false;
|
|
196
|
+
return true;
|
|
197
|
+
}
|
|
198
|
+
//#endregion
|
|
199
|
+
export { buildDecryptor };
|
|
@@ -10,11 +10,18 @@ export declare class PdfFile {
|
|
|
10
10
|
private readonly xref;
|
|
11
11
|
readonly trailer: PdfDict;
|
|
12
12
|
private readonly cache;
|
|
13
|
+
private readonly objStmCache;
|
|
14
|
+
private decryptor;
|
|
15
|
+
private encryptObjNum;
|
|
13
16
|
private constructor();
|
|
14
17
|
static parse(bytes: Uint8Array): PdfFile;
|
|
18
|
+
private initEncryption;
|
|
19
|
+
private readonly lengthResolver;
|
|
15
20
|
resolve(value: PdfValue): PdfValue;
|
|
21
|
+
private objectFromStream;
|
|
16
22
|
get(dict: PdfDict, key: string): PdfValue;
|
|
17
23
|
get catalog(): PdfDict;
|
|
24
|
+
get encryptionUnsupported(): boolean;
|
|
18
25
|
pages(): Array<PdfPage>;
|
|
19
26
|
private walkPageTree;
|
|
20
27
|
pageContent(page: PdfPage): Uint8Array;
|
|
@@ -1,6 +1,8 @@
|
|
|
1
1
|
import { PDF_NULL, PdfName, PdfRef, PdfStream } from "../pdf/objects.js";
|
|
2
|
+
import { buildDecryptor } from "./decrypt.js";
|
|
2
3
|
import { Lexer } from "./lexer.js";
|
|
3
4
|
import { parseIndirectObject, parseObject } from "./parser.js";
|
|
5
|
+
import { reversePredictor } from "./predictor.js";
|
|
4
6
|
import { unzlibSync } from "fflate";
|
|
5
7
|
//#region src/pdf-reader/document.ts
|
|
6
8
|
var DEFAULT_MEDIA_BOX = [
|
|
@@ -10,8 +12,12 @@ var DEFAULT_MEDIA_BOX = [
|
|
|
10
12
|
792
|
|
11
13
|
];
|
|
12
14
|
var MAX_PAGES = 5e4;
|
|
15
|
+
var MAX_OBJSTM_N = 2e5;
|
|
13
16
|
var PdfFile = class PdfFile {
|
|
14
17
|
cache = /* @__PURE__ */ new Map();
|
|
18
|
+
objStmCache = /* @__PURE__ */ new Map();
|
|
19
|
+
decryptor;
|
|
20
|
+
encryptObjNum = -1;
|
|
15
21
|
constructor(buf, xref, trailer) {
|
|
16
22
|
this.buf = buf;
|
|
17
23
|
this.xref = xref;
|
|
@@ -30,29 +36,66 @@ var PdfFile = class PdfFile {
|
|
|
30
36
|
} catch {}
|
|
31
37
|
if (xref.size === 0 || !(trailer.get("Root") instanceof PdfRef)) {
|
|
32
38
|
const scanned = bruteForceScan(bytes);
|
|
33
|
-
for (const [id,
|
|
39
|
+
for (const [id, entry] of scanned.xref) if (!xref.has(id)) xref.set(id, entry);
|
|
34
40
|
if (!(trailer.get("Root") instanceof PdfRef) && scanned.root) {
|
|
35
41
|
trailer = new Map(trailer);
|
|
36
42
|
trailer.set("Root", scanned.root);
|
|
37
43
|
}
|
|
38
44
|
}
|
|
39
|
-
|
|
45
|
+
const file = new PdfFile(bytes, xref, trailer);
|
|
46
|
+
file.initEncryption();
|
|
47
|
+
return file;
|
|
40
48
|
}
|
|
49
|
+
initEncryption() {
|
|
50
|
+
const encVal = this.trailer.get("Encrypt");
|
|
51
|
+
if (encVal === void 0) return;
|
|
52
|
+
if (encVal instanceof PdfRef) this.encryptObjNum = encVal.id;
|
|
53
|
+
const enc = this.resolve(encVal);
|
|
54
|
+
if (!(enc instanceof Map)) return;
|
|
55
|
+
const id = this.trailer.get("ID");
|
|
56
|
+
this.decryptor = buildDecryptor(enc, Array.isArray(id) ? id : void 0);
|
|
57
|
+
}
|
|
58
|
+
lengthResolver = (r) => {
|
|
59
|
+
const n = this.resolve(r);
|
|
60
|
+
return typeof n === "number" ? n : void 0;
|
|
61
|
+
};
|
|
41
62
|
resolve(value) {
|
|
42
63
|
if (!(value instanceof PdfRef)) return value;
|
|
43
64
|
const cached = this.cache.get(value.id);
|
|
44
65
|
if (cached !== void 0) return cached;
|
|
45
|
-
const
|
|
46
|
-
if (
|
|
66
|
+
const entry = this.xref.get(value.id);
|
|
67
|
+
if (entry === void 0) return PDF_NULL;
|
|
47
68
|
this.cache.set(value.id, PDF_NULL);
|
|
48
|
-
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
|
|
69
|
+
let result = PDF_NULL;
|
|
70
|
+
if (entry.kind === "uncompressed") {
|
|
71
|
+
if (entry.offset >= 0 && entry.offset < this.buf.length) {
|
|
72
|
+
const obj = parseIndirectObject(new Lexer(this.buf, entry.offset), this.lengthResolver);
|
|
73
|
+
result = obj ? obj.value : PDF_NULL;
|
|
74
|
+
if (this.decryptor && value.id !== this.encryptObjNum) result = this.decryptor.decrypt(result, value.id, obj?.generation ?? 0);
|
|
75
|
+
}
|
|
76
|
+
} else result = this.objectFromStream(entry.streamObj).get(value.id) ?? PDF_NULL;
|
|
53
77
|
this.cache.set(value.id, result);
|
|
54
78
|
return result;
|
|
55
79
|
}
|
|
80
|
+
objectFromStream(streamObj) {
|
|
81
|
+
const cached = this.objStmCache.get(streamObj);
|
|
82
|
+
if (cached) return cached;
|
|
83
|
+
const out = /* @__PURE__ */ new Map();
|
|
84
|
+
this.objStmCache.set(streamObj, out);
|
|
85
|
+
const entry = this.xref.get(streamObj);
|
|
86
|
+
if (!entry || entry.kind !== "uncompressed") return out;
|
|
87
|
+
const obj = parseIndirectObject(new Lexer(this.buf, entry.offset), this.lengthResolver);
|
|
88
|
+
if (!obj || !(obj.value instanceof PdfStream)) return out;
|
|
89
|
+
let stream = obj.value;
|
|
90
|
+
if (this.decryptor) {
|
|
91
|
+
const dec = this.decryptor.decrypt(stream, streamObj, obj.generation);
|
|
92
|
+
if (dec instanceof PdfStream) stream = dec;
|
|
93
|
+
}
|
|
94
|
+
const data = inflateStream(stream);
|
|
95
|
+
const first = numOf(this.resolve(stream.dict.get("First") ?? PDF_NULL));
|
|
96
|
+
for (const member of objStmHeader(data, numOf(this.resolve(stream.dict.get("N") ?? PDF_NULL)))) out.set(member.id, parseObject(new Lexer(data, first + member.off), this.lengthResolver));
|
|
97
|
+
return out;
|
|
98
|
+
}
|
|
56
99
|
get(dict, key) {
|
|
57
100
|
return this.resolve(dict.get(key) ?? PDF_NULL);
|
|
58
101
|
}
|
|
@@ -60,6 +103,9 @@ var PdfFile = class PdfFile {
|
|
|
60
103
|
const root = this.resolve(this.trailer.get("Root") ?? PDF_NULL);
|
|
61
104
|
return root instanceof Map ? root : /* @__PURE__ */ new Map();
|
|
62
105
|
}
|
|
106
|
+
get encryptionUnsupported() {
|
|
107
|
+
return this.trailer.get("Encrypt") !== void 0 && this.decryptor === void 0;
|
|
108
|
+
}
|
|
63
109
|
pages() {
|
|
64
110
|
const out = [];
|
|
65
111
|
const root = this.get(this.catalog, "Pages");
|
|
@@ -105,10 +151,12 @@ var PdfFile = class PdfFile {
|
|
|
105
151
|
let data = stream.data;
|
|
106
152
|
const filter = this.resolve(stream.dict.get("Filter") ?? PDF_NULL);
|
|
107
153
|
const filters = Array.isArray(filter) ? filter : [filter];
|
|
154
|
+
let flate = false;
|
|
108
155
|
for (const f of filters) if (f instanceof PdfName && (f.value === "FlateDecode" || f.value === "Fl")) try {
|
|
109
156
|
data = unzlibSync(data);
|
|
157
|
+
flate = true;
|
|
110
158
|
} catch {}
|
|
111
|
-
return data;
|
|
159
|
+
return flate ? applyStreamPredictor(this, stream.dict, data) : data;
|
|
112
160
|
}
|
|
113
161
|
};
|
|
114
162
|
function findStartXref(buf) {
|
|
@@ -121,25 +169,33 @@ function readXrefChain(buf, offset) {
|
|
|
121
169
|
const xref = /* @__PURE__ */ new Map();
|
|
122
170
|
let trailer = /* @__PURE__ */ new Map();
|
|
123
171
|
const visited = /* @__PURE__ */ new Set();
|
|
124
|
-
|
|
125
|
-
while (
|
|
172
|
+
const queue = [offset];
|
|
173
|
+
while (queue.length > 0) {
|
|
174
|
+
const at = queue.shift();
|
|
175
|
+
if (at < 0 || at >= buf.length || visited.has(at)) continue;
|
|
126
176
|
visited.add(at);
|
|
127
|
-
const section =
|
|
128
|
-
if (!section)
|
|
129
|
-
for (const [id,
|
|
177
|
+
const section = readXrefAt(buf, at);
|
|
178
|
+
if (!section) continue;
|
|
179
|
+
for (const [id, entry] of section.xref) if (!xref.has(id)) xref.set(id, entry);
|
|
130
180
|
if (trailer.size === 0) trailer = section.trailer;
|
|
181
|
+
const xrefStm = section.trailer.get("XRefStm");
|
|
182
|
+
if (typeof xrefStm === "number") queue.push(xrefStm);
|
|
131
183
|
const prev = section.trailer.get("Prev");
|
|
132
|
-
|
|
184
|
+
if (typeof prev === "number") queue.push(prev);
|
|
133
185
|
}
|
|
134
186
|
return {
|
|
135
187
|
xref,
|
|
136
188
|
trailer
|
|
137
189
|
};
|
|
138
190
|
}
|
|
139
|
-
function
|
|
191
|
+
function readXrefAt(buf, offset) {
|
|
140
192
|
const lexer = new Lexer(buf, offset);
|
|
141
193
|
const head = lexer.nextToken();
|
|
142
|
-
if (
|
|
194
|
+
if (head.kind === "keyword" && head.value === "xref") return readClassicXref(lexer);
|
|
195
|
+
const obj = parseIndirectObject(new Lexer(buf, offset));
|
|
196
|
+
if (obj && obj.value instanceof PdfStream) return readXrefStream(obj.value);
|
|
197
|
+
}
|
|
198
|
+
function readClassicXref(lexer) {
|
|
143
199
|
const xref = /* @__PURE__ */ new Map();
|
|
144
200
|
for (;;) {
|
|
145
201
|
const tok = lexer.nextToken();
|
|
@@ -154,7 +210,10 @@ function readXrefSection(buf, offset) {
|
|
|
154
210
|
const gen = lexer.nextToken();
|
|
155
211
|
const type = lexer.nextToken();
|
|
156
212
|
if (off.kind !== "num" || gen.kind !== "num" || type.kind !== "keyword") return void 0;
|
|
157
|
-
if (type.value === "n" && !xref.has(first + i)) xref.set(first + i,
|
|
213
|
+
if (type.value === "n" && !xref.has(first + i)) xref.set(first + i, {
|
|
214
|
+
kind: "uncompressed",
|
|
215
|
+
offset: off.value
|
|
216
|
+
});
|
|
158
217
|
}
|
|
159
218
|
}
|
|
160
219
|
const trailerVal = parseObject(lexer);
|
|
@@ -163,8 +222,99 @@ function readXrefSection(buf, offset) {
|
|
|
163
222
|
trailer: trailerVal instanceof Map ? trailerVal : /* @__PURE__ */ new Map()
|
|
164
223
|
};
|
|
165
224
|
}
|
|
225
|
+
function readXrefStream(stream) {
|
|
226
|
+
const dict = stream.dict;
|
|
227
|
+
const wv = dict.get("W");
|
|
228
|
+
if (!Array.isArray(wv) || wv.length < 3) return void 0;
|
|
229
|
+
const w0 = numOf(wv[0]);
|
|
230
|
+
const w1 = numOf(wv[1]);
|
|
231
|
+
const w2 = numOf(wv[2]);
|
|
232
|
+
const rowLen = w0 + w1 + w2;
|
|
233
|
+
if (rowLen <= 0) return void 0;
|
|
234
|
+
const data = inflateStream(stream);
|
|
235
|
+
const size = numOf(dict.get("Size"));
|
|
236
|
+
const indexV = dict.get("Index");
|
|
237
|
+
const index = Array.isArray(indexV) ? indexV.map(numOf) : [0, size];
|
|
238
|
+
const xref = /* @__PURE__ */ new Map();
|
|
239
|
+
let pos = 0;
|
|
240
|
+
for (let s = 0; s + 1 < index.length; s += 2) {
|
|
241
|
+
const start = index[s];
|
|
242
|
+
const count = index[s + 1];
|
|
243
|
+
for (let i = 0; i < count && pos + rowLen <= data.length; i++) {
|
|
244
|
+
const type = w0 === 0 ? 1 : readBE(data, pos, w0);
|
|
245
|
+
const f2 = readBE(data, pos + w0, w1);
|
|
246
|
+
const f3 = readBE(data, pos + w0 + w1, w2);
|
|
247
|
+
pos += rowLen;
|
|
248
|
+
const id = start + i;
|
|
249
|
+
if (xref.has(id)) continue;
|
|
250
|
+
if (type === 1) xref.set(id, {
|
|
251
|
+
kind: "uncompressed",
|
|
252
|
+
offset: f2
|
|
253
|
+
});
|
|
254
|
+
else if (type === 2) xref.set(id, {
|
|
255
|
+
kind: "compressed",
|
|
256
|
+
streamObj: f2,
|
|
257
|
+
index: f3
|
|
258
|
+
});
|
|
259
|
+
}
|
|
260
|
+
}
|
|
261
|
+
return {
|
|
262
|
+
xref,
|
|
263
|
+
trailer: dict
|
|
264
|
+
};
|
|
265
|
+
}
|
|
266
|
+
function objStmHeader(data, n) {
|
|
267
|
+
const out = [];
|
|
268
|
+
const lexer = new Lexer(data, 0);
|
|
269
|
+
for (let i = 0; i < Math.min(n, MAX_OBJSTM_N); i++) {
|
|
270
|
+
const idTok = lexer.nextToken();
|
|
271
|
+
const offTok = lexer.nextToken();
|
|
272
|
+
if (idTok.kind !== "num" || offTok.kind !== "num") break;
|
|
273
|
+
out.push({
|
|
274
|
+
id: idTok.value,
|
|
275
|
+
off: offTok.value
|
|
276
|
+
});
|
|
277
|
+
}
|
|
278
|
+
return out;
|
|
279
|
+
}
|
|
280
|
+
function inflateStream(stream) {
|
|
281
|
+
let data = stream.data;
|
|
282
|
+
const filter = stream.dict.get("Filter") ?? PDF_NULL;
|
|
283
|
+
const filters = Array.isArray(filter) ? filter : [filter];
|
|
284
|
+
for (const f of filters) if (f instanceof PdfName && (f.value === "FlateDecode" || f.value === "Fl")) try {
|
|
285
|
+
data = unzlibSync(data);
|
|
286
|
+
} catch {
|
|
287
|
+
return new Uint8Array(0);
|
|
288
|
+
}
|
|
289
|
+
const parmsVal = stream.dict.get("DecodeParms") ?? stream.dict.get("DP");
|
|
290
|
+
const parms = parmsVal instanceof Map ? parmsVal : Array.isArray(parmsVal) ? parmsVal.find((p) => p instanceof Map) : void 0;
|
|
291
|
+
if (parms) {
|
|
292
|
+
const predictor = numOf(parms.get("Predictor"));
|
|
293
|
+
if (predictor >= 2) data = reversePredictor(data, {
|
|
294
|
+
predictor,
|
|
295
|
+
colors: numOf(parms.get("Colors")) || 1,
|
|
296
|
+
bitsPerComponent: numOf(parms.get("BitsPerComponent")) || 8,
|
|
297
|
+
columns: numOf(parms.get("Columns")) || 1
|
|
298
|
+
});
|
|
299
|
+
}
|
|
300
|
+
return data;
|
|
301
|
+
}
|
|
302
|
+
function applyStreamPredictor(file, dict, data) {
|
|
303
|
+
const parmsVal = file.get(dict, "DecodeParms");
|
|
304
|
+
const parms = parmsVal instanceof Map ? parmsVal : void 0;
|
|
305
|
+
if (!parms) return data;
|
|
306
|
+
const predictor = numOf(file.get(parms, "Predictor"));
|
|
307
|
+
if (predictor < 2) return data;
|
|
308
|
+
return reversePredictor(data, {
|
|
309
|
+
predictor,
|
|
310
|
+
colors: numOf(file.get(parms, "Colors")) || 1,
|
|
311
|
+
bitsPerComponent: numOf(file.get(parms, "BitsPerComponent")) || 8,
|
|
312
|
+
columns: numOf(file.get(parms, "Columns")) || 1
|
|
313
|
+
});
|
|
314
|
+
}
|
|
166
315
|
function bruteForceScan(buf) {
|
|
167
316
|
const xref = /* @__PURE__ */ new Map();
|
|
317
|
+
const objStmObjs = [];
|
|
168
318
|
let root;
|
|
169
319
|
const lexer = new Lexer(buf);
|
|
170
320
|
let prev2;
|
|
@@ -175,13 +325,16 @@ function bruteForceScan(buf) {
|
|
|
175
325
|
const tok = lexer.nextToken();
|
|
176
326
|
if (tok.kind === "eof") break;
|
|
177
327
|
if (tok.kind === "keyword" && tok.value === "obj" && prev2 && prev1) {
|
|
178
|
-
xref.set(prev2.value,
|
|
179
|
-
|
|
180
|
-
|
|
181
|
-
|
|
182
|
-
|
|
183
|
-
|
|
184
|
-
|
|
328
|
+
xref.set(prev2.value, {
|
|
329
|
+
kind: "uncompressed",
|
|
330
|
+
offset: prev2.start
|
|
331
|
+
});
|
|
332
|
+
const value = parseIndirectObject(new Lexer(buf, prev2.start))?.value;
|
|
333
|
+
const dict = value instanceof PdfStream ? value.dict : value;
|
|
334
|
+
if (dict instanceof Map) {
|
|
335
|
+
const type = dict.get("Type");
|
|
336
|
+
if (type instanceof PdfName && type.value === "Catalog" && root === void 0) root = new PdfRef(prev2.value, prev1.value);
|
|
337
|
+
else if (type instanceof PdfName && type.value === "ObjStm") objStmObjs.push(prev2.value);
|
|
185
338
|
}
|
|
186
339
|
}
|
|
187
340
|
prev2 = prev1;
|
|
@@ -190,11 +343,41 @@ function bruteForceScan(buf) {
|
|
|
190
343
|
value: tok.value
|
|
191
344
|
} : void 0;
|
|
192
345
|
}
|
|
346
|
+
for (const streamObj of objStmObjs) {
|
|
347
|
+
const entry = xref.get(streamObj);
|
|
348
|
+
if (!entry || entry.kind !== "uncompressed") continue;
|
|
349
|
+
const obj = parseIndirectObject(new Lexer(buf, entry.offset));
|
|
350
|
+
if (!obj || !(obj.value instanceof PdfStream)) continue;
|
|
351
|
+
const data = inflateStream(obj.value);
|
|
352
|
+
const first = numOf(obj.value.dict.get("First") ?? PDF_NULL);
|
|
353
|
+
objStmHeader(data, numOf(obj.value.dict.get("N") ?? PDF_NULL)).forEach((member, index) => {
|
|
354
|
+
if (!xref.has(member.id)) xref.set(member.id, {
|
|
355
|
+
kind: "compressed",
|
|
356
|
+
streamObj,
|
|
357
|
+
index
|
|
358
|
+
});
|
|
359
|
+
if (root === void 0) {
|
|
360
|
+
const v = parseObject(new Lexer(data, first + member.off));
|
|
361
|
+
if (v instanceof Map) {
|
|
362
|
+
const t = v.get("Type");
|
|
363
|
+
if (t instanceof PdfName && t.value === "Catalog") root = new PdfRef(member.id, 0);
|
|
364
|
+
}
|
|
365
|
+
}
|
|
366
|
+
});
|
|
367
|
+
}
|
|
193
368
|
return {
|
|
194
369
|
xref,
|
|
195
370
|
root
|
|
196
371
|
};
|
|
197
372
|
}
|
|
373
|
+
function readBE(data, offset, width) {
|
|
374
|
+
let v = 0;
|
|
375
|
+
for (let i = 0; i < width; i++) v = v * 256 + (data[offset + i] ?? 0);
|
|
376
|
+
return v;
|
|
377
|
+
}
|
|
378
|
+
function numOf(v) {
|
|
379
|
+
return typeof v === "number" ? v : 0;
|
|
380
|
+
}
|
|
198
381
|
function readRectangle(value) {
|
|
199
382
|
if (!Array.isArray(value) || value.length < 4) return void 0;
|
|
200
383
|
const nums = value.slice(0, 4).map((v) => typeof v === "number" ? v : NaN);
|
|
@@ -1,4 +1,19 @@
|
|
|
1
1
|
import { BodyElement } from '../core/document-model/index.js';
|
|
2
2
|
import { FlowDoc } from '../core/ir/flow.js';
|
|
3
|
+
import { Loss, ResourceStore } from '../core/ir/index.js';
|
|
4
|
+
import { PdfImage } from './images.js';
|
|
5
|
+
import { PdfVector } from './vector.js';
|
|
6
|
+
export interface Reconstruction {
|
|
7
|
+
readonly doc: FlowDoc;
|
|
8
|
+
readonly losses: ReadonlyArray<Loss>;
|
|
9
|
+
}
|
|
3
10
|
export declare function paragraphBlock(text: string, outlineLevel?: number): BodyElement;
|
|
4
|
-
export
|
|
11
|
+
export interface TextSpan {
|
|
12
|
+
readonly text: string;
|
|
13
|
+
readonly href?: string;
|
|
14
|
+
}
|
|
15
|
+
export declare function paragraphFromRuns(spans: ReadonlyArray<TextSpan>, outlineLevel?: number): BodyElement;
|
|
16
|
+
export declare function imageBlock(image: PdfImage, resources: ResourceStore, alt?: string): BodyElement;
|
|
17
|
+
export declare function dedupeLosses(losses: ReadonlyArray<Loss>): Array<Loss>;
|
|
18
|
+
export declare function shapeBlock(v: PdfVector): BodyElement;
|
|
19
|
+
export declare function buildFlowDoc(body: ReadonlyArray<BodyElement>, resources?: ResourceStore): FlowDoc;
|
|
@@ -1,3 +1,4 @@
|
|
|
1
|
+
import { pt } from "../core/ir/units.js";
|
|
1
2
|
import { ResourceStore } from "../core/ir/resources.js";
|
|
2
3
|
import { EMPTY_STYLE_SHEET, resolveBodyStyles } from "../core/style-cascade/resolver.js";
|
|
3
4
|
import "../core/style-cascade/index.js";
|
|
@@ -14,14 +15,112 @@ function paragraphBlock(text, outlineLevel) {
|
|
|
14
15
|
}
|
|
15
16
|
};
|
|
16
17
|
}
|
|
17
|
-
function
|
|
18
|
+
function paragraphFromRuns(spans, outlineLevel) {
|
|
19
|
+
const merged = [];
|
|
20
|
+
for (const s of spans) {
|
|
21
|
+
const last = merged[merged.length - 1];
|
|
22
|
+
if (last && last.href === s.href) last.text += s.text;
|
|
23
|
+
else if (s.href !== void 0) merged.push({
|
|
24
|
+
text: s.text,
|
|
25
|
+
href: s.href
|
|
26
|
+
});
|
|
27
|
+
else merged.push({ text: s.text });
|
|
28
|
+
}
|
|
29
|
+
const runs = merged.map((m) => ({
|
|
30
|
+
text: m.text.replace(/\s+/g, " "),
|
|
31
|
+
href: m.href
|
|
32
|
+
})).filter((m) => m.text.length > 0);
|
|
33
|
+
if (runs.length > 0) {
|
|
34
|
+
runs[0].text = runs[0].text.replace(/^ /, "");
|
|
35
|
+
runs[runs.length - 1].text = runs[runs.length - 1].text.replace(/ $/, "");
|
|
36
|
+
}
|
|
37
|
+
return {
|
|
38
|
+
kind: "paragraph",
|
|
39
|
+
paragraph: {
|
|
40
|
+
properties: outlineLevel !== void 0 ? { outlineLevel } : {},
|
|
41
|
+
runs: runs.filter((r) => r.text.length > 0).map((r) => ({
|
|
42
|
+
text: r.text,
|
|
43
|
+
properties: {},
|
|
44
|
+
...r.href ? { href: r.href } : {}
|
|
45
|
+
}))
|
|
46
|
+
}
|
|
47
|
+
};
|
|
48
|
+
}
|
|
49
|
+
function imageBlock(image, resources, alt) {
|
|
50
|
+
return {
|
|
51
|
+
kind: "image",
|
|
52
|
+
image: {
|
|
53
|
+
resource: resources.put(image.bytes),
|
|
54
|
+
width: pt(image.widthPt),
|
|
55
|
+
height: pt(image.heightPt),
|
|
56
|
+
paragraphProperties: {},
|
|
57
|
+
...alt ? { altText: alt } : {}
|
|
58
|
+
}
|
|
59
|
+
};
|
|
60
|
+
}
|
|
61
|
+
function dedupeLosses(losses) {
|
|
62
|
+
const byDetail = /* @__PURE__ */ new Map();
|
|
63
|
+
for (const loss of losses) if (!byDetail.has(loss.detail)) byDetail.set(loss.detail, loss);
|
|
64
|
+
return [...byDetail.values()];
|
|
65
|
+
}
|
|
66
|
+
function shapeBlock(v) {
|
|
67
|
+
const w = v.maxX - v.minX;
|
|
68
|
+
const h = v.maxY - v.minY;
|
|
69
|
+
const fx = (x) => x - v.minX;
|
|
70
|
+
const fy = (y) => v.maxY - y;
|
|
71
|
+
const commands = v.segs.map((s) => {
|
|
72
|
+
switch (s.op) {
|
|
73
|
+
case "move": return {
|
|
74
|
+
cmd: "move",
|
|
75
|
+
x: fx(s.x),
|
|
76
|
+
y: fy(s.y)
|
|
77
|
+
};
|
|
78
|
+
case "line": return {
|
|
79
|
+
cmd: "line",
|
|
80
|
+
x: fx(s.x),
|
|
81
|
+
y: fy(s.y)
|
|
82
|
+
};
|
|
83
|
+
case "cubic": return {
|
|
84
|
+
cmd: "cubic",
|
|
85
|
+
x1: fx(s.x1),
|
|
86
|
+
y1: fy(s.y1),
|
|
87
|
+
x2: fx(s.x2),
|
|
88
|
+
y2: fy(s.y2),
|
|
89
|
+
x: fx(s.x),
|
|
90
|
+
y: fy(s.y)
|
|
91
|
+
};
|
|
92
|
+
case "close": return { cmd: "close" };
|
|
93
|
+
}
|
|
94
|
+
});
|
|
95
|
+
return {
|
|
96
|
+
kind: "shape",
|
|
97
|
+
shape: {
|
|
98
|
+
width: pt(w),
|
|
99
|
+
height: pt(h),
|
|
100
|
+
geometry: {
|
|
101
|
+
kind: "custom",
|
|
102
|
+
custom: {
|
|
103
|
+
pathWidth: w,
|
|
104
|
+
pathHeight: h,
|
|
105
|
+
commands
|
|
106
|
+
}
|
|
107
|
+
},
|
|
108
|
+
fill: {
|
|
109
|
+
kind: "solid",
|
|
110
|
+
colorHex: v.fillHex
|
|
111
|
+
},
|
|
112
|
+
paragraphProperties: {}
|
|
113
|
+
}
|
|
114
|
+
};
|
|
115
|
+
}
|
|
116
|
+
function buildFlowDoc(body, resources = new ResourceStore()) {
|
|
18
117
|
return {
|
|
19
118
|
kind: "flow",
|
|
20
119
|
body: resolveBodyStyles([...body], EMPTY_STYLE_SHEET),
|
|
21
120
|
sections: [],
|
|
22
121
|
styles: EMPTY_STYLE_SHEET,
|
|
23
|
-
resources
|
|
122
|
+
resources
|
|
24
123
|
};
|
|
25
124
|
}
|
|
26
125
|
//#endregion
|
|
27
|
-
export { buildFlowDoc, paragraphBlock };
|
|
126
|
+
export { buildFlowDoc, dedupeLosses, imageBlock, paragraphBlock, paragraphFromRuns, shapeBlock };
|