reamkit 1.5.0 → 1.7.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +2 -2
- package/dist/esm/core/converter/facade.js +6 -1
- package/dist/esm/core/converter/ream.js +2 -1
- package/dist/esm/core/document-model/types.d.ts +9 -1
- package/dist/esm/core/drawingml/chart-serializer.d.ts +2 -0
- package/dist/esm/core/drawingml/chart-serializer.js +53 -0
- package/dist/esm/core/spreadsheet-model/index.d.ts +1 -1
- package/dist/esm/core/spreadsheet-model/types.d.ts +5 -0
- package/dist/esm/excel/column-bands.d.ts +7 -0
- package/dist/esm/excel/column-bands.js +87 -0
- package/dist/esm/excel/conditional-format.js +22 -0
- package/dist/esm/excel/print-model.js +33 -11
- package/dist/esm/excel/worksheet-parser.js +26 -0
- package/dist/esm/excel/xlsx-writer.js +88 -17
- package/dist/esm/html/html-writer.js +100 -5
- package/dist/esm/layout/styled-layout.js +88 -19
- package/dist/esm/pdf-reader/cmap.d.ts +5 -0
- package/dist/esm/pdf-reader/cmap.js +72 -0
- package/dist/esm/pdf-reader/content.d.ts +52 -0
- package/dist/esm/pdf-reader/content.js +420 -0
- package/dist/esm/pdf-reader/crypto.d.ts +7 -0
- package/dist/esm/pdf-reader/crypto.js +609 -0
- package/dist/esm/pdf-reader/decrypt.d.ts +5 -0
- package/dist/esm/pdf-reader/decrypt.js +199 -0
- package/dist/esm/pdf-reader/document.d.ts +30 -0
- package/dist/esm/pdf-reader/document.js +413 -0
- package/dist/esm/pdf-reader/flow-build.d.ts +19 -0
- package/dist/esm/pdf-reader/flow-build.js +126 -0
- package/dist/esm/pdf-reader/font.d.ts +4 -0
- package/dist/esm/pdf-reader/font.js +69 -0
- package/dist/esm/pdf-reader/image-decode.d.ts +15 -0
- package/dist/esm/pdf-reader/image-decode.js +442 -0
- package/dist/esm/pdf-reader/images.d.ts +16 -0
- package/dist/esm/pdf-reader/images.js +90 -0
- package/dist/esm/pdf-reader/layout.d.ts +3 -0
- package/dist/esm/pdf-reader/layout.js +103 -0
- package/dist/esm/pdf-reader/lexer.d.ts +43 -0
- package/dist/esm/pdf-reader/lexer.js +250 -0
- package/dist/esm/pdf-reader/parser.d.ts +10 -0
- package/dist/esm/pdf-reader/parser.js +86 -0
- package/dist/esm/pdf-reader/png-encode.d.ts +2 -0
- package/dist/esm/pdf-reader/png-encode.js +98 -0
- package/dist/esm/pdf-reader/predictor.d.ts +7 -0
- package/dist/esm/pdf-reader/predictor.js +61 -0
- package/dist/esm/pdf-reader/reader.d.ts +4 -0
- package/dist/esm/pdf-reader/reader.js +50 -0
- package/dist/esm/pdf-reader/struct-tree.d.ts +14 -0
- package/dist/esm/pdf-reader/struct-tree.js +92 -0
- package/dist/esm/pdf-reader/tagged.d.ts +3 -0
- package/dist/esm/pdf-reader/tagged.js +141 -0
- package/dist/esm/pdf-reader/text.d.ts +3 -0
- package/dist/esm/pdf-reader/text.js +65 -0
- package/dist/esm/pdf-reader/vector.d.ts +12 -0
- package/dist/esm/pdf-reader/vector.js +54 -0
- package/dist/esm/word/docx-writer.js +96 -7
- package/dist/esm/word/omml-serializer.d.ts +2 -0
- package/dist/esm/word/omml-serializer.js +48 -0
- package/package.json +2 -2
|
@@ -0,0 +1,90 @@
|
|
|
1
|
+
import { PDF_NULL, PdfName, PdfStream } from "../pdf/objects.js";
|
|
2
|
+
import { FEATURES } from "../core/ir/features.js";
|
|
3
|
+
import { interpretContent, multiply } from "./content.js";
|
|
4
|
+
import { decodePdfImage } from "./image-decode.js";
|
|
5
|
+
//#region src/pdf-reader/images.ts
|
|
6
|
+
var NO_FONTS = /* @__PURE__ */ new Map();
|
|
7
|
+
var MAX_FORM_DEPTH = 12;
|
|
8
|
+
var MAX_IMAGES = 4096;
|
|
9
|
+
function collectPageImages(file, page) {
|
|
10
|
+
const images = [];
|
|
11
|
+
const lossByDetail = /* @__PURE__ */ new Map();
|
|
12
|
+
const visiting = /* @__PURE__ */ new Set();
|
|
13
|
+
const addLoss = (severity, detail) => {
|
|
14
|
+
if (!lossByDetail.has(detail)) lossByDetail.set(detail, {
|
|
15
|
+
severity,
|
|
16
|
+
feature: FEATURES.images,
|
|
17
|
+
detail
|
|
18
|
+
});
|
|
19
|
+
};
|
|
20
|
+
const walk = (resources, content, baseCtm, depth, inheritedMcid) => {
|
|
21
|
+
const xobjects = resources ? file.get(resources, "XObject") : PDF_NULL;
|
|
22
|
+
const xobjDict = xobjects instanceof Map ? xobjects : void 0;
|
|
23
|
+
for (const placement of interpretContent(content, NO_FONTS, baseCtm).images) {
|
|
24
|
+
if (images.length >= MAX_IMAGES) return;
|
|
25
|
+
const stream = xobjDict ? file.resolve(xobjDict.get(placement.name) ?? PDF_NULL) : PDF_NULL;
|
|
26
|
+
if (!(stream instanceof PdfStream)) continue;
|
|
27
|
+
const subtype = nameOf(file.get(stream.dict, "Subtype"));
|
|
28
|
+
const mcid = placement.mcid ?? inheritedMcid;
|
|
29
|
+
if (subtype === "Image") {
|
|
30
|
+
const decoded = decodePdfImage(file, stream);
|
|
31
|
+
if (decoded.ok) {
|
|
32
|
+
images.push(geometry(placement.ctm, decoded, mcid));
|
|
33
|
+
if (decoded.degraded) addLoss("degraded", decoded.degraded);
|
|
34
|
+
} else addLoss(decoded.severity, decoded.detail);
|
|
35
|
+
} else if (subtype === "Form" && depth < MAX_FORM_DEPTH && !visiting.has(stream)) {
|
|
36
|
+
visiting.add(stream);
|
|
37
|
+
const formRes = file.get(stream.dict, "Resources");
|
|
38
|
+
walk(formRes instanceof Map ? formRes : resources, file.streamData(stream), multiply(matrixOf(file, stream.dict), placement.ctm), depth + 1, mcid);
|
|
39
|
+
visiting.delete(stream);
|
|
40
|
+
}
|
|
41
|
+
}
|
|
42
|
+
};
|
|
43
|
+
walk(page.resources, file.pageContent(page), [
|
|
44
|
+
1,
|
|
45
|
+
0,
|
|
46
|
+
0,
|
|
47
|
+
1,
|
|
48
|
+
0,
|
|
49
|
+
0
|
|
50
|
+
], 0, void 0);
|
|
51
|
+
return {
|
|
52
|
+
images,
|
|
53
|
+
losses: [...lossByDetail.values()]
|
|
54
|
+
};
|
|
55
|
+
}
|
|
56
|
+
function geometry(ctm, decoded, mcid) {
|
|
57
|
+
return {
|
|
58
|
+
bytes: decoded.bytes,
|
|
59
|
+
format: decoded.format,
|
|
60
|
+
widthPt: Math.hypot(ctm[0], ctm[1]) || 1,
|
|
61
|
+
heightPt: Math.hypot(ctm[2], ctm[3]) || 1,
|
|
62
|
+
x: ctm[4],
|
|
63
|
+
y: ctm[5],
|
|
64
|
+
...mcid !== void 0 ? { mcid } : {}
|
|
65
|
+
};
|
|
66
|
+
}
|
|
67
|
+
function matrixOf(file, dict) {
|
|
68
|
+
const m = file.resolve(dict.get("Matrix") ?? PDF_NULL);
|
|
69
|
+
if (Array.isArray(m) && m.length >= 6 && m.every((v) => typeof v === "number")) return [
|
|
70
|
+
m[0],
|
|
71
|
+
m[1],
|
|
72
|
+
m[2],
|
|
73
|
+
m[3],
|
|
74
|
+
m[4],
|
|
75
|
+
m[5]
|
|
76
|
+
];
|
|
77
|
+
return [
|
|
78
|
+
1,
|
|
79
|
+
0,
|
|
80
|
+
0,
|
|
81
|
+
1,
|
|
82
|
+
0,
|
|
83
|
+
0
|
|
84
|
+
];
|
|
85
|
+
}
|
|
86
|
+
function nameOf(v) {
|
|
87
|
+
return v instanceof PdfName ? v.value : "";
|
|
88
|
+
}
|
|
89
|
+
//#endregion
|
|
90
|
+
export { collectPageImages };
|
|
@@ -0,0 +1,103 @@
|
|
|
1
|
+
import { ResourceStore } from "../core/ir/resources.js";
|
|
2
|
+
import { buildFlowDoc, dedupeLosses, imageBlock, paragraphFromRuns, shapeBlock } from "./flow-build.js";
|
|
3
|
+
import { collectPageImages } from "./images.js";
|
|
4
|
+
import { extractPageText } from "./text.js";
|
|
5
|
+
import { collectPageVectors } from "./vector.js";
|
|
6
|
+
//#region src/pdf-reader/layout.ts
|
|
7
|
+
function reconstructByLayout(file) {
|
|
8
|
+
const pages = file.pages();
|
|
9
|
+
const pageLines = pages.map((page) => groupIntoLines(extractPageText(file, page)).filter((l) => l.text.length > 0));
|
|
10
|
+
const medianFont = median(pageLines.flat().map((l) => l.fontSize)) || 12;
|
|
11
|
+
const resources = new ResourceStore();
|
|
12
|
+
const losses = [];
|
|
13
|
+
const body = [];
|
|
14
|
+
pages.forEach((page, i) => {
|
|
15
|
+
const blocks = [];
|
|
16
|
+
for (const para of groupIntoParagraphs(pageLines[i])) blocks.push({
|
|
17
|
+
top: para.top,
|
|
18
|
+
el: paragraphFromRuns(para.spans, headingLevel(para.fontSize, medianFont))
|
|
19
|
+
});
|
|
20
|
+
const imgs = collectPageImages(file, page);
|
|
21
|
+
losses.push(...imgs.losses);
|
|
22
|
+
for (const img of imgs.images) blocks.push({
|
|
23
|
+
top: img.y + img.heightPt,
|
|
24
|
+
el: imageBlock(img, resources)
|
|
25
|
+
});
|
|
26
|
+
for (const v of collectPageVectors(file, page)) blocks.push({
|
|
27
|
+
top: v.maxY,
|
|
28
|
+
el: shapeBlock(v)
|
|
29
|
+
});
|
|
30
|
+
blocks.sort((a, b) => b.top - a.top);
|
|
31
|
+
for (const block of blocks) body.push(block.el);
|
|
32
|
+
});
|
|
33
|
+
return {
|
|
34
|
+
doc: buildFlowDoc(body, resources),
|
|
35
|
+
losses: dedupeLosses(losses)
|
|
36
|
+
};
|
|
37
|
+
}
|
|
38
|
+
function groupIntoLines(runs) {
|
|
39
|
+
const sorted = [...runs].sort((a, b) => b.y - a.y || a.x - b.x);
|
|
40
|
+
const clusters = [];
|
|
41
|
+
for (const run of sorted) {
|
|
42
|
+
const last = clusters[clusters.length - 1];
|
|
43
|
+
const tol = Math.max(1, (run.fontSizePt || 10) * .5);
|
|
44
|
+
if (last && Math.abs(last.y - run.y) <= tol) {
|
|
45
|
+
last.runs.push(run);
|
|
46
|
+
last.fontSize = Math.max(last.fontSize, run.fontSizePt || 0);
|
|
47
|
+
} else clusters.push({
|
|
48
|
+
y: run.y,
|
|
49
|
+
fontSize: run.fontSizePt || 10,
|
|
50
|
+
runs: [run]
|
|
51
|
+
});
|
|
52
|
+
}
|
|
53
|
+
return clusters.map((c) => {
|
|
54
|
+
const ordered = c.runs.sort((a, b) => a.x - b.x);
|
|
55
|
+
const fontSize = c.fontSize || 10;
|
|
56
|
+
const spans = lineSpans(ordered, fontSize);
|
|
57
|
+
return {
|
|
58
|
+
y: c.y,
|
|
59
|
+
fontSize,
|
|
60
|
+
text: spans.map((s) => s.text).join("").replace(/\s+/g, " ").trim(),
|
|
61
|
+
spans
|
|
62
|
+
};
|
|
63
|
+
});
|
|
64
|
+
}
|
|
65
|
+
function lineSpans(runs, fontSize) {
|
|
66
|
+
const spans = [];
|
|
67
|
+
let prevEnd;
|
|
68
|
+
for (const run of runs) {
|
|
69
|
+
if (prevEnd !== void 0 && run.x - prevEnd > fontSize * .25) spans.push({ text: " " });
|
|
70
|
+
spans.push(run.href !== void 0 ? {
|
|
71
|
+
text: run.text,
|
|
72
|
+
href: run.href
|
|
73
|
+
} : { text: run.text });
|
|
74
|
+
prevEnd = run.x + run.text.length * (run.fontSizePt || fontSize) * .5;
|
|
75
|
+
}
|
|
76
|
+
return spans;
|
|
77
|
+
}
|
|
78
|
+
function groupIntoParagraphs(lines) {
|
|
79
|
+
const groups = [];
|
|
80
|
+
let prevY;
|
|
81
|
+
for (const line of lines) {
|
|
82
|
+
const gap = prevY !== void 0 ? prevY - line.y : 0;
|
|
83
|
+
if (groups.length === 0 || prevY !== void 0 && gap > line.fontSize * 1.5) groups.push([]);
|
|
84
|
+
groups[groups.length - 1].push(line);
|
|
85
|
+
prevY = line.y;
|
|
86
|
+
}
|
|
87
|
+
return groups.map((g) => ({
|
|
88
|
+
spans: g.flatMap((l, i) => i > 0 ? [{ text: " " }, ...l.spans] : [...l.spans]),
|
|
89
|
+
fontSize: Math.max(...g.map((l) => l.fontSize)),
|
|
90
|
+
top: g[0].y
|
|
91
|
+
}));
|
|
92
|
+
}
|
|
93
|
+
function headingLevel(fontSize, medianFont) {
|
|
94
|
+
if (fontSize >= medianFont * 1.5) return 0;
|
|
95
|
+
if (fontSize >= medianFont * 1.25) return 1;
|
|
96
|
+
}
|
|
97
|
+
function median(values) {
|
|
98
|
+
if (values.length === 0) return 0;
|
|
99
|
+
const sorted = [...values].sort((a, b) => a - b);
|
|
100
|
+
return sorted[Math.floor(sorted.length / 2)];
|
|
101
|
+
}
|
|
102
|
+
//#endregion
|
|
103
|
+
export { reconstructByLayout };
|
|
@@ -0,0 +1,43 @@
|
|
|
1
|
+
export type Token = {
|
|
2
|
+
readonly kind: 'num';
|
|
3
|
+
readonly value: number;
|
|
4
|
+
} | {
|
|
5
|
+
readonly kind: 'name';
|
|
6
|
+
readonly value: string;
|
|
7
|
+
} | {
|
|
8
|
+
readonly kind: 'str';
|
|
9
|
+
readonly value: string;
|
|
10
|
+
} | {
|
|
11
|
+
readonly kind: 'hexstr';
|
|
12
|
+
readonly bytes: Uint8Array;
|
|
13
|
+
} | {
|
|
14
|
+
readonly kind: 'arrayOpen';
|
|
15
|
+
} | {
|
|
16
|
+
readonly kind: 'arrayClose';
|
|
17
|
+
} | {
|
|
18
|
+
readonly kind: 'dictOpen';
|
|
19
|
+
} | {
|
|
20
|
+
readonly kind: 'dictClose';
|
|
21
|
+
} | {
|
|
22
|
+
readonly kind: 'keyword';
|
|
23
|
+
readonly value: string;
|
|
24
|
+
} | {
|
|
25
|
+
readonly kind: 'eof';
|
|
26
|
+
};
|
|
27
|
+
export declare class Lexer {
|
|
28
|
+
private readonly buf;
|
|
29
|
+
pos: number;
|
|
30
|
+
constructor(buf: Uint8Array, pos?: number);
|
|
31
|
+
get length(): number;
|
|
32
|
+
byteAt(i: number): number;
|
|
33
|
+
skipWhitespace(): void;
|
|
34
|
+
nextToken(): Token;
|
|
35
|
+
private readNumber;
|
|
36
|
+
private readName;
|
|
37
|
+
private readKeyword;
|
|
38
|
+
private readHexString;
|
|
39
|
+
private readLiteralString;
|
|
40
|
+
indexOfAscii(needle: string, from: number): number;
|
|
41
|
+
readStreamBody(length: number | undefined): Uint8Array;
|
|
42
|
+
}
|
|
43
|
+
export declare function latin1(bytes: Uint8Array): string;
|
|
@@ -0,0 +1,250 @@
|
|
|
1
|
+
//#region src/pdf-reader/lexer.ts
|
|
2
|
+
function isWhitespace(b) {
|
|
3
|
+
return b === 0 || b === 9 || b === 10 || b === 12 || b === 13 || b === 32;
|
|
4
|
+
}
|
|
5
|
+
function isDelimiter(b) {
|
|
6
|
+
return b === 40 || b === 41 || b === 60 || b === 62 || b === 91 || b === 93 || b === 123 || b === 125 || b === 47 || b === 37;
|
|
7
|
+
}
|
|
8
|
+
function isRegular(b) {
|
|
9
|
+
return !isWhitespace(b) && !isDelimiter(b);
|
|
10
|
+
}
|
|
11
|
+
function hexVal(b) {
|
|
12
|
+
if (b >= 48 && b <= 57) return b - 48;
|
|
13
|
+
if (b >= 65 && b <= 70) return b - 65 + 10;
|
|
14
|
+
if (b >= 97 && b <= 102) return b - 97 + 10;
|
|
15
|
+
return -1;
|
|
16
|
+
}
|
|
17
|
+
var Lexer = class {
|
|
18
|
+
pos;
|
|
19
|
+
constructor(buf, pos = 0) {
|
|
20
|
+
this.buf = buf;
|
|
21
|
+
this.pos = pos;
|
|
22
|
+
}
|
|
23
|
+
get length() {
|
|
24
|
+
return this.buf.length;
|
|
25
|
+
}
|
|
26
|
+
byteAt(i) {
|
|
27
|
+
return i >= 0 && i < this.buf.length ? this.buf[i] : -1;
|
|
28
|
+
}
|
|
29
|
+
skipWhitespace() {
|
|
30
|
+
const buf = this.buf;
|
|
31
|
+
while (this.pos < buf.length) {
|
|
32
|
+
const b = buf[this.pos];
|
|
33
|
+
if (isWhitespace(b)) this.pos++;
|
|
34
|
+
else if (b === 37) {
|
|
35
|
+
this.pos++;
|
|
36
|
+
while (this.pos < buf.length && buf[this.pos] !== 10 && buf[this.pos] !== 13) this.pos++;
|
|
37
|
+
} else break;
|
|
38
|
+
}
|
|
39
|
+
}
|
|
40
|
+
nextToken() {
|
|
41
|
+
this.skipWhitespace();
|
|
42
|
+
const buf = this.buf;
|
|
43
|
+
if (this.pos >= buf.length) return { kind: "eof" };
|
|
44
|
+
const b = buf[this.pos];
|
|
45
|
+
switch (b) {
|
|
46
|
+
case 91:
|
|
47
|
+
this.pos++;
|
|
48
|
+
return { kind: "arrayOpen" };
|
|
49
|
+
case 93:
|
|
50
|
+
this.pos++;
|
|
51
|
+
return { kind: "arrayClose" };
|
|
52
|
+
case 60:
|
|
53
|
+
if (buf[this.pos + 1] === 60) {
|
|
54
|
+
this.pos += 2;
|
|
55
|
+
return { kind: "dictOpen" };
|
|
56
|
+
}
|
|
57
|
+
return this.readHexString();
|
|
58
|
+
case 62:
|
|
59
|
+
if (buf[this.pos + 1] === 62) {
|
|
60
|
+
this.pos += 2;
|
|
61
|
+
return { kind: "dictClose" };
|
|
62
|
+
}
|
|
63
|
+
this.pos++;
|
|
64
|
+
return this.nextToken();
|
|
65
|
+
case 40: return this.readLiteralString();
|
|
66
|
+
case 47: return this.readName();
|
|
67
|
+
case 123:
|
|
68
|
+
case 125:
|
|
69
|
+
this.pos++;
|
|
70
|
+
return {
|
|
71
|
+
kind: "keyword",
|
|
72
|
+
value: String.fromCharCode(b)
|
|
73
|
+
};
|
|
74
|
+
}
|
|
75
|
+
if (b === 43 || b === 45 || b === 46 || b >= 48 && b <= 57) return this.readNumber();
|
|
76
|
+
if (isRegular(b)) return this.readKeyword();
|
|
77
|
+
this.pos++;
|
|
78
|
+
return this.nextToken();
|
|
79
|
+
}
|
|
80
|
+
readNumber() {
|
|
81
|
+
const buf = this.buf;
|
|
82
|
+
const start = this.pos;
|
|
83
|
+
if (buf[this.pos] === 43 || buf[this.pos] === 45) this.pos++;
|
|
84
|
+
while (this.pos < buf.length) {
|
|
85
|
+
const b = buf[this.pos];
|
|
86
|
+
if (b >= 48 && b <= 57 || b === 46) this.pos++;
|
|
87
|
+
else break;
|
|
88
|
+
}
|
|
89
|
+
const text = latin1(buf.subarray(start, this.pos));
|
|
90
|
+
const value = Number(text);
|
|
91
|
+
return {
|
|
92
|
+
kind: "num",
|
|
93
|
+
value: Number.isFinite(value) ? value : 0
|
|
94
|
+
};
|
|
95
|
+
}
|
|
96
|
+
readName() {
|
|
97
|
+
const buf = this.buf;
|
|
98
|
+
this.pos++;
|
|
99
|
+
const out = [];
|
|
100
|
+
while (this.pos < buf.length) {
|
|
101
|
+
const b = buf[this.pos];
|
|
102
|
+
if (!isRegular(b)) break;
|
|
103
|
+
if (b === 35 && this.pos + 2 < buf.length) {
|
|
104
|
+
const hi = hexVal(buf[this.pos + 1]);
|
|
105
|
+
const lo = hexVal(buf[this.pos + 2]);
|
|
106
|
+
if (hi >= 0 && lo >= 0) {
|
|
107
|
+
out.push(hi * 16 + lo);
|
|
108
|
+
this.pos += 3;
|
|
109
|
+
continue;
|
|
110
|
+
}
|
|
111
|
+
}
|
|
112
|
+
out.push(b);
|
|
113
|
+
this.pos++;
|
|
114
|
+
}
|
|
115
|
+
return {
|
|
116
|
+
kind: "name",
|
|
117
|
+
value: latin1(Uint8Array.from(out))
|
|
118
|
+
};
|
|
119
|
+
}
|
|
120
|
+
readKeyword() {
|
|
121
|
+
const buf = this.buf;
|
|
122
|
+
const start = this.pos;
|
|
123
|
+
while (this.pos < buf.length && isRegular(buf[this.pos])) this.pos++;
|
|
124
|
+
return {
|
|
125
|
+
kind: "keyword",
|
|
126
|
+
value: latin1(buf.subarray(start, this.pos))
|
|
127
|
+
};
|
|
128
|
+
}
|
|
129
|
+
readHexString() {
|
|
130
|
+
const buf = this.buf;
|
|
131
|
+
this.pos++;
|
|
132
|
+
const out = [];
|
|
133
|
+
let hi = -1;
|
|
134
|
+
while (this.pos < buf.length) {
|
|
135
|
+
const b = buf[this.pos];
|
|
136
|
+
this.pos++;
|
|
137
|
+
if (b === 62) break;
|
|
138
|
+
const v = hexVal(b);
|
|
139
|
+
if (v < 0) continue;
|
|
140
|
+
if (hi < 0) hi = v;
|
|
141
|
+
else {
|
|
142
|
+
out.push(hi * 16 + v);
|
|
143
|
+
hi = -1;
|
|
144
|
+
}
|
|
145
|
+
}
|
|
146
|
+
if (hi >= 0) out.push(hi * 16);
|
|
147
|
+
return {
|
|
148
|
+
kind: "hexstr",
|
|
149
|
+
bytes: Uint8Array.from(out)
|
|
150
|
+
};
|
|
151
|
+
}
|
|
152
|
+
readLiteralString() {
|
|
153
|
+
const buf = this.buf;
|
|
154
|
+
this.pos++;
|
|
155
|
+
const out = [];
|
|
156
|
+
let depth = 1;
|
|
157
|
+
while (this.pos < buf.length) {
|
|
158
|
+
const b = buf[this.pos];
|
|
159
|
+
this.pos++;
|
|
160
|
+
if (b === 92) {
|
|
161
|
+
if (this.pos >= buf.length) break;
|
|
162
|
+
const e = buf[this.pos];
|
|
163
|
+
this.pos++;
|
|
164
|
+
switch (e) {
|
|
165
|
+
case 110:
|
|
166
|
+
out.push(10);
|
|
167
|
+
break;
|
|
168
|
+
case 114:
|
|
169
|
+
out.push(13);
|
|
170
|
+
break;
|
|
171
|
+
case 116:
|
|
172
|
+
out.push(9);
|
|
173
|
+
break;
|
|
174
|
+
case 98:
|
|
175
|
+
out.push(8);
|
|
176
|
+
break;
|
|
177
|
+
case 102:
|
|
178
|
+
out.push(12);
|
|
179
|
+
break;
|
|
180
|
+
case 10: break;
|
|
181
|
+
case 13:
|
|
182
|
+
if (buf[this.pos] === 10) this.pos++;
|
|
183
|
+
break;
|
|
184
|
+
default: if (e >= 48 && e <= 55) {
|
|
185
|
+
let oct = e - 48;
|
|
186
|
+
for (let k = 0; k < 2 && this.pos < buf.length; k++) {
|
|
187
|
+
const d = buf[this.pos];
|
|
188
|
+
if (d < 48 || d > 55) break;
|
|
189
|
+
oct = oct * 8 + (d - 48);
|
|
190
|
+
this.pos++;
|
|
191
|
+
}
|
|
192
|
+
out.push(oct & 255);
|
|
193
|
+
} else out.push(e);
|
|
194
|
+
}
|
|
195
|
+
continue;
|
|
196
|
+
}
|
|
197
|
+
if (b === 40) {
|
|
198
|
+
depth++;
|
|
199
|
+
out.push(b);
|
|
200
|
+
continue;
|
|
201
|
+
}
|
|
202
|
+
if (b === 41) {
|
|
203
|
+
depth--;
|
|
204
|
+
if (depth === 0) break;
|
|
205
|
+
out.push(b);
|
|
206
|
+
continue;
|
|
207
|
+
}
|
|
208
|
+
out.push(b);
|
|
209
|
+
}
|
|
210
|
+
return {
|
|
211
|
+
kind: "str",
|
|
212
|
+
value: latin1(Uint8Array.from(out))
|
|
213
|
+
};
|
|
214
|
+
}
|
|
215
|
+
indexOfAscii(needle, from) {
|
|
216
|
+
const buf = this.buf;
|
|
217
|
+
const n = needle.length;
|
|
218
|
+
outer: for (let i = from; i <= buf.length - n; i++) {
|
|
219
|
+
for (let j = 0; j < n; j++) if (buf[i + j] !== needle.charCodeAt(j)) continue outer;
|
|
220
|
+
return i;
|
|
221
|
+
}
|
|
222
|
+
return -1;
|
|
223
|
+
}
|
|
224
|
+
readStreamBody(length) {
|
|
225
|
+
const buf = this.buf;
|
|
226
|
+
if (buf[this.pos] === 13 && buf[this.pos + 1] === 10) this.pos += 2;
|
|
227
|
+
else if (buf[this.pos] === 10 || buf[this.pos] === 13) this.pos += 1;
|
|
228
|
+
const start = this.pos;
|
|
229
|
+
if (length !== void 0 && length >= 0 && start + length <= buf.length) {
|
|
230
|
+
this.pos = start + length;
|
|
231
|
+
return buf.subarray(start, start + length);
|
|
232
|
+
}
|
|
233
|
+
const es = this.indexOfAscii("endstream", start);
|
|
234
|
+
const end = es < 0 ? buf.length : es;
|
|
235
|
+
let dataEnd = end;
|
|
236
|
+
if (dataEnd > start && buf[dataEnd - 1] === 10) {
|
|
237
|
+
dataEnd--;
|
|
238
|
+
if (dataEnd > start && buf[dataEnd - 1] === 13) dataEnd--;
|
|
239
|
+
} else if (dataEnd > start && buf[dataEnd - 1] === 13) dataEnd--;
|
|
240
|
+
this.pos = end;
|
|
241
|
+
return buf.subarray(start, dataEnd);
|
|
242
|
+
}
|
|
243
|
+
};
|
|
244
|
+
function latin1(bytes) {
|
|
245
|
+
let s = "";
|
|
246
|
+
for (const b of bytes) s += String.fromCharCode(b);
|
|
247
|
+
return s;
|
|
248
|
+
}
|
|
249
|
+
//#endregion
|
|
250
|
+
export { Lexer };
|
|
@@ -0,0 +1,10 @@
|
|
|
1
|
+
import { Lexer } from './lexer.js';
|
|
2
|
+
import { PdfValue, PdfRef } from '../pdf/objects.js';
|
|
3
|
+
export type LengthResolver = (ref: PdfRef) => number | undefined;
|
|
4
|
+
export interface IndirectObject {
|
|
5
|
+
readonly id: number;
|
|
6
|
+
readonly generation: number;
|
|
7
|
+
readonly value: PdfValue;
|
|
8
|
+
}
|
|
9
|
+
export declare function parseObject(lexer: Lexer, resolveLength?: LengthResolver): PdfValue;
|
|
10
|
+
export declare function parseIndirectObject(lexer: Lexer, resolveLength?: LengthResolver): IndirectObject | undefined;
|
|
@@ -0,0 +1,86 @@
|
|
|
1
|
+
import { PDF_NULL, PdfHexString, PdfName, PdfRef, PdfStream } from "../pdf/objects.js";
|
|
2
|
+
//#region src/pdf-reader/parser.ts
|
|
3
|
+
function parseObject(lexer, resolveLength) {
|
|
4
|
+
return parseValue(lexer, lexer.nextToken(), resolveLength);
|
|
5
|
+
}
|
|
6
|
+
function parseIndirectObject(lexer, resolveLength) {
|
|
7
|
+
const idTok = lexer.nextToken();
|
|
8
|
+
if (idTok.kind !== "num") return void 0;
|
|
9
|
+
const genTok = lexer.nextToken();
|
|
10
|
+
if (genTok.kind !== "num") return void 0;
|
|
11
|
+
const objTok = lexer.nextToken();
|
|
12
|
+
if (objTok.kind !== "keyword" || objTok.value !== "obj") return void 0;
|
|
13
|
+
const value = parseObject(lexer, resolveLength);
|
|
14
|
+
const save = lexer.pos;
|
|
15
|
+
const end = lexer.nextToken();
|
|
16
|
+
if (!(end.kind === "keyword" && end.value === "endobj")) lexer.pos = save;
|
|
17
|
+
return {
|
|
18
|
+
id: idTok.value,
|
|
19
|
+
generation: genTok.value,
|
|
20
|
+
value
|
|
21
|
+
};
|
|
22
|
+
}
|
|
23
|
+
function parseValue(lexer, tok, resolveLength) {
|
|
24
|
+
switch (tok.kind) {
|
|
25
|
+
case "num": return parseNumberOrRef(lexer, tok.value);
|
|
26
|
+
case "name": return new PdfName(tok.value);
|
|
27
|
+
case "str": return tok.value;
|
|
28
|
+
case "hexstr": return new PdfHexString(tok.bytes);
|
|
29
|
+
case "arrayOpen": return parseArray(lexer, resolveLength);
|
|
30
|
+
case "dictOpen": return parseDictOrStream(lexer, resolveLength);
|
|
31
|
+
case "keyword":
|
|
32
|
+
if (tok.value === "true") return true;
|
|
33
|
+
if (tok.value === "false") return false;
|
|
34
|
+
return PDF_NULL;
|
|
35
|
+
default: return PDF_NULL;
|
|
36
|
+
}
|
|
37
|
+
}
|
|
38
|
+
function parseNumberOrRef(lexer, first) {
|
|
39
|
+
if (!Number.isInteger(first) || first < 0) return first;
|
|
40
|
+
const save = lexer.pos;
|
|
41
|
+
const gen = lexer.nextToken();
|
|
42
|
+
if (gen.kind === "num" && Number.isInteger(gen.value)) {
|
|
43
|
+
const r = lexer.nextToken();
|
|
44
|
+
if (r.kind === "keyword" && r.value === "R") return new PdfRef(first, gen.value);
|
|
45
|
+
}
|
|
46
|
+
lexer.pos = save;
|
|
47
|
+
return first;
|
|
48
|
+
}
|
|
49
|
+
function parseArray(lexer, resolveLength) {
|
|
50
|
+
const out = [];
|
|
51
|
+
for (;;) {
|
|
52
|
+
const tok = lexer.nextToken();
|
|
53
|
+
if (tok.kind === "arrayClose" || tok.kind === "eof") break;
|
|
54
|
+
out.push(parseValue(lexer, tok, resolveLength));
|
|
55
|
+
}
|
|
56
|
+
return out;
|
|
57
|
+
}
|
|
58
|
+
function parseDictOrStream(lexer, resolveLength) {
|
|
59
|
+
const map = /* @__PURE__ */ new Map();
|
|
60
|
+
for (;;) {
|
|
61
|
+
const keyTok = lexer.nextToken();
|
|
62
|
+
if (keyTok.kind === "dictClose" || keyTok.kind === "eof") break;
|
|
63
|
+
if (keyTok.kind !== "name") {
|
|
64
|
+
if (keyTok.kind === "arrayOpen" || keyTok.kind === "dictOpen") parseValue(lexer, keyTok, resolveLength);
|
|
65
|
+
continue;
|
|
66
|
+
}
|
|
67
|
+
const value = parseObject(lexer, resolveLength);
|
|
68
|
+
map.set(keyTok.value, value);
|
|
69
|
+
}
|
|
70
|
+
const save = lexer.pos;
|
|
71
|
+
const next = lexer.nextToken();
|
|
72
|
+
if (next.kind === "keyword" && next.value === "stream") {
|
|
73
|
+
const lengthVal = map.get("Length");
|
|
74
|
+
let length;
|
|
75
|
+
if (typeof lengthVal === "number") length = lengthVal;
|
|
76
|
+
else if (lengthVal instanceof PdfRef && resolveLength) length = resolveLength(lengthVal);
|
|
77
|
+
const data = lexer.readStreamBody(length);
|
|
78
|
+
const endTok = lexer.nextToken();
|
|
79
|
+
if (!(endTok.kind === "keyword" && endTok.value === "endstream")) {}
|
|
80
|
+
return new PdfStream(map, data);
|
|
81
|
+
}
|
|
82
|
+
lexer.pos = save;
|
|
83
|
+
return map;
|
|
84
|
+
}
|
|
85
|
+
//#endregion
|
|
86
|
+
export { parseIndirectObject, parseObject };
|
|
@@ -0,0 +1,98 @@
|
|
|
1
|
+
import { zlibSync } from "fflate";
|
|
2
|
+
//#region src/pdf-reader/png-encode.ts
|
|
3
|
+
var COLOR_TYPE = {
|
|
4
|
+
gray: 0,
|
|
5
|
+
rgb: 2,
|
|
6
|
+
"gray-alpha": 4,
|
|
7
|
+
rgba: 6
|
|
8
|
+
};
|
|
9
|
+
var CHANNELS = {
|
|
10
|
+
gray: 1,
|
|
11
|
+
rgb: 3,
|
|
12
|
+
"gray-alpha": 2,
|
|
13
|
+
rgba: 4
|
|
14
|
+
};
|
|
15
|
+
var SIGNATURE = Uint8Array.from([
|
|
16
|
+
137,
|
|
17
|
+
80,
|
|
18
|
+
78,
|
|
19
|
+
71,
|
|
20
|
+
13,
|
|
21
|
+
10,
|
|
22
|
+
26,
|
|
23
|
+
10
|
|
24
|
+
]);
|
|
25
|
+
function encodePng(width, height, color, samples) {
|
|
26
|
+
const stride = width * CHANNELS[color];
|
|
27
|
+
const raw = new Uint8Array(height * (stride + 1));
|
|
28
|
+
for (let y = 0; y < height; y++) {
|
|
29
|
+
const dst = y * (stride + 1);
|
|
30
|
+
raw[dst] = 0;
|
|
31
|
+
raw.set(samples.subarray(y * stride, y * stride + stride), dst + 1);
|
|
32
|
+
}
|
|
33
|
+
const idat = zlibSync(raw);
|
|
34
|
+
const ihdr = new Uint8Array(13);
|
|
35
|
+
writeU32(ihdr, 0, width);
|
|
36
|
+
writeU32(ihdr, 4, height);
|
|
37
|
+
ihdr[8] = 8;
|
|
38
|
+
ihdr[9] = COLOR_TYPE[color];
|
|
39
|
+
ihdr[10] = 0;
|
|
40
|
+
ihdr[11] = 0;
|
|
41
|
+
ihdr[12] = 0;
|
|
42
|
+
return concat([
|
|
43
|
+
SIGNATURE,
|
|
44
|
+
chunk("IHDR", ihdr),
|
|
45
|
+
chunk("IDAT", idat),
|
|
46
|
+
chunk("IEND", new Uint8Array(0))
|
|
47
|
+
]);
|
|
48
|
+
}
|
|
49
|
+
function chunk(type, data) {
|
|
50
|
+
const typeBytes = Uint8Array.from([
|
|
51
|
+
type.charCodeAt(0),
|
|
52
|
+
type.charCodeAt(1),
|
|
53
|
+
type.charCodeAt(2),
|
|
54
|
+
type.charCodeAt(3)
|
|
55
|
+
]);
|
|
56
|
+
const out = new Uint8Array(12 + data.length);
|
|
57
|
+
writeU32(out, 0, data.length);
|
|
58
|
+
out.set(typeBytes, 4);
|
|
59
|
+
out.set(data, 8);
|
|
60
|
+
const crcInput = new Uint8Array(4 + data.length);
|
|
61
|
+
crcInput.set(typeBytes, 0);
|
|
62
|
+
crcInput.set(data, 4);
|
|
63
|
+
writeU32(out, 8 + data.length, crc32(crcInput));
|
|
64
|
+
return out;
|
|
65
|
+
}
|
|
66
|
+
function writeU32(buf, offset, value) {
|
|
67
|
+
buf[offset] = value >>> 24 & 255;
|
|
68
|
+
buf[offset + 1] = value >>> 16 & 255;
|
|
69
|
+
buf[offset + 2] = value >>> 8 & 255;
|
|
70
|
+
buf[offset + 3] = value & 255;
|
|
71
|
+
}
|
|
72
|
+
function concat(parts) {
|
|
73
|
+
let total = 0;
|
|
74
|
+
for (const p of parts) total += p.length;
|
|
75
|
+
const out = new Uint8Array(total);
|
|
76
|
+
let off = 0;
|
|
77
|
+
for (const p of parts) {
|
|
78
|
+
out.set(p, off);
|
|
79
|
+
off += p.length;
|
|
80
|
+
}
|
|
81
|
+
return out;
|
|
82
|
+
}
|
|
83
|
+
var crcTable;
|
|
84
|
+
function crc32(bytes) {
|
|
85
|
+
if (!crcTable) {
|
|
86
|
+
crcTable = new Uint32Array(256);
|
|
87
|
+
for (let n = 0; n < 256; n++) {
|
|
88
|
+
let c = n;
|
|
89
|
+
for (let k = 0; k < 8; k++) c = c & 1 ? 3988292384 ^ c >>> 1 : c >>> 1;
|
|
90
|
+
crcTable[n] = c >>> 0;
|
|
91
|
+
}
|
|
92
|
+
}
|
|
93
|
+
let crc = 4294967295;
|
|
94
|
+
for (const b of bytes) crc = crcTable[(crc ^ b) & 255] ^ crc >>> 8;
|
|
95
|
+
return (crc ^ 4294967295) >>> 0;
|
|
96
|
+
}
|
|
97
|
+
//#endregion
|
|
98
|
+
export { encodePng };
|