reamkit 1.29.0 → 1.30.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/esm/core/document-model/types.d.ts +2 -0
- package/dist/esm/core/drawingml/shape-render.js +13 -1
- package/dist/esm/core/fonts/index.d.ts +1 -1
- package/dist/esm/core/fonts/remote-fonts.d.ts +8 -0
- package/dist/esm/core/fonts/remote-fonts.js +100 -17
- package/dist/esm/core/ir/flow.d.ts +17 -0
- package/dist/esm/index.d.ts +1 -1
- package/dist/esm/pdf-reader/annot-draw.js +93 -1
- package/dist/esm/pdf-reader/annots.d.ts +18 -0
- package/dist/esm/pdf-reader/annots.js +86 -9
- package/dist/esm/pdf-reader/cmap.js +5 -2
- package/dist/esm/pdf-reader/content.d.ts +17 -4
- package/dist/esm/pdf-reader/content.js +163 -11
- package/dist/esm/pdf-reader/display.js +16 -0
- package/dist/esm/pdf-reader/embedded-fonts.d.ts +24 -0
- package/dist/esm/pdf-reader/embedded-fonts.js +48 -9
- package/dist/esm/pdf-reader/flow-build.d.ts +96 -5
- package/dist/esm/pdf-reader/flow-build.js +210 -23
- package/dist/esm/pdf-reader/font.d.ts +27 -1
- package/dist/esm/pdf-reader/font.js +313 -29
- package/dist/esm/pdf-reader/glyf-outline.js +13 -2
- package/dist/esm/pdf-reader/glyph-shapes.d.ts +18 -0
- package/dist/esm/pdf-reader/glyph-shapes.js +57 -0
- package/dist/esm/pdf-reader/image-decode.js +70 -4
- package/dist/esm/pdf-reader/images.d.ts +5 -0
- package/dist/esm/pdf-reader/images.js +4 -2
- package/dist/esm/pdf-reader/jbig2.d.ts +40 -1
- package/dist/esm/pdf-reader/jbig2.js +78 -16
- package/dist/esm/pdf-reader/jpeg.d.ts +6 -3
- package/dist/esm/pdf-reader/jpeg.js +21 -1
- package/dist/esm/pdf-reader/layout.d.ts +26 -2
- package/dist/esm/pdf-reader/layout.js +638 -92
- package/dist/esm/pdf-reader/lexer.d.ts +10 -0
- package/dist/esm/pdf-reader/lexer.js +17 -0
- package/dist/esm/pdf-reader/pattern-tint.d.ts +11 -1
- package/dist/esm/pdf-reader/pattern-tint.js +21 -3
- package/dist/esm/pdf-reader/regions.d.ts +25 -0
- package/dist/esm/pdf-reader/regions.js +167 -0
- package/dist/esm/pdf-reader/shading.d.ts +58 -2
- package/dist/esm/pdf-reader/shading.js +181 -11
- package/dist/esm/pdf-reader/struct-tree.js +112 -8
- package/dist/esm/pdf-reader/tagged.js +27 -7
- package/dist/esm/pdf-reader/text-rules.js +1 -1
- package/dist/esm/pdf-reader/text.js +9 -14
- package/dist/esm/pdf-reader/vector.d.ts +8 -2
- package/dist/esm/pdf-reader/vector.js +34 -5
- package/dist/esm/word/docx-writer.js +137 -36
- package/dist/esm/word/drawing-parser.js +7 -2
- package/dist/esm/word/paragraph-properties.js +2 -0
- package/package.json +5 -3
|
@@ -23,17 +23,17 @@ function readStructTree(file) {
|
|
|
23
23
|
return pg instanceof Map ? pageMap.get(pg) : void 0;
|
|
24
24
|
};
|
|
25
25
|
const seen = /* @__PURE__ */ new Set();
|
|
26
|
+
const typeOf = roleResolver(file, stRoot);
|
|
26
27
|
const read = (value, parentPage) => {
|
|
27
28
|
const elem = file.resolve(value);
|
|
28
29
|
if (!(elem instanceof Map) || seen.has(elem) || seen.size > MAX_NODES) return void 0;
|
|
29
30
|
seen.add(elem);
|
|
30
31
|
const ownPage = pageIndexOf(elem.get("Pg") ?? PDF_NULL) ?? parentPage;
|
|
31
|
-
const
|
|
32
|
-
const children = [];
|
|
32
|
+
const content = [];
|
|
33
33
|
for (const kid of kidList(file, elem.get("K"))) {
|
|
34
34
|
const rk = file.resolve(kid);
|
|
35
35
|
if (typeof rk === "number") {
|
|
36
|
-
if (ownPage >= 0)
|
|
36
|
+
if (ownPage >= 0) content.push({
|
|
37
37
|
page: ownPage,
|
|
38
38
|
mcid: rk
|
|
39
39
|
});
|
|
@@ -42,22 +42,24 @@ function readStructTree(file) {
|
|
|
42
42
|
if (kind === "MCR") {
|
|
43
43
|
const m = rk.get("MCID");
|
|
44
44
|
const page = pageIndexOf(rk.get("Pg") ?? PDF_NULL) ?? ownPage;
|
|
45
|
-
if (typeof m === "number" && page >= 0)
|
|
45
|
+
if (typeof m === "number" && page >= 0) content.push({
|
|
46
46
|
page,
|
|
47
47
|
mcid: m
|
|
48
48
|
});
|
|
49
49
|
} else if (kind === "OBJR") {} else {
|
|
50
50
|
const child = read(rk, ownPage);
|
|
51
|
-
if (child)
|
|
51
|
+
if (!child) continue;
|
|
52
|
+
if (inline(file, rk, child.type)) content.push(...allMcids(child));
|
|
53
|
+
else content.push(child);
|
|
52
54
|
}
|
|
53
55
|
}
|
|
54
56
|
}
|
|
57
|
+
const type = typeOf(nameOf(elem.get("S")));
|
|
55
58
|
const alt = elem.get("Alt");
|
|
56
59
|
const { colSpan, rowSpan } = readSpans(file, elem.get("A") ?? PDF_NULL);
|
|
57
60
|
return {
|
|
58
|
-
type
|
|
59
|
-
|
|
60
|
-
children,
|
|
61
|
+
type,
|
|
62
|
+
...settle(type, content),
|
|
61
63
|
...typeof alt === "string" ? { alt } : {},
|
|
62
64
|
...colSpan > 1 ? { colSpan } : {},
|
|
63
65
|
...rowSpan > 1 ? { rowSpan } : {}
|
|
@@ -71,6 +73,108 @@ function readStructTree(file) {
|
|
|
71
73
|
children: roots
|
|
72
74
|
};
|
|
73
75
|
}
|
|
76
|
+
/**
|
|
77
|
+
* An element's own marked content and its child elements, as the node keeps
|
|
78
|
+
* them.
|
|
79
|
+
*
|
|
80
|
+
* An element holds text of its own AND block elements only where the tree
|
|
81
|
+
* mixes them — a heading carrying its number as a child, a paragraph with a
|
|
82
|
+
* formula set between two of its sentences. Kept as two lists, the order
|
|
83
|
+
* between them was gone and the element's own text was never read at all:
|
|
84
|
+
* bug1937438_mml_from_latex.pdf's heading came back as "1" without "A small
|
|
85
|
+
* example", and its sentence around a formula without its words. Each
|
|
86
|
+
* stretch of the element's own content becomes a paragraph of its own
|
|
87
|
+
* standing where it stood — except in a figure, whose marked content is its
|
|
88
|
+
* picture.
|
|
89
|
+
*/
|
|
90
|
+
function settle(type, content) {
|
|
91
|
+
const nodes = content.filter((c) => "type" in c);
|
|
92
|
+
const own = content.filter((c) => !("type" in c));
|
|
93
|
+
if (nodes.length === 0 || own.length === 0 || type === "Figure") return {
|
|
94
|
+
mcids: own,
|
|
95
|
+
children: nodes
|
|
96
|
+
};
|
|
97
|
+
const children = [];
|
|
98
|
+
let stretch = [];
|
|
99
|
+
const close = () => {
|
|
100
|
+
if (stretch.length > 0) children.push({
|
|
101
|
+
type: "P",
|
|
102
|
+
mcids: stretch,
|
|
103
|
+
children: []
|
|
104
|
+
});
|
|
105
|
+
stretch = [];
|
|
106
|
+
};
|
|
107
|
+
for (const c of content) if ("type" in c) {
|
|
108
|
+
close();
|
|
109
|
+
children.push(c);
|
|
110
|
+
} else stretch.push(c);
|
|
111
|
+
close();
|
|
112
|
+
return {
|
|
113
|
+
mcids: [],
|
|
114
|
+
children
|
|
115
|
+
};
|
|
116
|
+
}
|
|
117
|
+
/** Every marked content reference under a node, in reading order. */
|
|
118
|
+
function allMcids(node) {
|
|
119
|
+
return [...node.mcids, ...node.children.flatMap(allMcids)];
|
|
120
|
+
}
|
|
121
|
+
/**
|
|
122
|
+
* §14.8.4.4 — the inline-level structure types (PDF 2.0 adds `Em`, `Strong`,
|
|
123
|
+
* `Sub`), a formula (§14.8.4.5.5, which is set in a line as often as out of
|
|
124
|
+
* one), and a list label, which is the start of its item's line.
|
|
125
|
+
*/
|
|
126
|
+
var INLINE = new Set([
|
|
127
|
+
"Span",
|
|
128
|
+
"Quote",
|
|
129
|
+
"Reference",
|
|
130
|
+
"BibEntry",
|
|
131
|
+
"Code",
|
|
132
|
+
"Link",
|
|
133
|
+
"Annot",
|
|
134
|
+
"Ruby",
|
|
135
|
+
"RB",
|
|
136
|
+
"RT",
|
|
137
|
+
"RP",
|
|
138
|
+
"Warichu",
|
|
139
|
+
"WT",
|
|
140
|
+
"WP",
|
|
141
|
+
"Em",
|
|
142
|
+
"Strong",
|
|
143
|
+
"Sub",
|
|
144
|
+
"Formula",
|
|
145
|
+
"Lbl"
|
|
146
|
+
]);
|
|
147
|
+
/** The MathML namespace (ISO 32000-2 §14.8.6): an element in it is mathematics. */
|
|
148
|
+
var MATHML = "http://www.w3.org/1998/Math/MathML";
|
|
149
|
+
/** Whether an element is inline — by its standard type, or as MathML. */
|
|
150
|
+
function inline(file, elem, type) {
|
|
151
|
+
if (INLINE.has(type)) return true;
|
|
152
|
+
const ns = file.resolve(elem.get("NS") ?? PDF_NULL);
|
|
153
|
+
const uri = ns instanceof Map ? file.resolve(ns.get("NS") ?? PDF_NULL) : void 0;
|
|
154
|
+
return typeof uri === "string" && uri === MATHML;
|
|
155
|
+
}
|
|
156
|
+
/**
|
|
157
|
+
* §14.8.4.2 `/RoleMap` — a document's own structure types, mapped to the
|
|
158
|
+
* standard ones they stand for, followed until a type maps no further. A
|
|
159
|
+
* LaTeX document names its elements `section`, `text-unit` and
|
|
160
|
+
* `section-number`, and maps them to `H1`, `Part` and `Span`: read by their
|
|
161
|
+
* own names, none of them meant anything.
|
|
162
|
+
*/
|
|
163
|
+
function roleResolver(file, stRoot) {
|
|
164
|
+
const map = file.resolve(stRoot.get("RoleMap") ?? PDF_NULL);
|
|
165
|
+
if (!(map instanceof Map)) return (type) => type;
|
|
166
|
+
return (type) => {
|
|
167
|
+
let at = type;
|
|
168
|
+
for (let hop = 0; hop < MAX_ROLE_HOPS; hop++) {
|
|
169
|
+
const next = file.resolve(map.get(at) ?? PDF_NULL);
|
|
170
|
+
if (!(next instanceof PdfName) || next.value === at) break;
|
|
171
|
+
at = next.value;
|
|
172
|
+
}
|
|
173
|
+
return at;
|
|
174
|
+
};
|
|
175
|
+
}
|
|
176
|
+
/** How far a chain of role mappings is followed before it is taken for a cycle. */
|
|
177
|
+
var MAX_ROLE_HOPS = 8;
|
|
74
178
|
function kidList(file, kVal) {
|
|
75
179
|
if (kVal === void 0) return [];
|
|
76
180
|
const k = file.resolve(kVal);
|
|
@@ -2,10 +2,11 @@ import { pt } from "../core/ir/units.js";
|
|
|
2
2
|
import { ResourceStore } from "../core/ir/resources.js";
|
|
3
3
|
import { FEATURES } from "../core/ir/features.js";
|
|
4
4
|
import { collectEmbeddedFonts } from "./embedded-fonts.js";
|
|
5
|
+
import { collectFaceFamilies } from "./font.js";
|
|
5
6
|
import { extractPageText } from "./text.js";
|
|
6
7
|
import { collectPageVectors } from "./vector.js";
|
|
7
8
|
import { displayOf, placeRuns, placeVectors } from "./display.js";
|
|
8
|
-
import { buildFlowDoc, dedupeLosses, imageBlock, paragraphBlock, paragraphFromRuns, sectionFromPdfPages, shapeBlock, withMeasuredMargins } from "./flow-build.js";
|
|
9
|
+
import { buildFlowDoc, dedupeLosses, imageBlock, paragraphBlock, paragraphFromRuns, sectionFromPdfPages, shapeBlock, spaceAfter, withMeasuredMargins } from "./flow-build.js";
|
|
9
10
|
import { collectPageImages } from "./images.js";
|
|
10
11
|
import { markDrawnRules } from "./text-rules.js";
|
|
11
12
|
import { endedParagraph } from "./layout.js";
|
|
@@ -78,10 +79,14 @@ function reconstructTaggedPdf(file) {
|
|
|
78
79
|
const textOf = (node) => squash(node.mcids.map(({ page, mcid }) => runsOfMcid(page, mcid).map((r) => r.text).join("")).join(" "));
|
|
79
80
|
const spansOf = (node) => {
|
|
80
81
|
const spans = [];
|
|
81
|
-
|
|
82
|
-
|
|
83
|
-
|
|
84
|
-
|
|
82
|
+
let last;
|
|
83
|
+
for (const { page, mcid } of node.mcids) {
|
|
84
|
+
const runs = runsOfMcid(page, mcid);
|
|
85
|
+
const first = runs[0];
|
|
86
|
+
if (last !== void 0 && first !== void 0 && spacedApart(last, first)) spans.push(spaceAfter(spans[spans.length - 1]));
|
|
87
|
+
for (const run of runs) spans.push(spanOf(run));
|
|
88
|
+
last = runs[runs.length - 1] ?? last;
|
|
89
|
+
}
|
|
85
90
|
return spans;
|
|
86
91
|
};
|
|
87
92
|
/** One run as the span that carries everything the page showed it with. */
|
|
@@ -156,8 +161,10 @@ function reconstructTaggedPdf(file) {
|
|
|
156
161
|
groups[groups.length - 1].push(line);
|
|
157
162
|
prev = line;
|
|
158
163
|
}
|
|
164
|
+
const ends = (spans) => /\s$/u.test(spans.at(-1)?.text ?? "");
|
|
165
|
+
const opens = (spans) => /^\s/u.test(spans[0]?.text ?? "");
|
|
159
166
|
return groups.map((g) => ({
|
|
160
|
-
spans: g.flatMap((l, i) => i > 0
|
|
167
|
+
spans: g.flatMap((l, i) => i > 0 && !ends(g[i - 1].spans) && !opens(l.spans) ? [spaceAfter(g[i - 1].spans.at(-1)), ...l.spans] : [...l.spans]),
|
|
161
168
|
set: {
|
|
162
169
|
top: g[0].y,
|
|
163
170
|
bottom: g[g.length - 1].y,
|
|
@@ -323,7 +330,7 @@ function reconstructTaggedPdf(file) {
|
|
|
323
330
|
const reached = [...claimed].reduce((n, r) => n + r.text.length, 0);
|
|
324
331
|
if (onPage > 0 && reached * 2 < onPage) return void 0;
|
|
325
332
|
return {
|
|
326
|
-
doc: buildFlowDoc(body, resources, withMeasuredMargins(sectionFromPdfPages(pages), shown, placedRuns, pageImages.map((p) => p.images)), collectEmbeddedFonts(file, pages, imageLosses)),
|
|
333
|
+
doc: buildFlowDoc(body, resources, withMeasuredMargins(sectionFromPdfPages(pages), shown, placedRuns, pageImages.map((p) => p.images)), collectEmbeddedFonts(file, pages, imageLosses), [], void 0, collectFaceFamilies(file, pages)),
|
|
327
334
|
losses: imageLosses
|
|
328
335
|
};
|
|
329
336
|
}
|
|
@@ -414,6 +421,19 @@ function equalGrid(raw) {
|
|
|
414
421
|
grid: Array.from({ length: numCols }, () => w)
|
|
415
422
|
};
|
|
416
423
|
}
|
|
424
|
+
/**
|
|
425
|
+
* Whether the page shows a space between one run and the next: the second
|
|
426
|
+
* starts another line, or stands clear of the first by more than a kern —
|
|
427
|
+
* and neither already carries the space.
|
|
428
|
+
*/
|
|
429
|
+
function spacedApart(prev, next) {
|
|
430
|
+
if (/\s$/u.test(prev.text) || /^\s/u.test(next.text)) return false;
|
|
431
|
+
const size = Math.max(prev.fontSizePt, next.fontSizePt, 1);
|
|
432
|
+
if (Math.abs(prev.y - next.y) > size * .5) return true;
|
|
433
|
+
return next.x - prev.endX > size * MCID_SPACE_EM;
|
|
434
|
+
}
|
|
435
|
+
/** A gap between two stretches of marked content this wide, in ems, is a word space. */
|
|
436
|
+
var MCID_SPACE_EM = .15;
|
|
417
437
|
function squash(text) {
|
|
418
438
|
return text.replace(/\s+/g, " ").trim();
|
|
419
439
|
}
|
|
@@ -95,7 +95,7 @@ var SAME_HEIGHT_PT = .5;
|
|
|
95
95
|
* separate table rules is not.
|
|
96
96
|
*/
|
|
97
97
|
function joinRules(vectors) {
|
|
98
|
-
const rules = vectors.filter((v) => isRule(v));
|
|
98
|
+
const rules = vectors.filter((v) => v.glyph !== true && isRule(v));
|
|
99
99
|
const byRow = /* @__PURE__ */ new Map();
|
|
100
100
|
for (const v of rules) {
|
|
101
101
|
const mid = (v.minY + v.maxY) / 2;
|
|
@@ -2,10 +2,10 @@ import { PDF_NULL, PdfName, PdfStream } from "../pdf/objects.js";
|
|
|
2
2
|
import { buildColorSpaceMap, buildShadingMap } from "./shading.js";
|
|
3
3
|
import { IDENTITY, interpretContent, multiply } from "./content.js";
|
|
4
4
|
import { textMarkupOf } from "./annot-draw.js";
|
|
5
|
-
import { collectPageAppearances } from "./annots.js";
|
|
6
|
-
import { hiddenProperties, hiddenXObject } from "./optional-content.js";
|
|
5
|
+
import { appearanceContent, collectPageAppearances } from "./annots.js";
|
|
7
6
|
import { buildContentFont } from "./font.js";
|
|
8
|
-
import {
|
|
7
|
+
import { hiddenProperties, hiddenXObject } from "./optional-content.js";
|
|
8
|
+
import { patternTint, tintedHex } from "./pattern-tint.js";
|
|
9
9
|
//#region src/pdf-reader/text.ts
|
|
10
10
|
var MAX_FORM_DEPTH = 8;
|
|
11
11
|
/**
|
|
@@ -22,7 +22,12 @@ var MAX_FORM_DEPTH = 8;
|
|
|
22
22
|
function extractPageText(file, page) {
|
|
23
23
|
const runs = [];
|
|
24
24
|
collectRuns(file, page.resources, file.pageContent(page), IDENTITY, 0, /* @__PURE__ */ new Set(), runs);
|
|
25
|
-
|
|
25
|
+
const own = runs.length;
|
|
26
|
+
for (const appearance of collectPageAppearances(file, page)) collectRuns(file, appearance.resources ?? page.resources, appearanceContent(file, appearance), appearance.ctm, 1, new Set([appearance.stream]), runs);
|
|
27
|
+
for (let i = own; i < runs.length; i++) runs[i] = {
|
|
28
|
+
...runs[i],
|
|
29
|
+
annotation: true
|
|
30
|
+
};
|
|
26
31
|
const links = collectLinks(file, page);
|
|
27
32
|
const marks = collectTextMarkup(file, page);
|
|
28
33
|
const shown = withoutRestrikes(runs);
|
|
@@ -244,16 +249,6 @@ function withPatternColour(file, resources, run, visiting) {
|
|
|
244
249
|
visiting.delete(stream);
|
|
245
250
|
}
|
|
246
251
|
}
|
|
247
|
-
/** A colour laid over white paper at `coverage` strength, as a 6-hex string. */
|
|
248
|
-
function tintedHex(colorHex, coverage) {
|
|
249
|
-
const k = Math.min(1, Math.max(0, coverage));
|
|
250
|
-
if (k >= 1) return colorHex;
|
|
251
|
-
const channel = (at) => {
|
|
252
|
-
const c = Number.parseInt(colorHex.slice(at, at + 2), 16);
|
|
253
|
-
return Math.round(255 - (255 - (Number.isFinite(c) ? c : 0)) * k).toString(16).toUpperCase().padStart(2, "0");
|
|
254
|
-
};
|
|
255
|
-
return `${channel(0)}${channel(2)}${channel(4)}`;
|
|
256
|
-
}
|
|
257
252
|
/**
|
|
258
253
|
* The `/Font` resources of one dictionary, built into interpreter fonts. Shared
|
|
259
254
|
* with the path and picture walks, which need them for one thing only: a Type 3
|
|
@@ -1,5 +1,5 @@
|
|
|
1
|
+
import { PageGradient } from './shading.js';
|
|
1
2
|
import { PathSeg } from './content.js';
|
|
2
|
-
import { ShapeGradient } from '../core/vector.js';
|
|
3
3
|
import { Loss } from '../core/ir/index.js';
|
|
4
4
|
import { PdfFile, PdfPage } from './document.js';
|
|
5
5
|
/**
|
|
@@ -21,7 +21,7 @@ export interface PdfVector {
|
|
|
21
21
|
/** Present iff a qualifying solid fill survived (EP10). */
|
|
22
22
|
readonly fillHex?: string;
|
|
23
23
|
/** Present iff a shading-pattern fill survived (EP16c). */
|
|
24
|
-
readonly gradient?:
|
|
24
|
+
readonly gradient?: PageGradient;
|
|
25
25
|
/** §11.6.4.4 — how opaque the fill is, when the page asked for less than all. */
|
|
26
26
|
readonly alpha?: number;
|
|
27
27
|
/**
|
|
@@ -33,6 +33,12 @@ export interface PdfVector {
|
|
|
33
33
|
readonly strokeHex?: string;
|
|
34
34
|
/** Stroke width in page-space points (EP11). */
|
|
35
35
|
readonly lineWidth?: number;
|
|
36
|
+
/** §8.4.3.6 — the stroke's dash lengths in page-space points, where it is dashed. */
|
|
37
|
+
readonly dash?: ReadonlyArray<number>;
|
|
38
|
+
/** §8.4.3.3 — the stroke's cap, where it is not the butt cap. */
|
|
39
|
+
readonly cap?: 'round' | 'square';
|
|
40
|
+
/** §11.6.4.4 `/CA` — how opaque the stroke is, where less than all. */
|
|
41
|
+
readonly strokeAlpha?: number;
|
|
36
42
|
readonly minX: number;
|
|
37
43
|
readonly minY: number;
|
|
38
44
|
readonly maxX: number;
|
|
@@ -2,9 +2,10 @@ import { PDF_NULL, PdfName, PdfStream } from "../pdf/objects.js";
|
|
|
2
2
|
import { FEATURES } from "../core/ir/features.js";
|
|
3
3
|
import { buildAlphaMap, buildColorSpaceMap, buildShadingMap, gradientShading, shadingTypeOf } from "./shading.js";
|
|
4
4
|
import { IDENTITY, interpretContent, multiply } from "./content.js";
|
|
5
|
-
import { collectPageAppearances } from "./annots.js";
|
|
5
|
+
import { appearanceContent, collectPageAppearances } from "./annots.js";
|
|
6
6
|
import { hiddenProperties, hiddenXObject } from "./optional-content.js";
|
|
7
7
|
import { buildFonts } from "./text.js";
|
|
8
|
+
import { patternTint, tintedHex } from "./pattern-tint.js";
|
|
8
9
|
//#region src/pdf-reader/vector.ts
|
|
9
10
|
function paintedVectors(file, page) {
|
|
10
11
|
const out = [];
|
|
@@ -22,6 +23,31 @@ function paintedVectors(file, page) {
|
|
|
22
23
|
stateCache.set(resources, made);
|
|
23
24
|
return made;
|
|
24
25
|
};
|
|
26
|
+
const tints = /* @__PURE__ */ new Map();
|
|
27
|
+
const tinted = (resources, vector) => {
|
|
28
|
+
const name = vector.patternName;
|
|
29
|
+
if (name === void 0) return vector;
|
|
30
|
+
let known = tints.get(resources);
|
|
31
|
+
if (!known) tints.set(resources, known = /* @__PURE__ */ new Map());
|
|
32
|
+
const key = `${name}\u0000${vector.patternPaint ?? ""}`;
|
|
33
|
+
let hex = known.get(key);
|
|
34
|
+
if (!known.has(key)) {
|
|
35
|
+
let tint;
|
|
36
|
+
try {
|
|
37
|
+
tint = patternTint(file, resources, name, vector.patternPaint);
|
|
38
|
+
} catch {
|
|
39
|
+
tint = void 0;
|
|
40
|
+
}
|
|
41
|
+
hex = tint ? tintedHex(tint.colorHex, tint.coverage) : void 0;
|
|
42
|
+
known.set(key, hex);
|
|
43
|
+
}
|
|
44
|
+
if (hex === void 0) return vector;
|
|
45
|
+
const { patternName: _named, patternPaint: _paint, ...plain } = vector;
|
|
46
|
+
return {
|
|
47
|
+
...plain,
|
|
48
|
+
fillHex: hex
|
|
49
|
+
};
|
|
50
|
+
};
|
|
25
51
|
const walk = (resources, content, baseCtm, depth, prefix) => {
|
|
26
52
|
if (out.length >= MAX_VECTORS) return;
|
|
27
53
|
const xobjects = resources ? file.get(resources, "XObject") : PDF_NULL;
|
|
@@ -34,7 +60,7 @@ function paintedVectors(file, page) {
|
|
|
34
60
|
const found = dict instanceof Map ? file.resolve(dict.get(paint.name) ?? PDF_NULL) : PDF_NULL;
|
|
35
61
|
const sh = found instanceof PdfStream ? found.dict : found instanceof Map ? found : void 0;
|
|
36
62
|
if (sh && shadingTypeOf(file, sh) === 1) continue;
|
|
37
|
-
const gradient = sh ? gradientShading(file, sh) : void 0;
|
|
63
|
+
const gradient = sh ? gradientShading(file, sh, paint.ctm) : void 0;
|
|
38
64
|
if (paint.masked) {
|
|
39
65
|
bareShadings++;
|
|
40
66
|
continue;
|
|
@@ -74,7 +100,7 @@ function paintedVectors(file, page) {
|
|
|
74
100
|
if (out.length >= MAX_VECTORS) return;
|
|
75
101
|
if (event.vector) {
|
|
76
102
|
out.push({
|
|
77
|
-
...event.vector,
|
|
103
|
+
...tinted(resources, event.vector),
|
|
78
104
|
orderKey: [...prefix, event.order]
|
|
79
105
|
});
|
|
80
106
|
continue;
|
|
@@ -102,7 +128,7 @@ function paintedVectors(file, page) {
|
|
|
102
128
|
};
|
|
103
129
|
walk(page.resources, file.pageContent(page), IDENTITY, 0, []);
|
|
104
130
|
collectPageAppearances(file, page).forEach((appearance, index) => {
|
|
105
|
-
walk(appearance.resources ?? page.resources, file
|
|
131
|
+
walk(appearance.resources ?? page.resources, appearanceContent(file, appearance), appearance.ctm, 1, [Number.MAX_SAFE_INTEGER, index]);
|
|
106
132
|
});
|
|
107
133
|
return {
|
|
108
134
|
placements: out,
|
|
@@ -196,7 +222,10 @@ function collectPageVectors(file, page, occupied = []) {
|
|
|
196
222
|
...filled && v.darkens === true ? { darkens: true } : {},
|
|
197
223
|
...stroked ? {
|
|
198
224
|
strokeHex: v.strokeHex,
|
|
199
|
-
...v.lineWidth !== void 0 ? { lineWidth: v.lineWidth } : {}
|
|
225
|
+
...v.lineWidth !== void 0 ? { lineWidth: v.lineWidth } : {},
|
|
226
|
+
...v.dash !== void 0 ? { dash: v.dash } : {},
|
|
227
|
+
...v.cap !== void 0 ? { cap: v.cap } : {},
|
|
228
|
+
...v.strokeAlpha !== void 0 ? { strokeAlpha: v.strokeAlpha } : {}
|
|
200
229
|
} : {},
|
|
201
230
|
...b,
|
|
202
231
|
...v.glyph === true ? { glyph: true } : {},
|