reamkit 1.29.0 → 1.31.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +60 -24
- package/dist/esm/core/converter/project.js +3 -1
- package/dist/esm/core/crypto/offcrypto.js +1 -1
- package/dist/esm/core/document-model/index.d.ts +1 -1
- package/dist/esm/core/document-model/types.d.ts +65 -0
- package/dist/esm/core/drawingml/shape-render.js +13 -1
- package/dist/esm/core/font/index.d.ts +2 -0
- package/dist/esm/core/font/ttf-build.d.ts +113 -0
- package/dist/esm/core/font/ttf-build.js +1224 -0
- package/dist/esm/core/font/ttf-subset.d.ts +19 -0
- package/dist/esm/core/font/ttf-subset.js +14 -1
- package/dist/esm/core/fonts/index.d.ts +1 -1
- package/dist/esm/core/fonts/provider.d.ts +17 -0
- package/dist/esm/core/fonts/provider.js +27 -2
- package/dist/esm/core/fonts/remote-fonts.d.ts +8 -0
- package/dist/esm/core/fonts/remote-fonts.js +108 -18
- package/dist/esm/core/ir/flow.d.ts +88 -1
- package/dist/esm/core/numbering/index.d.ts +1 -1
- package/dist/esm/core/numbering/state.d.ts +11 -1
- package/dist/esm/core/numbering/state.js +10 -1
- package/dist/esm/core/style-cascade/resolver.js +35 -4
- package/dist/esm/core/style-cascade/types.d.ts +19 -1
- package/dist/esm/core/style-cascade/types.js +3 -0
- package/dist/esm/index.d.ts +2 -2
- package/dist/esm/layout/page-doc.d.ts +10 -3
- package/dist/esm/layout/styled-layout.d.ts +48 -7
- package/dist/esm/layout/styled-layout.js +881 -119
- package/dist/esm/layout/turned-section.d.ts +38 -0
- package/dist/esm/layout/turned-section.js +193 -0
- package/dist/esm/pdf/styled-page-emitter.js +78 -2
- package/dist/esm/pdf-reader/annot-draw.js +93 -1
- package/dist/esm/pdf-reader/annots.d.ts +18 -0
- package/dist/esm/pdf-reader/annots.js +86 -9
- package/dist/esm/pdf-reader/cff-outline.d.ts +27 -0
- package/dist/esm/pdf-reader/cff-outline.js +169 -21
- package/dist/esm/pdf-reader/cmap.js +5 -2
- package/dist/esm/pdf-reader/content.d.ts +70 -4
- package/dist/esm/pdf-reader/content.js +172 -12
- package/dist/esm/pdf-reader/display.d.ts +36 -0
- package/dist/esm/pdf-reader/display.js +82 -1
- package/dist/esm/pdf-reader/document.js +5 -1
- package/dist/esm/pdf-reader/embedded-fonts.d.ts +25 -0
- package/dist/esm/pdf-reader/embedded-fonts.js +78 -9
- package/dist/esm/pdf-reader/encodings.d.ts +8 -0
- package/dist/esm/pdf-reader/encodings.js +25 -3
- package/dist/esm/pdf-reader/face-outlines.d.ts +78 -0
- package/dist/esm/pdf-reader/face-outlines.js +362 -0
- package/dist/esm/pdf-reader/figures.d.ts +52 -0
- package/dist/esm/pdf-reader/figures.js +433 -0
- package/dist/esm/pdf-reader/flow-build.d.ts +224 -7
- package/dist/esm/pdf-reader/flow-build.js +545 -39
- package/dist/esm/pdf-reader/font.d.ts +27 -1
- package/dist/esm/pdf-reader/font.js +632 -63
- package/dist/esm/pdf-reader/glyf-outline.d.ts +33 -0
- package/dist/esm/pdf-reader/glyf-outline.js +148 -3
- package/dist/esm/pdf-reader/glyph-names.js +154 -1
- package/dist/esm/pdf-reader/glyph-shapes.d.ts +18 -0
- package/dist/esm/pdf-reader/glyph-shapes.js +57 -0
- package/dist/esm/pdf-reader/image-decode.js +70 -4
- package/dist/esm/pdf-reader/images.d.ts +5 -0
- package/dist/esm/pdf-reader/images.js +4 -2
- package/dist/esm/pdf-reader/jbig2.d.ts +40 -1
- package/dist/esm/pdf-reader/jbig2.js +78 -16
- package/dist/esm/pdf-reader/jpeg.d.ts +6 -3
- package/dist/esm/pdf-reader/jpeg.js +21 -1
- package/dist/esm/pdf-reader/layout.d.ts +119 -2
- package/dist/esm/pdf-reader/layout.js +2219 -184
- package/dist/esm/pdf-reader/lexer.d.ts +10 -0
- package/dist/esm/pdf-reader/lexer.js +17 -0
- package/dist/esm/pdf-reader/page-numbers.d.ts +53 -0
- package/dist/esm/pdf-reader/page-numbers.js +167 -0
- package/dist/esm/pdf-reader/pattern-tint.d.ts +11 -1
- package/dist/esm/pdf-reader/pattern-tint.js +21 -3
- package/dist/esm/pdf-reader/regions.d.ts +25 -0
- package/dist/esm/pdf-reader/regions.js +167 -0
- package/dist/esm/pdf-reader/shading.d.ts +58 -2
- package/dist/esm/pdf-reader/shading.js +181 -11
- package/dist/esm/pdf-reader/struct-tree.js +112 -8
- package/dist/esm/pdf-reader/tagged.js +207 -28
- package/dist/esm/pdf-reader/text-rules.js +1 -1
- package/dist/esm/pdf-reader/text.d.ts +6 -3
- package/dist/esm/pdf-reader/text.js +106 -22
- package/dist/esm/pdf-reader/type1-outline.d.ts +11 -0
- package/dist/esm/pdf-reader/type1-outline.js +63 -8
- package/dist/esm/pdf-reader/vector.d.ts +8 -2
- package/dist/esm/pdf-reader/vector.js +105 -6
- package/dist/esm/word/doc/doc-reader.js +6 -2
- package/dist/esm/word/doc/doc-text.d.ts +6 -0
- package/dist/esm/word/doc/doc-text.js +19 -1
- package/dist/esm/word/document-parser.d.ts +2 -2
- package/dist/esm/word/document-parser.js +11 -1
- package/dist/esm/word/docx-reader.js +5 -3
- package/dist/esm/word/docx-writer.js +341 -54
- package/dist/esm/word/drawing-parser.d.ts +5 -3
- package/dist/esm/word/drawing-parser.js +56 -10
- package/dist/esm/word/font-embed.d.ts +30 -0
- package/dist/esm/word/font-embed.js +173 -0
- package/dist/esm/word/font-table.d.ts +10 -0
- package/dist/esm/word/font-table.js +13 -1
- package/dist/esm/word/index.js +1 -1
- package/dist/esm/word/numbering-parser.d.ts +3 -1
- package/dist/esm/word/numbering-parser.js +2 -1
- package/dist/esm/word/paragraph-properties.d.ts +7 -6
- package/dist/esm/word/paragraph-properties.js +16 -2
- package/dist/esm/word/run-properties.js +26 -0
- package/package.json +12 -5
|
@@ -2,13 +2,15 @@ import { pt } from "../core/ir/units.js";
|
|
|
2
2
|
import { ResourceStore } from "../core/ir/resources.js";
|
|
3
3
|
import { FEATURES } from "../core/ir/features.js";
|
|
4
4
|
import { collectEmbeddedFonts } from "./embedded-fonts.js";
|
|
5
|
+
import { collectFaceFamilies } from "./font.js";
|
|
6
|
+
import { displayOf, placeRuns, placeVectors, textFrameOf, wordsTurnOf } from "./display.js";
|
|
7
|
+
import { ASCENDER, CARRIER_LINE_PT, buildFlowDoc, dedupeLosses, floatOntoSheet, imageBlock, paragraphBlock, paragraphFromRuns, sectionFromPdfPages, sectionOnSheet, shapeBlock, spaceAfter, withMeasuredMargins } from "./flow-build.js";
|
|
8
|
+
import { faceOutlinesOf, kernedFaces, pageSpacing } from "./face-outlines.js";
|
|
5
9
|
import { extractPageText } from "./text.js";
|
|
6
10
|
import { collectPageVectors } from "./vector.js";
|
|
7
|
-
import { displayOf, placeRuns, placeVectors } from "./display.js";
|
|
8
|
-
import { buildFlowDoc, dedupeLosses, imageBlock, paragraphBlock, paragraphFromRuns, sectionFromPdfPages, shapeBlock, withMeasuredMargins } from "./flow-build.js";
|
|
9
11
|
import { collectPageImages } from "./images.js";
|
|
10
12
|
import { markDrawnRules } from "./text-rules.js";
|
|
11
|
-
import { endedParagraph } from "./layout.js";
|
|
13
|
+
import { FOOTER_PART, HEADER_PART, endedParagraph, footerBand, numberedFrom, numberingOf, pageTextEdges, runningFoot, stepsBetweenWords } from "./layout.js";
|
|
12
14
|
import { readStructTree } from "./struct-tree.js";
|
|
13
15
|
//#region src/pdf-reader/tagged.ts
|
|
14
16
|
var ASSUMED_CONTENT_WIDTH_PT = 468;
|
|
@@ -30,8 +32,15 @@ function reconstructTaggedPdf(file) {
|
|
|
30
32
|
const root = readStructTree(file);
|
|
31
33
|
if (!root) return void 0;
|
|
32
34
|
const pages = file.pages();
|
|
33
|
-
const
|
|
34
|
-
const
|
|
35
|
+
const sheets = pages.map((page) => displayOf(page));
|
|
36
|
+
const painted = /* @__PURE__ */ new Map();
|
|
37
|
+
const extracted = pages.map((page) => extractPageText(file, page, painted));
|
|
38
|
+
const spacing = pageSpacing(extracted);
|
|
39
|
+
const onSheets = extracted.map((runs, i) => placeRuns(runs, sheets[i]));
|
|
40
|
+
const worded = onSheets.filter((runs) => runs.some((r) => r.text.trim() !== ""));
|
|
41
|
+
const turned = worded.length > 0 && worded.every((runs) => wordsTurnOf(runs) === 270);
|
|
42
|
+
const shown = turned ? sheets.map((sheet) => textFrameOf(sheet)) : sheets;
|
|
43
|
+
const placedRuns = turned ? extracted.map((runs, i) => placeRuns(runs, shown[i])) : onSheets;
|
|
35
44
|
const vectorLosses = [];
|
|
36
45
|
const pageVectors = pages.map((page, i) => {
|
|
37
46
|
const lifted = collectPageVectors(file, page, collectPageImages(file, page).images.map((img) => ({
|
|
@@ -78,10 +87,14 @@ function reconstructTaggedPdf(file) {
|
|
|
78
87
|
const textOf = (node) => squash(node.mcids.map(({ page, mcid }) => runsOfMcid(page, mcid).map((r) => r.text).join("")).join(" "));
|
|
79
88
|
const spansOf = (node) => {
|
|
80
89
|
const spans = [];
|
|
81
|
-
|
|
82
|
-
|
|
83
|
-
|
|
84
|
-
|
|
90
|
+
let last;
|
|
91
|
+
for (const { page, mcid } of node.mcids) {
|
|
92
|
+
const runs = runsOfMcid(page, mcid);
|
|
93
|
+
const first = runs[0];
|
|
94
|
+
if (last !== void 0 && first !== void 0 && spacedApart(last, first)) spans.push(spaceAfter(spans[spans.length - 1]));
|
|
95
|
+
for (const run of runs) spans.push(spanOf(run));
|
|
96
|
+
last = runs[runs.length - 1] ?? last;
|
|
97
|
+
}
|
|
85
98
|
return spans;
|
|
86
99
|
};
|
|
87
100
|
/** One run as the span that carries everything the page showed it with. */
|
|
@@ -101,6 +114,23 @@ function reconstructTaggedPdf(file) {
|
|
|
101
114
|
});
|
|
102
115
|
const collectText = (node) => squash([textOf(node), ...node.children.map(collectText)].join(" "));
|
|
103
116
|
const setting = /* @__PURE__ */ new Map();
|
|
117
|
+
const pageOf = /* @__PURE__ */ new Map();
|
|
118
|
+
/** The highest baseline a page's own paragraphs stand on, page space (y up). */
|
|
119
|
+
const highestOf = (own) => {
|
|
120
|
+
const tops = own.flatMap((el) => {
|
|
121
|
+
const set = setting.get(el);
|
|
122
|
+
return set ? [set.top] : [];
|
|
123
|
+
});
|
|
124
|
+
return tops.length > 0 ? Math.max(...tops) : void 0;
|
|
125
|
+
};
|
|
126
|
+
/** The lowest baseline a page's own paragraphs reach, page space (y up). */
|
|
127
|
+
const lowestOf = (own) => {
|
|
128
|
+
const bottoms = own.flatMap((el) => {
|
|
129
|
+
const set = setting.get(el);
|
|
130
|
+
return set ? [set.bottom] : [];
|
|
131
|
+
});
|
|
132
|
+
return bottoms.length > 0 ? Math.min(...bottoms) : void 0;
|
|
133
|
+
};
|
|
104
134
|
/** The topmost and bottommost baseline under a node, and its largest face. */
|
|
105
135
|
function baselinesOf(node) {
|
|
106
136
|
let top;
|
|
@@ -156,8 +186,10 @@ function reconstructTaggedPdf(file) {
|
|
|
156
186
|
groups[groups.length - 1].push(line);
|
|
157
187
|
prev = line;
|
|
158
188
|
}
|
|
189
|
+
const ends = (spans) => /\s$/u.test(spans.at(-1)?.text ?? "");
|
|
190
|
+
const opens = (spans) => /^\s/u.test(spans[0]?.text ?? "");
|
|
159
191
|
return groups.map((g) => ({
|
|
160
|
-
spans: g.flatMap((l, i) => i > 0
|
|
192
|
+
spans: g.flatMap((l, i) => i > 0 && !ends(g[i - 1].spans) && !opens(l.spans) ? [spaceAfter(g[i - 1].spans.at(-1)), ...l.spans] : [...l.spans]),
|
|
161
193
|
set: {
|
|
162
194
|
top: g[0].y,
|
|
163
195
|
bottom: g[g.length - 1].y,
|
|
@@ -198,33 +230,48 @@ function reconstructTaggedPdf(file) {
|
|
|
198
230
|
}).filter((l) => l.spans.some((sp) => sp.text.trim().length > 0));
|
|
199
231
|
}
|
|
200
232
|
function emit(node, out) {
|
|
233
|
+
const on = (el) => {
|
|
234
|
+
const page = firstPageOf(node);
|
|
235
|
+
if (page !== void 0) pageOf.set(el, page);
|
|
236
|
+
return el;
|
|
237
|
+
};
|
|
201
238
|
if (node.type === "Table") {
|
|
202
239
|
const table = buildTable(node);
|
|
203
|
-
if (table) out.push(table);
|
|
240
|
+
if (table) out.push(on(table));
|
|
204
241
|
return;
|
|
205
242
|
}
|
|
206
243
|
if (node.type === "Figure") {
|
|
207
244
|
for (const img of imagesForNode(node)) {
|
|
208
245
|
emitted.add(img);
|
|
209
|
-
out.push(imageBlock(img, resources, node.alt));
|
|
246
|
+
out.push(on(imageBlock(img, resources, node.alt)));
|
|
210
247
|
}
|
|
211
248
|
return;
|
|
212
249
|
}
|
|
213
250
|
if (node.type === "LI") {
|
|
214
251
|
const text = collectText(node);
|
|
215
|
-
if (text.length > 0) out.push(paragraphBlock(text, void 0));
|
|
252
|
+
if (text.length > 0) out.push(on(paragraphBlock(text, void 0)));
|
|
216
253
|
return;
|
|
217
254
|
}
|
|
218
255
|
if (node.children.length === 0) {
|
|
219
256
|
if (textOf(node).length > 0) for (const part of settingsOf(node)) {
|
|
220
257
|
const el = paragraphFromRuns(part.spans, headingLevel(node.type));
|
|
221
258
|
setting.set(el, part.set);
|
|
222
|
-
out.push(el);
|
|
259
|
+
out.push(on(el));
|
|
223
260
|
}
|
|
224
261
|
return;
|
|
225
262
|
}
|
|
226
263
|
for (const child of node.children) emit(child, out);
|
|
227
264
|
}
|
|
265
|
+
/** The first page any marked content under a node stands on. */
|
|
266
|
+
function firstPageOf(node) {
|
|
267
|
+
let first;
|
|
268
|
+
const visit = (n) => {
|
|
269
|
+
for (const { page } of n.mcids) if (first === void 0 || page < first) first = page;
|
|
270
|
+
for (const child of n.children) visit(child);
|
|
271
|
+
};
|
|
272
|
+
visit(node);
|
|
273
|
+
return first;
|
|
274
|
+
}
|
|
228
275
|
function buildTable(tableNode) {
|
|
229
276
|
const raw = [];
|
|
230
277
|
const collectRows = (n) => {
|
|
@@ -288,9 +335,14 @@ function reconstructTaggedPdf(file) {
|
|
|
288
335
|
cells
|
|
289
336
|
};
|
|
290
337
|
}
|
|
291
|
-
const
|
|
292
|
-
emit(root,
|
|
293
|
-
|
|
338
|
+
const named = [];
|
|
339
|
+
emit(root, named);
|
|
340
|
+
const byPage = pages.map(() => []);
|
|
341
|
+
let current = 0;
|
|
342
|
+
for (const el of named) {
|
|
343
|
+
current = Math.max(current, pageOf.get(el) ?? current);
|
|
344
|
+
byPage[current].push(el);
|
|
345
|
+
}
|
|
294
346
|
let zOrder = -1e6;
|
|
295
347
|
imageLosses.push(...vectorLosses);
|
|
296
348
|
pages.forEach((_page, index) => {
|
|
@@ -299,31 +351,104 @@ function reconstructTaggedPdf(file) {
|
|
|
299
351
|
top: shown[index].height
|
|
300
352
|
};
|
|
301
353
|
const taken = ruled[index]?.consumed;
|
|
354
|
+
const drawn = [];
|
|
302
355
|
for (const v of pageVectors[index] ?? []) {
|
|
303
356
|
if (taken?.has(v) === true) continue;
|
|
304
|
-
|
|
357
|
+
const shape = shapeBlock(v, frame, zOrder++, true);
|
|
358
|
+
drawn.push(turned ? floatOntoSheet(shape, shown[index].height) : shape);
|
|
305
359
|
}
|
|
360
|
+
byPage[index].unshift(...drawn);
|
|
306
361
|
});
|
|
307
|
-
const orphans = [];
|
|
308
362
|
pageImages.forEach((p, page) => {
|
|
309
|
-
|
|
310
|
-
|
|
311
|
-
|
|
312
|
-
|
|
363
|
+
const left = p.images.filter((img) => !emitted.has(img)).sort((a, b) => b.y - a.y);
|
|
364
|
+
byPage[page].push(...left.map((img) => imageBlock(img, resources)));
|
|
365
|
+
});
|
|
366
|
+
if (byPage.every((own) => own.length === 0)) return void 0;
|
|
367
|
+
const loose = placedRuns.map((runs, i) => shown[i].sheet ? [] : runs.filter((r) => !claimed.has(r)));
|
|
368
|
+
const foot = runningFoot(loose, shown, "foot");
|
|
369
|
+
const head = runningFoot(loose, shown, "head");
|
|
370
|
+
const numberedBand = foot?.numbered === true ? foot : head?.numbered === true ? head : void 0;
|
|
371
|
+
const numbering = numberingOf(numberedBand);
|
|
372
|
+
const lifted = (r, i) => foot?.lift[i]?.has(r) === true || head?.lift[i]?.has(r) === true;
|
|
373
|
+
const measured = withMeasuredMargins(sectionFromPdfPages(pages, shown[0]), shown, placedRuns.map((runs, i) => runs.filter((r) => !lifted(r, i))), pageImages.map((p) => p.images), foot?.band);
|
|
374
|
+
const body = [];
|
|
375
|
+
const top = measured?.margins?.top ?? 0;
|
|
376
|
+
const bottom = measured?.margins?.bottom ?? 0;
|
|
377
|
+
const ends = byPage.map((own) => lowestOf(own));
|
|
378
|
+
const tops = byPage.map((own) => highestOf(own));
|
|
379
|
+
const sectionsFrom = (numbering?.runs ?? []).map((run) => run.from).filter((from) => from > 0);
|
|
380
|
+
const endsAt = [];
|
|
381
|
+
let sectionFrom = 0;
|
|
382
|
+
byPage.forEach((own, index) => {
|
|
383
|
+
if (sectionsFrom.includes(index)) {
|
|
384
|
+
endsAt.push({
|
|
385
|
+
at: body.length,
|
|
386
|
+
from: sectionFrom
|
|
387
|
+
});
|
|
388
|
+
sectionFrom = index;
|
|
389
|
+
}
|
|
390
|
+
const height = shown[index]?.height ?? 0;
|
|
391
|
+
spaceParagraphs(own, setting, height);
|
|
392
|
+
const ended = ends[index - 1];
|
|
393
|
+
const opens = index > 0 && (own.length === 0 || ended === void 0 || ended - bottom > (height - top - bottom) * SHORT_PAGE_SHARE);
|
|
394
|
+
const lead = own.findIndex((el) => !floats(el));
|
|
395
|
+
const set = lead >= 0 ? setting.get(own[lead]) : void 0;
|
|
396
|
+
const fresh = index === 0 || opens || sectionsFrom.includes(index);
|
|
397
|
+
const highest = tops[index];
|
|
398
|
+
const first = set !== void 0 && highest !== void 0 && set.top >= highest - set.size;
|
|
399
|
+
if (fresh && first && measured?.margins) {
|
|
400
|
+
const gap = height - top - (set.top + set.size * ASCENDER);
|
|
401
|
+
if (gap > 1) own[lead] = spacedBefore(own[lead], gap);
|
|
402
|
+
}
|
|
403
|
+
if (opens && !sectionsFrom.includes(index)) body.push(pageBreak(true));
|
|
404
|
+
else if (index === 0 && own.length === 0 && pages.length > 1) body.push(pageBreak(false));
|
|
405
|
+
body.push(...own);
|
|
313
406
|
});
|
|
314
|
-
orphans.sort((a, b) => a.page - b.page || b.img.y - a.img.y);
|
|
315
|
-
for (const { img } of orphans) body.push(imageBlock(img, resources));
|
|
316
|
-
if (body.length === 0) return void 0;
|
|
317
407
|
if (placedRuns.some((page) => page.some((r) => r.text.includes("�")))) imageLosses.push({
|
|
318
408
|
severity: "dropped",
|
|
319
409
|
feature: FEATURES.text,
|
|
320
410
|
detail: "some glyphs map to no character — the font states no /ToUnicode and its program says nothing either, so that text is unrecoverable"
|
|
321
411
|
});
|
|
322
|
-
const onPage = placedRuns.
|
|
412
|
+
const onPage = placedRuns.flatMap((runs, i) => runs.filter((r) => !lifted(r, i))).reduce((n, r) => n + r.text.length, 0);
|
|
323
413
|
const reached = [...claimed].reduce((n, r) => n + r.text.length, 0);
|
|
324
414
|
if (onPage > 0 && reached * 2 < onPage) return void 0;
|
|
415
|
+
endsAt.push({
|
|
416
|
+
at: body.length,
|
|
417
|
+
from: sectionFrom
|
|
418
|
+
});
|
|
419
|
+
const stepped0 = stepsBetweenWords(placedRuns[0] ?? []);
|
|
420
|
+
const edges0 = pageTextEdges(placedRuns[0] ?? []);
|
|
421
|
+
const numeralOf = (of) => {
|
|
422
|
+
if (of === void 0 || of !== numberedBand || numbering === void 0) return void 0;
|
|
423
|
+
const first = of.lift.findIndex((set) => set.size > 0);
|
|
424
|
+
return first >= 0 ? numbering.numbers[first]?.text : void 0;
|
|
425
|
+
};
|
|
426
|
+
const footBand = foot ? footerBand(foot.band, stepped0, edges0, foot.numbered, numeralOf(foot)) : [];
|
|
427
|
+
const headBand = head ? footerBand(head.band, stepped0, edges0, head.numbered, numeralOf(head)) : [];
|
|
428
|
+
const sectionAt = (from) => {
|
|
429
|
+
const base = sectionOnSheet(measured, shown[0]);
|
|
430
|
+
return base ? {
|
|
431
|
+
...base,
|
|
432
|
+
...numberedFrom(numbering, from),
|
|
433
|
+
...footBand.length > 0 ? { footers: [{
|
|
434
|
+
type: "default",
|
|
435
|
+
relationshipId: FOOTER_PART
|
|
436
|
+
}] } : {},
|
|
437
|
+
...headBand.length > 0 ? { headers: [{
|
|
438
|
+
type: "default",
|
|
439
|
+
relationshipId: HEADER_PART
|
|
440
|
+
}] } : {}
|
|
441
|
+
} : base;
|
|
442
|
+
};
|
|
443
|
+
const sections = endsAt.length > 1 ? endsAt.flatMap((end) => {
|
|
444
|
+
const properties = sectionAt(end.from);
|
|
445
|
+
return properties ? [{
|
|
446
|
+
properties,
|
|
447
|
+
endIndex: end.at
|
|
448
|
+
}] : [];
|
|
449
|
+
}) : [];
|
|
325
450
|
return {
|
|
326
|
-
doc: buildFlowDoc(body, resources,
|
|
451
|
+
doc: buildFlowDoc(body, resources, sectionAt(0), collectEmbeddedFonts(file, pages, imageLosses), sections, footBand.length > 0 || headBand.length > 0 ? new Map([...footBand.length > 0 ? [[FOOTER_PART, footBand]] : [], ...headBand.length > 0 ? [[HEADER_PART, headBand]] : []]) : void 0, collectFaceFamilies(file, pages), faceOutlinesOf(painted, spacing), kernedFaces(spacing)),
|
|
327
452
|
losses: imageLosses
|
|
328
453
|
};
|
|
329
454
|
}
|
|
@@ -414,9 +539,63 @@ function equalGrid(raw) {
|
|
|
414
539
|
grid: Array.from({ length: numCols }, () => w)
|
|
415
540
|
};
|
|
416
541
|
}
|
|
542
|
+
/**
|
|
543
|
+
* Whether the page shows a space between one run and the next: the second
|
|
544
|
+
* starts another line, or stands clear of the first by more than a kern —
|
|
545
|
+
* and neither already carries the space.
|
|
546
|
+
*/
|
|
547
|
+
function spacedApart(prev, next) {
|
|
548
|
+
if (/\s$/u.test(prev.text) || /^\s/u.test(next.text)) return false;
|
|
549
|
+
const size = Math.max(prev.fontSizePt, next.fontSizePt, 1);
|
|
550
|
+
if (Math.abs(prev.y - next.y) > size * .5) return true;
|
|
551
|
+
return next.x - prev.endX > size * MCID_SPACE_EM;
|
|
552
|
+
}
|
|
553
|
+
/** A gap between two stretches of marked content this wide, in ems, is a word space. */
|
|
554
|
+
var MCID_SPACE_EM = .15;
|
|
417
555
|
function squash(text) {
|
|
418
556
|
return text.replace(/\s+/g, " ").trim();
|
|
419
557
|
}
|
|
558
|
+
/**
|
|
559
|
+
* How much of a page's text area the page may leave empty below its last line
|
|
560
|
+
* and still be FULL: a page that ends higher than this ended on purpose.
|
|
561
|
+
*/
|
|
562
|
+
var SHORT_PAGE_SHARE = .25;
|
|
563
|
+
/** Whether an element FLOATS — a drawing anchored to its page, which takes no room. */
|
|
564
|
+
function floats(el) {
|
|
565
|
+
return el.kind === "image" && el.image.float !== void 0 || el.kind === "shape" && el.shape.float !== void 0;
|
|
566
|
+
}
|
|
567
|
+
/** A paragraph given the space the page left above it; anything else as it is. */
|
|
568
|
+
function spacedBefore(el, before) {
|
|
569
|
+
if (el.kind !== "paragraph") return el;
|
|
570
|
+
const properties = {
|
|
571
|
+
...el.paragraph.properties,
|
|
572
|
+
spacingBefore: pt(before)
|
|
573
|
+
};
|
|
574
|
+
return {
|
|
575
|
+
...el,
|
|
576
|
+
paragraph: {
|
|
577
|
+
...el.paragraph,
|
|
578
|
+
properties
|
|
579
|
+
}
|
|
580
|
+
};
|
|
581
|
+
}
|
|
582
|
+
/**
|
|
583
|
+
* An empty paragraph that takes no room: the carrier of a page break, or of
|
|
584
|
+
* nothing at all on a blank first sheet the next page's break has to follow.
|
|
585
|
+
*/
|
|
586
|
+
function pageBreak(breaks) {
|
|
587
|
+
return {
|
|
588
|
+
kind: "paragraph",
|
|
589
|
+
paragraph: {
|
|
590
|
+
properties: {
|
|
591
|
+
...breaks ? { pageBreakBefore: true } : {},
|
|
592
|
+
spacingLine: CARRIER_LINE_PT,
|
|
593
|
+
spacingLineRule: "exact"
|
|
594
|
+
},
|
|
595
|
+
runs: []
|
|
596
|
+
}
|
|
597
|
+
};
|
|
598
|
+
}
|
|
420
599
|
function headingLevel(type) {
|
|
421
600
|
const m = /^H([1-6])$/.exec(type);
|
|
422
601
|
return m ? Number(m[1]) - 1 : void 0;
|
|
@@ -95,7 +95,7 @@ var SAME_HEIGHT_PT = .5;
|
|
|
95
95
|
* separate table rules is not.
|
|
96
96
|
*/
|
|
97
97
|
function joinRules(vectors) {
|
|
98
|
-
const rules = vectors.filter((v) => isRule(v));
|
|
98
|
+
const rules = vectors.filter((v) => v.glyph !== true && isRule(v));
|
|
99
99
|
const byRow = /* @__PURE__ */ new Map();
|
|
100
100
|
for (const v of rules) {
|
|
101
101
|
const mid = (v.minY + v.maxY) / 2;
|
|
@@ -1,4 +1,5 @@
|
|
|
1
1
|
import { ContentFont, TextRun } from './content.js';
|
|
2
|
+
import { ShownCodes } from './face-outlines.js';
|
|
2
3
|
import { PdfDict } from '../pdf/objects.js';
|
|
3
4
|
import { PdfFile, PdfPage } from './document.js';
|
|
4
5
|
/**
|
|
@@ -8,11 +9,13 @@ import { PdfFile, PdfPage } from './document.js';
|
|
|
8
9
|
* origin falls inside a `/Link` annotation's `/Rect` with that link's URI (EP8)
|
|
9
10
|
* so hyperlinks survive.
|
|
10
11
|
*
|
|
11
|
-
* @param file
|
|
12
|
-
* @param page
|
|
12
|
+
* @param file The owning {@link PdfFile}.
|
|
13
|
+
* @param page The page to extract.
|
|
14
|
+
* @param painted Where to gather the codes each font painted, for a caller
|
|
15
|
+
* that embeds the faces (see `./face-outlines`).
|
|
13
16
|
* @returns The page's runs, each carrying an `href` when it sits under a link.
|
|
14
17
|
*/
|
|
15
|
-
export declare function extractPageText(file: PdfFile, page: PdfPage): Array<TextRun>;
|
|
18
|
+
export declare function extractPageText(file: PdfFile, page: PdfPage, painted?: ShownCodes): Array<TextRun>;
|
|
16
19
|
/**
|
|
17
20
|
* The `/Font` resources of one dictionary, built into interpreter fonts. Shared
|
|
18
21
|
* with the path and picture walks, which need them for one thing only: a Type 3
|
|
@@ -2,10 +2,11 @@ import { PDF_NULL, PdfName, PdfStream } from "../pdf/objects.js";
|
|
|
2
2
|
import { buildColorSpaceMap, buildShadingMap } from "./shading.js";
|
|
3
3
|
import { IDENTITY, interpretContent, multiply } from "./content.js";
|
|
4
4
|
import { textMarkupOf } from "./annot-draw.js";
|
|
5
|
-
import { collectPageAppearances } from "./annots.js";
|
|
6
|
-
import { hiddenProperties, hiddenXObject } from "./optional-content.js";
|
|
5
|
+
import { appearanceContent, collectPageAppearances } from "./annots.js";
|
|
7
6
|
import { buildContentFont } from "./font.js";
|
|
8
|
-
import {
|
|
7
|
+
import { hiddenProperties, hiddenXObject } from "./optional-content.js";
|
|
8
|
+
import { addShown } from "./face-outlines.js";
|
|
9
|
+
import { patternTint, tintedHex } from "./pattern-tint.js";
|
|
9
10
|
//#region src/pdf-reader/text.ts
|
|
10
11
|
var MAX_FORM_DEPTH = 8;
|
|
11
12
|
/**
|
|
@@ -15,17 +16,24 @@ var MAX_FORM_DEPTH = 8;
|
|
|
15
16
|
* origin falls inside a `/Link` annotation's `/Rect` with that link's URI (EP8)
|
|
16
17
|
* so hyperlinks survive.
|
|
17
18
|
*
|
|
18
|
-
* @param file
|
|
19
|
-
* @param page
|
|
19
|
+
* @param file The owning {@link PdfFile}.
|
|
20
|
+
* @param page The page to extract.
|
|
21
|
+
* @param painted Where to gather the codes each font painted, for a caller
|
|
22
|
+
* that embeds the faces (see `./face-outlines`).
|
|
20
23
|
* @returns The page's runs, each carrying an `href` when it sits under a link.
|
|
21
24
|
*/
|
|
22
|
-
function extractPageText(file, page) {
|
|
25
|
+
function extractPageText(file, page, painted) {
|
|
23
26
|
const runs = [];
|
|
24
|
-
collectRuns(file, page.resources, file.pageContent(page), IDENTITY, 0, /* @__PURE__ */ new Set(), runs);
|
|
25
|
-
|
|
27
|
+
collectRuns(file, page.resources, file.pageContent(page), IDENTITY, 0, /* @__PURE__ */ new Set(), runs, painted);
|
|
28
|
+
const own = runs.length;
|
|
29
|
+
for (const appearance of collectPageAppearances(file, page)) collectRuns(file, appearance.resources ?? page.resources, appearanceContent(file, appearance), appearance.ctm, 1, new Set([appearance.stream]), runs, painted);
|
|
30
|
+
for (let i = own; i < runs.length; i++) runs[i] = {
|
|
31
|
+
...runs[i],
|
|
32
|
+
annotation: true
|
|
33
|
+
};
|
|
26
34
|
const links = collectLinks(file, page);
|
|
27
35
|
const marks = collectTextMarkup(file, page);
|
|
28
|
-
const shown = withoutRestrikes(runs);
|
|
36
|
+
const shown = withAccentsComposed(withoutRestrikes(runs));
|
|
29
37
|
if (links.length === 0 && marks.length === 0) return shown;
|
|
30
38
|
return shown.flatMap((run) => {
|
|
31
39
|
const link = links.find((l) => inRect(run.x, run.y, l.rect));
|
|
@@ -163,6 +171,91 @@ function withoutRestrikes(runs) {
|
|
|
163
171
|
}
|
|
164
172
|
/** How near, in ems, a re-strike of the same text lands to the one it thickens. */
|
|
165
173
|
var RESTRIKE_EM = .08;
|
|
174
|
+
/**
|
|
175
|
+
* §9.4.3 — an accent struck over a letter, composed with it.
|
|
176
|
+
*
|
|
177
|
+
* TeX sets an accented letter its font has no glyph for as two: the accent,
|
|
178
|
+
* and the letter drawn back under it — "ï" is a dieresis and a dotless i.
|
|
179
|
+
* Read as they come, comments.pdf's "naïve" came back "na¨ıve", the accent a
|
|
180
|
+
* character of the word. A run that is one spacing accent, with the pen taken
|
|
181
|
+
* back under it for the run after it, is the accent of that run's first letter
|
|
182
|
+
* — or, where the accent was struck after its letter, of the last letter of the
|
|
183
|
+
* run before it. The two are written as the one character Unicode composes them
|
|
184
|
+
* into, a dotless i or j taking back its dot's place under the accent.
|
|
185
|
+
*
|
|
186
|
+
* @param runs The page's runs, in painting order.
|
|
187
|
+
* @returns The runs with each such accent composed into its letter.
|
|
188
|
+
*/
|
|
189
|
+
function withAccentsComposed(runs) {
|
|
190
|
+
const out = [];
|
|
191
|
+
for (let i = 0; i < runs.length; i++) {
|
|
192
|
+
const run = runs[i];
|
|
193
|
+
const mark = ACCENTS.get(run.text);
|
|
194
|
+
const next = runs[i + 1];
|
|
195
|
+
const prev = out[out.length - 1];
|
|
196
|
+
if (mark !== void 0 && next !== void 0 && struckOver(run, next, "first")) {
|
|
197
|
+
const [first = "", ...rest] = [...next.text];
|
|
198
|
+
out.push({
|
|
199
|
+
...next,
|
|
200
|
+
text: accented(first, mark) + rest.join("")
|
|
201
|
+
});
|
|
202
|
+
i++;
|
|
203
|
+
continue;
|
|
204
|
+
}
|
|
205
|
+
if (mark !== void 0 && prev !== void 0 && struckOver(run, prev, "last")) {
|
|
206
|
+
const chars = [...prev.text];
|
|
207
|
+
const last = chars.pop() ?? "";
|
|
208
|
+
out[out.length - 1] = {
|
|
209
|
+
...prev,
|
|
210
|
+
text: chars.join("") + accented(last, mark)
|
|
211
|
+
};
|
|
212
|
+
continue;
|
|
213
|
+
}
|
|
214
|
+
out.push(run);
|
|
215
|
+
}
|
|
216
|
+
return out;
|
|
217
|
+
}
|
|
218
|
+
/**
|
|
219
|
+
* Whether an accent stands over the first letter of a run (the pen taken back
|
|
220
|
+
* under it) or over the last (the pen taken back to strike it): on one
|
|
221
|
+
* baseline, give or take the lift an accent over a capital takes, and
|
|
222
|
+
* overlapping it by more than a hair.
|
|
223
|
+
*/
|
|
224
|
+
function struckOver(accent, base, which) {
|
|
225
|
+
if (accent.angleDeg !== void 0 || base.angleDeg !== void 0) return false;
|
|
226
|
+
const size = base.fontSizePt || accent.fontSizePt || 10;
|
|
227
|
+
if (Math.abs(accent.y - base.y) > size * ACCENT_LIFT_EM) return false;
|
|
228
|
+
if (!/^\p{L}/u.test(which === "first" ? base.text : [...base.text].at(-1) ?? "")) return false;
|
|
229
|
+
const hair = size * ACCENT_OVERLAP_EM;
|
|
230
|
+
return which === "first" ? base.x < accent.endX - hair && base.x > accent.x - size : accent.x < base.endX - hair && accent.x > base.x;
|
|
231
|
+
}
|
|
232
|
+
/** A letter with an accent composed onto it: a dotless i or j takes back its dot's place. */
|
|
233
|
+
function accented(letter, mark) {
|
|
234
|
+
return ((letter === "ı" ? "i" : letter === "ȷ" ? "j" : letter) + mark).normalize("NFC");
|
|
235
|
+
}
|
|
236
|
+
/** How far over the line, in ems, TeX lifts an accent to stand over a capital. */
|
|
237
|
+
var ACCENT_LIFT_EM = .6;
|
|
238
|
+
/** How far, in ems, an accent must overlap its letter to stand over it. */
|
|
239
|
+
var ACCENT_OVERLAP_EM = .05;
|
|
240
|
+
/** The spacing accents TeX strikes over a letter, and the combining marks they are. */
|
|
241
|
+
var ACCENTS = new Map([
|
|
242
|
+
["`", "̀"],
|
|
243
|
+
["´", "́"],
|
|
244
|
+
["ˆ", "̂"],
|
|
245
|
+
["^", "̂"],
|
|
246
|
+
["˜", "̃"],
|
|
247
|
+
["~", "̃"],
|
|
248
|
+
["¯", "̄"],
|
|
249
|
+
["ˉ", "̄"],
|
|
250
|
+
["˘", "̆"],
|
|
251
|
+
["˙", "̇"],
|
|
252
|
+
["¨", "̈"],
|
|
253
|
+
["˚", "̊"],
|
|
254
|
+
["˝", "̋"],
|
|
255
|
+
["ˇ", "̌"],
|
|
256
|
+
["¸", "̧"],
|
|
257
|
+
["˛", "̨"]
|
|
258
|
+
]);
|
|
166
259
|
var spaceCache = /* @__PURE__ */ new WeakMap();
|
|
167
260
|
/** The page's shading patterns, read once per resource dictionary. */
|
|
168
261
|
var shadingCache = /* @__PURE__ */ new WeakMap();
|
|
@@ -182,14 +275,15 @@ function spacesOf(file, resources) {
|
|
|
182
275
|
spaceCache.set(resources, made);
|
|
183
276
|
return made;
|
|
184
277
|
}
|
|
185
|
-
function collectRuns(file, resources, content, baseCtm, depth, visiting, out) {
|
|
278
|
+
function collectRuns(file, resources, content, baseCtm, depth, visiting, out, shown) {
|
|
186
279
|
const result = interpretContent(content, buildFonts(file, resources), baseCtm, shadingsOf(file, resources), void 0, spacesOf(file, resources), hiddenProperties(file, resources));
|
|
187
280
|
out.push(...result.texts.map((r) => withPatternColour(file, resources, r, visiting)));
|
|
281
|
+
if (shown) addShown(shown, result.shown);
|
|
188
282
|
if (depth >= MAX_FORM_DEPTH) return;
|
|
189
283
|
for (const glyph of result.glyphs) {
|
|
190
284
|
if (visiting.has(glyph.stream)) continue;
|
|
191
285
|
visiting.add(glyph.stream);
|
|
192
|
-
collectRuns(file, glyph.resources ?? resources, file.streamData(glyph.stream), glyph.ctm, depth + 1, visiting, out);
|
|
286
|
+
collectRuns(file, glyph.resources ?? resources, file.streamData(glyph.stream), glyph.ctm, depth + 1, visiting, out, shown);
|
|
193
287
|
visiting.delete(glyph.stream);
|
|
194
288
|
}
|
|
195
289
|
if (!resources) return;
|
|
@@ -203,7 +297,7 @@ function collectRuns(file, resources, content, baseCtm, depth, visiting, out) {
|
|
|
203
297
|
if (!(sub instanceof PdfName) || sub.value !== "Form") continue;
|
|
204
298
|
visiting.add(stream);
|
|
205
299
|
const formRes = file.get(stream.dict, "Resources");
|
|
206
|
-
collectRuns(file, formRes instanceof Map ? formRes : resources, file.streamData(stream), multiply(matrixOf(file, stream.dict), placement.ctm), depth + 1, visiting, out);
|
|
300
|
+
collectRuns(file, formRes instanceof Map ? formRes : resources, file.streamData(stream), multiply(matrixOf(file, stream.dict), placement.ctm), depth + 1, visiting, out, shown);
|
|
207
301
|
visiting.delete(stream);
|
|
208
302
|
}
|
|
209
303
|
}
|
|
@@ -244,16 +338,6 @@ function withPatternColour(file, resources, run, visiting) {
|
|
|
244
338
|
visiting.delete(stream);
|
|
245
339
|
}
|
|
246
340
|
}
|
|
247
|
-
/** A colour laid over white paper at `coverage` strength, as a 6-hex string. */
|
|
248
|
-
function tintedHex(colorHex, coverage) {
|
|
249
|
-
const k = Math.min(1, Math.max(0, coverage));
|
|
250
|
-
if (k >= 1) return colorHex;
|
|
251
|
-
const channel = (at) => {
|
|
252
|
-
const c = Number.parseInt(colorHex.slice(at, at + 2), 16);
|
|
253
|
-
return Math.round(255 - (255 - (Number.isFinite(c) ? c : 0)) * k).toString(16).toUpperCase().padStart(2, "0");
|
|
254
|
-
};
|
|
255
|
-
return `${channel(0)}${channel(2)}${channel(4)}`;
|
|
256
|
-
}
|
|
257
341
|
/**
|
|
258
342
|
* The `/Font` resources of one dictionary, built into interpreter fonts. Shared
|
|
259
343
|
* with the path and picture walks, which need them for one thing only: a Type 3
|
|
@@ -10,6 +10,17 @@ export interface Type1Font {
|
|
|
10
10
|
readonly has: (name: string) => boolean;
|
|
11
11
|
/** The program's own `/Encoding`: code → glyph name, where it states one. */
|
|
12
12
|
readonly encoding?: ReadonlyMap<number, string>;
|
|
13
|
+
/**
|
|
14
|
+
* The embedding the program's licence allows — the OS/2 `fsType` Adobe
|
|
15
|
+
* writes into `FontInfo` as `/FSType`, where the program states one.
|
|
16
|
+
*/
|
|
17
|
+
readonly fsType?: number;
|
|
18
|
+
/**
|
|
19
|
+
* §6.4 — how far the pen moves after a glyph, in thousandths of an em: the
|
|
20
|
+
* width its `hsbw` (or `sbw`) states. `undefined` where the program holds no
|
|
21
|
+
* such glyph or its charstring states no width first.
|
|
22
|
+
*/
|
|
23
|
+
readonly advance: (name: string) => number | undefined;
|
|
13
24
|
}
|
|
14
25
|
/**
|
|
15
26
|
* Read an embedded Type 1 program.
|