extract-pdf 0.1.20 → 0.1.22
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +74 -74
- package/package.json +1 -1
- package/src/models/annotation.ts +41 -41
- package/src/models/block-type.ts +203 -203
- package/src/models/line-converter.ts +224 -224
- package/src/models/metadata.ts +29 -29
- package/src/models/page.ts +16 -16
- package/src/models/parse-result.ts +32 -32
- package/src/models/parsed-elements.ts +29 -29
- package/src/models/stashing-stream.ts +86 -86
- package/src/models/text-item-line-grouper.ts +41 -41
- package/src/models/word.ts +31 -31
- package/src/pdf-to-html.ts +225 -225
- package/src/transforms/base/to-line-item-block-transform.ts +29 -29
- package/src/transforms/base/to-line-item-transform.ts +29 -29
- package/src/transforms/base/to-text-item-transform.ts +28 -28
- package/src/transforms/block/detect-code-quote-blocks.ts +57 -57
- package/src/transforms/block/detect-list-levels.ts +64 -64
- package/src/transforms/block/gather-blocks.ts +113 -113
- package/src/transforms/calculate-global-stats.ts +132 -132
- package/src/transforms/line-item/compact-lines.ts +92 -92
- package/src/transforms/line-item/detect-headers.ts +173 -173
- package/src/transforms/line-item/detect-list-items.ts +68 -68
- package/src/transforms/line-item/detect-toc.ts +459 -459
- package/src/transforms/line-item/remove-repetitive-elements.ts +101 -101
- package/src/transforms/line-item/vertical-to-horizontal.ts +90 -90
- package/src/transforms/to-html.ts +46 -46
- package/src/utils/is-url-pdf.ts +33 -33
- package/src/utils/page-item-functions.ts +35 -35
- package/src/utils/string-functions.ts +124 -124
|
@@ -1,459 +1,459 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* @description Line-item transformation that identifies Table of Contents pages
|
|
3
|
-
* (≥75% of lines end with a page number) and extracts TOC links with hierarchy
|
|
4
|
-
* levels determined by x-position or font differences. Then cross-references
|
|
5
|
-
* each TOC link against document pages via `HeadlineFinder` to tag the actual
|
|
6
|
-
* heading text with H2–H6 types, falling back to font-height range matching for
|
|
7
|
-
* headings that could not be located by exact text.
|
|
8
|
-
*/
|
|
9
|
-
|
|
10
|
-
import ToLineItemTransformation from "../base/to-line-item-transform";
|
|
11
|
-
import ParseResult from "../../models/parse-result";
|
|
12
|
-
import LineItem from "../../models/line-item";
|
|
13
|
-
import Word from "../../models/word";
|
|
14
|
-
import HeadlineFinder from "../../models/headline-finder";
|
|
15
|
-
import { REMOVED_ANNOTATION, ADDED_ANNOTATION } from "../../models/annotation";
|
|
16
|
-
import BlockType from "../../models/block-type";
|
|
17
|
-
import {
|
|
18
|
-
isDigit,
|
|
19
|
-
isNumber,
|
|
20
|
-
wordMatch,
|
|
21
|
-
hasOnly,
|
|
22
|
-
} from "../../utils/string-functions";
|
|
23
|
-
|
|
24
|
-
// Detect table of contents pages plus linked headlines
|
|
25
|
-
export default class DetectTOC extends ToLineItemTransformation {
|
|
26
|
-
constructor() {
|
|
27
|
-
super("Detect TOC");
|
|
28
|
-
}
|
|
29
|
-
|
|
30
|
-
transform(parseResult: ParseResult): ParseResult {
|
|
31
|
-
const tocPages = [];
|
|
32
|
-
const maxPagesToEvaluate = Math.min(20, parseResult.pages.length);
|
|
33
|
-
const linkLeveler = new LinkLeveler();
|
|
34
|
-
|
|
35
|
-
var tocLinks = [];
|
|
36
|
-
var lastTocPage;
|
|
37
|
-
var headlineItem;
|
|
38
|
-
parseResult.pages.slice(0, maxPagesToEvaluate).forEach((page) => {
|
|
39
|
-
var lineItemsWithDigits = 0;
|
|
40
|
-
const unknownLines = new Set();
|
|
41
|
-
const pageTocLinks = [];
|
|
42
|
-
var lastWordsWithoutNumber;
|
|
43
|
-
var lastLine;
|
|
44
|
-
// find lines with words containing only "." ...
|
|
45
|
-
const tocLines = page.items.filter((line) =>
|
|
46
|
-
line.words.includes((word) => hasOnly(word.string, ".")),
|
|
47
|
-
);
|
|
48
|
-
// ... and ending with a number per page
|
|
49
|
-
tocLines.forEach((line) => {
|
|
50
|
-
var words = line.words.filter((word) => !hasOnly(word.string, "."));
|
|
51
|
-
const digits = [];
|
|
52
|
-
while (words.length > 0 && isNumber(words[words.length - 1].string)) {
|
|
53
|
-
const lastWord = words.pop();
|
|
54
|
-
digits.unshift(lastWord.string);
|
|
55
|
-
}
|
|
56
|
-
|
|
57
|
-
if (digits.length === 0 && words.length > 0) {
|
|
58
|
-
const lastWord = words[words.length - 1];
|
|
59
|
-
while (
|
|
60
|
-
isDigit(lastWord.string.charCodeAt(lastWord.string.length - 1))
|
|
61
|
-
) {
|
|
62
|
-
digits.unshift(lastWord.string.charAt(lastWord.string.length - 1));
|
|
63
|
-
lastWord.string = lastWord.string.substring(
|
|
64
|
-
0,
|
|
65
|
-
lastWord.string.length - 1,
|
|
66
|
-
);
|
|
67
|
-
}
|
|
68
|
-
}
|
|
69
|
-
var endsWithDigit = digits.length > 0;
|
|
70
|
-
if (endsWithDigit) {
|
|
71
|
-
endsWithDigit = true;
|
|
72
|
-
if (lastWordsWithoutNumber) {
|
|
73
|
-
// 2-line item ?
|
|
74
|
-
words.push(...lastWordsWithoutNumber);
|
|
75
|
-
lastWordsWithoutNumber = null;
|
|
76
|
-
}
|
|
77
|
-
pageTocLinks.push(
|
|
78
|
-
new TocLink({
|
|
79
|
-
pageNumber: parseInt(digits.join("")),
|
|
80
|
-
lineItem: new LineItem({ ...line, words }),
|
|
81
|
-
}),
|
|
82
|
-
);
|
|
83
|
-
lineItemsWithDigits++;
|
|
84
|
-
} else {
|
|
85
|
-
if (!headlineItem) {
|
|
86
|
-
headlineItem = line;
|
|
87
|
-
} else {
|
|
88
|
-
if (lastWordsWithoutNumber) {
|
|
89
|
-
unknownLines.add(lastLine);
|
|
90
|
-
}
|
|
91
|
-
lastWordsWithoutNumber = words;
|
|
92
|
-
lastLine = line;
|
|
93
|
-
}
|
|
94
|
-
}
|
|
95
|
-
});
|
|
96
|
-
|
|
97
|
-
// page has been processed
|
|
98
|
-
if ((lineItemsWithDigits * 100) / page.items.length > 75) {
|
|
99
|
-
tocPages.push(page.index + 1);
|
|
100
|
-
lastTocPage = page;
|
|
101
|
-
linkLeveler.levelPageItems(pageTocLinks);
|
|
102
|
-
tocLinks.push(...pageTocLinks);
|
|
103
|
-
|
|
104
|
-
const newBlocks = [];
|
|
105
|
-
page.items.forEach((line) => {
|
|
106
|
-
if (!unknownLines.has(line)) {
|
|
107
|
-
line.annotation = REMOVED_ANNOTATION;
|
|
108
|
-
}
|
|
109
|
-
newBlocks.push(line);
|
|
110
|
-
if (line === headlineItem) {
|
|
111
|
-
newBlocks.push(
|
|
112
|
-
new LineItem({
|
|
113
|
-
...line,
|
|
114
|
-
type: BlockType.H2,
|
|
115
|
-
annotation: ADDED_ANNOTATION,
|
|
116
|
-
}),
|
|
117
|
-
);
|
|
118
|
-
}
|
|
119
|
-
});
|
|
120
|
-
page.items = newBlocks;
|
|
121
|
-
} else {
|
|
122
|
-
headlineItem = null;
|
|
123
|
-
}
|
|
124
|
-
});
|
|
125
|
-
|
|
126
|
-
// all pages have been processed
|
|
127
|
-
var foundHeadlines = tocLinks.length;
|
|
128
|
-
const notFoundHeadlines = [];
|
|
129
|
-
const foundBySize = [];
|
|
130
|
-
const headlineTypeToHeightRange = {}; // H1={min:23, max:25}
|
|
131
|
-
|
|
132
|
-
if (tocPages.length > 0) {
|
|
133
|
-
// Add TOC items
|
|
134
|
-
tocLinks.forEach((tocLink) => {
|
|
135
|
-
lastTocPage.items.push(
|
|
136
|
-
new LineItem({
|
|
137
|
-
words: [
|
|
138
|
-
new Word({
|
|
139
|
-
string: " ".repeat(tocLink.level * 3) + "-",
|
|
140
|
-
}),
|
|
141
|
-
].concat(tocLink.lineItem.words),
|
|
142
|
-
type: BlockType.TOC,
|
|
143
|
-
annotation: ADDED_ANNOTATION,
|
|
144
|
-
}),
|
|
145
|
-
);
|
|
146
|
-
});
|
|
147
|
-
|
|
148
|
-
// Add linked headers
|
|
149
|
-
const pageMapping = detectPageMappingNumber(
|
|
150
|
-
parseResult.pages.filter((page) => page.index > lastTocPage.index),
|
|
151
|
-
tocLinks,
|
|
152
|
-
);
|
|
153
|
-
tocLinks.forEach((tocLink) => {
|
|
154
|
-
var linkedPage = parseResult.pages[tocLink.pageNumber + pageMapping];
|
|
155
|
-
var foundHealineItems;
|
|
156
|
-
if (linkedPage) {
|
|
157
|
-
foundHealineItems = findHeadlineItems(
|
|
158
|
-
linkedPage,
|
|
159
|
-
tocLink.lineItem.text(),
|
|
160
|
-
);
|
|
161
|
-
if (!foundHealineItems) {
|
|
162
|
-
// pages are off by 1 ?
|
|
163
|
-
linkedPage =
|
|
164
|
-
parseResult.pages[tocLink.pageNumber + pageMapping + 1];
|
|
165
|
-
if (linkedPage) {
|
|
166
|
-
foundHealineItems = findHeadlineItems(
|
|
167
|
-
linkedPage,
|
|
168
|
-
tocLink.lineItem.text(),
|
|
169
|
-
);
|
|
170
|
-
}
|
|
171
|
-
}
|
|
172
|
-
}
|
|
173
|
-
if (foundHealineItems) {
|
|
174
|
-
addHeadlineItems(
|
|
175
|
-
linkedPage,
|
|
176
|
-
tocLink,
|
|
177
|
-
foundHealineItems,
|
|
178
|
-
headlineTypeToHeightRange,
|
|
179
|
-
);
|
|
180
|
-
} else {
|
|
181
|
-
notFoundHeadlines.push(tocLink);
|
|
182
|
-
}
|
|
183
|
-
});
|
|
184
|
-
|
|
185
|
-
// Try to find linked headers by height
|
|
186
|
-
var fromPage = lastTocPage.index + 2;
|
|
187
|
-
var lastNotFound = [];
|
|
188
|
-
const rollupLastNotFound = (currentPageNumber) => {
|
|
189
|
-
if (lastNotFound.length > 0) {
|
|
190
|
-
lastNotFound.forEach((notFoundTocLink) => {
|
|
191
|
-
const headlineType = BlockType.headlineByLevel(
|
|
192
|
-
notFoundTocLink.level + 2,
|
|
193
|
-
);
|
|
194
|
-
const heightRange = headlineTypeToHeightRange[headlineType.name];
|
|
195
|
-
if (heightRange) {
|
|
196
|
-
const [pageIndex, lineIndex] = findPageAndLineFromHeadline(
|
|
197
|
-
parseResult.pages,
|
|
198
|
-
notFoundTocLink,
|
|
199
|
-
heightRange,
|
|
200
|
-
fromPage,
|
|
201
|
-
currentPageNumber,
|
|
202
|
-
);
|
|
203
|
-
if (lineIndex > -1) {
|
|
204
|
-
const page = parseResult.pages[pageIndex];
|
|
205
|
-
page.items[lineIndex].annotation = REMOVED_ANNOTATION;
|
|
206
|
-
page.items.splice(
|
|
207
|
-
lineIndex + 1,
|
|
208
|
-
0,
|
|
209
|
-
new LineItem({
|
|
210
|
-
...notFoundTocLink.lineItem,
|
|
211
|
-
type: headlineType,
|
|
212
|
-
annotation: ADDED_ANNOTATION,
|
|
213
|
-
}),
|
|
214
|
-
);
|
|
215
|
-
foundBySize.push(notFoundTocLink);
|
|
216
|
-
}
|
|
217
|
-
}
|
|
218
|
-
});
|
|
219
|
-
lastNotFound = [];
|
|
220
|
-
}
|
|
221
|
-
};
|
|
222
|
-
if (notFoundHeadlines.length > 0) {
|
|
223
|
-
tocLinks.forEach((tocLink) => {
|
|
224
|
-
if (notFoundHeadlines.includes(tocLink)) {
|
|
225
|
-
lastNotFound.push(tocLink);
|
|
226
|
-
} else {
|
|
227
|
-
rollupLastNotFound(tocLink.pageNumber);
|
|
228
|
-
fromPage = tocLink.pageNumber;
|
|
229
|
-
}
|
|
230
|
-
});
|
|
231
|
-
if (lastNotFound.length > 0) {
|
|
232
|
-
rollupLastNotFound(parseResult.pages.length);
|
|
233
|
-
}
|
|
234
|
-
}
|
|
235
|
-
}
|
|
236
|
-
|
|
237
|
-
const messages = [];
|
|
238
|
-
messages.push("Detected " + tocPages.length + " table of content pages");
|
|
239
|
-
if (tocPages.length > 0) {
|
|
240
|
-
messages.push(
|
|
241
|
-
"TOC headline heights: " + JSON.stringify(headlineTypeToHeightRange),
|
|
242
|
-
);
|
|
243
|
-
messages.push(
|
|
244
|
-
"Found TOC headlines: " +
|
|
245
|
-
(foundHeadlines - notFoundHeadlines.length + foundBySize.length) +
|
|
246
|
-
"/" +
|
|
247
|
-
foundHeadlines,
|
|
248
|
-
);
|
|
249
|
-
}
|
|
250
|
-
if (notFoundHeadlines.length > 0) {
|
|
251
|
-
messages.push(
|
|
252
|
-
"Found TOC headlines (by size): " +
|
|
253
|
-
foundBySize.map((tocLink) => tocLink.lineItem.text()),
|
|
254
|
-
);
|
|
255
|
-
messages.push(
|
|
256
|
-
"Missing TOC headlines: " +
|
|
257
|
-
notFoundHeadlines
|
|
258
|
-
.filter((fTocLink) => !foundBySize.includes(fTocLink))
|
|
259
|
-
.map(
|
|
260
|
-
(tocLink) => tocLink.lineItem.text() + "=>" + tocLink.pageNumber,
|
|
261
|
-
),
|
|
262
|
-
);
|
|
263
|
-
}
|
|
264
|
-
return new ParseResult({
|
|
265
|
-
...parseResult,
|
|
266
|
-
globals: {
|
|
267
|
-
...parseResult.globals,
|
|
268
|
-
tocPages,
|
|
269
|
-
headlineTypeToHeightRange,
|
|
270
|
-
},
|
|
271
|
-
messages,
|
|
272
|
-
});
|
|
273
|
-
}
|
|
274
|
-
}
|
|
275
|
-
|
|
276
|
-
// Find out how the TOC page link actualy translates to the page.index
|
|
277
|
-
function detectPageMappingNumber(pages, tocLinks) {
|
|
278
|
-
for (var tocLink of tocLinks) {
|
|
279
|
-
const page = findPageWithHeadline(pages, tocLink.lineItem.text());
|
|
280
|
-
if (page) {
|
|
281
|
-
return page.index - tocLink.pageNumber;
|
|
282
|
-
}
|
|
283
|
-
}
|
|
284
|
-
return null;
|
|
285
|
-
}
|
|
286
|
-
|
|
287
|
-
function findPageWithHeadline(pages, headline) {
|
|
288
|
-
for (var page of pages) {
|
|
289
|
-
if (findHeadlineItems(page, headline)) {
|
|
290
|
-
return page;
|
|
291
|
-
}
|
|
292
|
-
}
|
|
293
|
-
return null;
|
|
294
|
-
}
|
|
295
|
-
|
|
296
|
-
function findHeadlineItems(page, headline) {
|
|
297
|
-
const headlineFinder = new HeadlineFinder({ headline });
|
|
298
|
-
var lineIndex = 0;
|
|
299
|
-
for (var line of page.items) {
|
|
300
|
-
const headlineItems = headlineFinder.consume(line);
|
|
301
|
-
if (headlineItems) {
|
|
302
|
-
return { lineIndex, headlineItems };
|
|
303
|
-
}
|
|
304
|
-
lineIndex++;
|
|
305
|
-
}
|
|
306
|
-
return null;
|
|
307
|
-
}
|
|
308
|
-
|
|
309
|
-
function addHeadlineItems(
|
|
310
|
-
page,
|
|
311
|
-
tocLink,
|
|
312
|
-
foundItems,
|
|
313
|
-
headlineTypeToHeightRange,
|
|
314
|
-
) {
|
|
315
|
-
foundItems.headlineItems.forEach(
|
|
316
|
-
(item) => (item.annotation = REMOVED_ANNOTATION),
|
|
317
|
-
);
|
|
318
|
-
const headlineType = BlockType.headlineByLevel(tocLink.level + 2);
|
|
319
|
-
const headlineHeight = foundItems.headlineItems.reduce(
|
|
320
|
-
(max, item) => Math.max(max, item.height),
|
|
321
|
-
0,
|
|
322
|
-
);
|
|
323
|
-
page.items.splice(
|
|
324
|
-
foundItems.lineIndex + 1,
|
|
325
|
-
0,
|
|
326
|
-
new LineItem({
|
|
327
|
-
...foundItems.headlineItems[0],
|
|
328
|
-
words: tocLink.lineItem.words,
|
|
329
|
-
height: headlineHeight,
|
|
330
|
-
type: headlineType,
|
|
331
|
-
annotation: ADDED_ANNOTATION,
|
|
332
|
-
}),
|
|
333
|
-
);
|
|
334
|
-
var range = headlineTypeToHeightRange[headlineType.name];
|
|
335
|
-
if (range) {
|
|
336
|
-
range.min = Math.min(range.min, headlineHeight);
|
|
337
|
-
range.max = Math.max(range.max, headlineHeight);
|
|
338
|
-
} else {
|
|
339
|
-
range = {
|
|
340
|
-
min: headlineHeight,
|
|
341
|
-
max: headlineHeight,
|
|
342
|
-
};
|
|
343
|
-
headlineTypeToHeightRange[headlineType.name] = range;
|
|
344
|
-
}
|
|
345
|
-
}
|
|
346
|
-
|
|
347
|
-
function findPageAndLineFromHeadline(
|
|
348
|
-
pages,
|
|
349
|
-
tocLink,
|
|
350
|
-
heightRange,
|
|
351
|
-
fromPage,
|
|
352
|
-
toPage,
|
|
353
|
-
) {
|
|
354
|
-
const linkText = tocLink.lineItem.text().toUpperCase();
|
|
355
|
-
for (var i = fromPage; i <= toPage; i++) {
|
|
356
|
-
const page = pages[i - 1];
|
|
357
|
-
if (page) {
|
|
358
|
-
const lineIndex = page.items.findIndex((line) => {
|
|
359
|
-
if (
|
|
360
|
-
!line.type &&
|
|
361
|
-
!line.annotation &&
|
|
362
|
-
line.height >= heightRange.min &&
|
|
363
|
-
line.height <= heightRange.max
|
|
364
|
-
) {
|
|
365
|
-
const match = wordMatch(linkText, line.text());
|
|
366
|
-
return match >= 0.5;
|
|
367
|
-
}
|
|
368
|
-
return false;
|
|
369
|
-
});
|
|
370
|
-
if (lineIndex > -1) return [i - 1, lineIndex];
|
|
371
|
-
}
|
|
372
|
-
}
|
|
373
|
-
return [-1, -1];
|
|
374
|
-
}
|
|
375
|
-
|
|
376
|
-
class LinkLeveler {
|
|
377
|
-
levelByMethod: any;
|
|
378
|
-
uniqueFonts: any[];
|
|
379
|
-
|
|
380
|
-
constructor() {
|
|
381
|
-
this.levelByMethod = null;
|
|
382
|
-
this.uniqueFonts = [];
|
|
383
|
-
}
|
|
384
|
-
|
|
385
|
-
levelPageItems(tocLinks /*: TocLink[] */) {
|
|
386
|
-
if (!this.levelByMethod) {
|
|
387
|
-
const uniqueX = this.calculateUniqueX(tocLinks);
|
|
388
|
-
if (uniqueX.length > 1) {
|
|
389
|
-
this.levelByMethod = this.levelByXDiff;
|
|
390
|
-
} else {
|
|
391
|
-
const uniqueFonts = this.calculateUniqueFonts(tocLinks);
|
|
392
|
-
if (uniqueFonts.length > 1) {
|
|
393
|
-
this.uniqueFonts = uniqueFonts;
|
|
394
|
-
this.levelByMethod = this.levelByFont;
|
|
395
|
-
} else {
|
|
396
|
-
this.levelByMethod = this.levelToZero;
|
|
397
|
-
}
|
|
398
|
-
}
|
|
399
|
-
}
|
|
400
|
-
this.levelByMethod(tocLinks);
|
|
401
|
-
}
|
|
402
|
-
|
|
403
|
-
levelByXDiff(tocLinks) {
|
|
404
|
-
const uniqueX = this.calculateUniqueX(tocLinks);
|
|
405
|
-
tocLinks.forEach((link) => {
|
|
406
|
-
link.level = uniqueX.indexOf(link.lineItem.x);
|
|
407
|
-
});
|
|
408
|
-
}
|
|
409
|
-
|
|
410
|
-
levelByFont(tocLinks) {
|
|
411
|
-
tocLinks.forEach((link) => {
|
|
412
|
-
link.level = this.uniqueFonts.indexOf(link.lineItem.font);
|
|
413
|
-
});
|
|
414
|
-
}
|
|
415
|
-
|
|
416
|
-
levelToZero(tocLinks) {
|
|
417
|
-
tocLinks.forEach((link) => {
|
|
418
|
-
link.level = 0;
|
|
419
|
-
});
|
|
420
|
-
}
|
|
421
|
-
|
|
422
|
-
calculateUniqueX(tocLinks) {
|
|
423
|
-
var uniqueX = tocLinks.reduce(function (uniquesArray, link) {
|
|
424
|
-
if (uniquesArray.indexOf(link.lineItem.x) < 0)
|
|
425
|
-
uniquesArray.push(link.lineItem.x);
|
|
426
|
-
return uniquesArray;
|
|
427
|
-
}, []);
|
|
428
|
-
|
|
429
|
-
uniqueX.sort((a, b) => {
|
|
430
|
-
return a - b;
|
|
431
|
-
});
|
|
432
|
-
|
|
433
|
-
return uniqueX;
|
|
434
|
-
}
|
|
435
|
-
|
|
436
|
-
calculateUniqueFonts(tocLinks) {
|
|
437
|
-
var uniqueFont = tocLinks.reduce(function (uniquesArray, link) {
|
|
438
|
-
if (uniquesArray.indexOf(link.lineItem.font) < 0)
|
|
439
|
-
uniquesArray.push(link.lineItem.font);
|
|
440
|
-
return uniquesArray;
|
|
441
|
-
}, []);
|
|
442
|
-
|
|
443
|
-
return uniqueFont;
|
|
444
|
-
}
|
|
445
|
-
}
|
|
446
|
-
|
|
447
|
-
class TocLink {
|
|
448
|
-
lineItem: any;
|
|
449
|
-
pageNumber: number;
|
|
450
|
-
level: number;
|
|
451
|
-
|
|
452
|
-
constructor(options: any) {
|
|
453
|
-
this.lineItem = options.lineItem;
|
|
454
|
-
this.pageNumber = options.pageNumber;
|
|
455
|
-
this.level = 0;
|
|
456
|
-
}
|
|
457
|
-
}
|
|
458
|
-
|
|
459
|
-
|
|
1
|
+
/**
|
|
2
|
+
* @description Line-item transformation that identifies Table of Contents pages
|
|
3
|
+
* (≥75% of lines end with a page number) and extracts TOC links with hierarchy
|
|
4
|
+
* levels determined by x-position or font differences. Then cross-references
|
|
5
|
+
* each TOC link against document pages via `HeadlineFinder` to tag the actual
|
|
6
|
+
* heading text with H2–H6 types, falling back to font-height range matching for
|
|
7
|
+
* headings that could not be located by exact text.
|
|
8
|
+
*/
|
|
9
|
+
|
|
10
|
+
import ToLineItemTransformation from "../base/to-line-item-transform";
|
|
11
|
+
import ParseResult from "../../models/parse-result";
|
|
12
|
+
import LineItem from "../../models/line-item";
|
|
13
|
+
import Word from "../../models/word";
|
|
14
|
+
import HeadlineFinder from "../../models/headline-finder";
|
|
15
|
+
import { REMOVED_ANNOTATION, ADDED_ANNOTATION } from "../../models/annotation";
|
|
16
|
+
import BlockType from "../../models/block-type";
|
|
17
|
+
import {
|
|
18
|
+
isDigit,
|
|
19
|
+
isNumber,
|
|
20
|
+
wordMatch,
|
|
21
|
+
hasOnly,
|
|
22
|
+
} from "../../utils/string-functions";
|
|
23
|
+
|
|
24
|
+
// Detect table of contents pages plus linked headlines
|
|
25
|
+
export default class DetectTOC extends ToLineItemTransformation {
|
|
26
|
+
constructor() {
|
|
27
|
+
super("Detect TOC");
|
|
28
|
+
}
|
|
29
|
+
|
|
30
|
+
transform(parseResult: ParseResult): ParseResult {
|
|
31
|
+
const tocPages = [];
|
|
32
|
+
const maxPagesToEvaluate = Math.min(20, parseResult.pages.length);
|
|
33
|
+
const linkLeveler = new LinkLeveler();
|
|
34
|
+
|
|
35
|
+
var tocLinks = [];
|
|
36
|
+
var lastTocPage;
|
|
37
|
+
var headlineItem;
|
|
38
|
+
parseResult.pages.slice(0, maxPagesToEvaluate).forEach((page) => {
|
|
39
|
+
var lineItemsWithDigits = 0;
|
|
40
|
+
const unknownLines = new Set();
|
|
41
|
+
const pageTocLinks = [];
|
|
42
|
+
var lastWordsWithoutNumber;
|
|
43
|
+
var lastLine;
|
|
44
|
+
// find lines with words containing only "." ...
|
|
45
|
+
const tocLines = page.items.filter((line) =>
|
|
46
|
+
line.words.includes((word) => hasOnly(word.string, ".")),
|
|
47
|
+
);
|
|
48
|
+
// ... and ending with a number per page
|
|
49
|
+
tocLines.forEach((line) => {
|
|
50
|
+
var words = line.words.filter((word) => !hasOnly(word.string, "."));
|
|
51
|
+
const digits = [];
|
|
52
|
+
while (words.length > 0 && isNumber(words[words.length - 1].string)) {
|
|
53
|
+
const lastWord = words.pop();
|
|
54
|
+
digits.unshift(lastWord.string);
|
|
55
|
+
}
|
|
56
|
+
|
|
57
|
+
if (digits.length === 0 && words.length > 0) {
|
|
58
|
+
const lastWord = words[words.length - 1];
|
|
59
|
+
while (
|
|
60
|
+
isDigit(lastWord.string.charCodeAt(lastWord.string.length - 1))
|
|
61
|
+
) {
|
|
62
|
+
digits.unshift(lastWord.string.charAt(lastWord.string.length - 1));
|
|
63
|
+
lastWord.string = lastWord.string.substring(
|
|
64
|
+
0,
|
|
65
|
+
lastWord.string.length - 1,
|
|
66
|
+
);
|
|
67
|
+
}
|
|
68
|
+
}
|
|
69
|
+
var endsWithDigit = digits.length > 0;
|
|
70
|
+
if (endsWithDigit) {
|
|
71
|
+
endsWithDigit = true;
|
|
72
|
+
if (lastWordsWithoutNumber) {
|
|
73
|
+
// 2-line item ?
|
|
74
|
+
words.push(...lastWordsWithoutNumber);
|
|
75
|
+
lastWordsWithoutNumber = null;
|
|
76
|
+
}
|
|
77
|
+
pageTocLinks.push(
|
|
78
|
+
new TocLink({
|
|
79
|
+
pageNumber: parseInt(digits.join("")),
|
|
80
|
+
lineItem: new LineItem({ ...line, words }),
|
|
81
|
+
}),
|
|
82
|
+
);
|
|
83
|
+
lineItemsWithDigits++;
|
|
84
|
+
} else {
|
|
85
|
+
if (!headlineItem) {
|
|
86
|
+
headlineItem = line;
|
|
87
|
+
} else {
|
|
88
|
+
if (lastWordsWithoutNumber) {
|
|
89
|
+
unknownLines.add(lastLine);
|
|
90
|
+
}
|
|
91
|
+
lastWordsWithoutNumber = words;
|
|
92
|
+
lastLine = line;
|
|
93
|
+
}
|
|
94
|
+
}
|
|
95
|
+
});
|
|
96
|
+
|
|
97
|
+
// page has been processed
|
|
98
|
+
if ((lineItemsWithDigits * 100) / page.items.length > 75) {
|
|
99
|
+
tocPages.push(page.index + 1);
|
|
100
|
+
lastTocPage = page;
|
|
101
|
+
linkLeveler.levelPageItems(pageTocLinks);
|
|
102
|
+
tocLinks.push(...pageTocLinks);
|
|
103
|
+
|
|
104
|
+
const newBlocks = [];
|
|
105
|
+
page.items.forEach((line) => {
|
|
106
|
+
if (!unknownLines.has(line)) {
|
|
107
|
+
line.annotation = REMOVED_ANNOTATION;
|
|
108
|
+
}
|
|
109
|
+
newBlocks.push(line);
|
|
110
|
+
if (line === headlineItem) {
|
|
111
|
+
newBlocks.push(
|
|
112
|
+
new LineItem({
|
|
113
|
+
...line,
|
|
114
|
+
type: BlockType.H2,
|
|
115
|
+
annotation: ADDED_ANNOTATION,
|
|
116
|
+
}),
|
|
117
|
+
);
|
|
118
|
+
}
|
|
119
|
+
});
|
|
120
|
+
page.items = newBlocks;
|
|
121
|
+
} else {
|
|
122
|
+
headlineItem = null;
|
|
123
|
+
}
|
|
124
|
+
});
|
|
125
|
+
|
|
126
|
+
// all pages have been processed
|
|
127
|
+
var foundHeadlines = tocLinks.length;
|
|
128
|
+
const notFoundHeadlines = [];
|
|
129
|
+
const foundBySize = [];
|
|
130
|
+
const headlineTypeToHeightRange = {}; // H1={min:23, max:25}
|
|
131
|
+
|
|
132
|
+
if (tocPages.length > 0) {
|
|
133
|
+
// Add TOC items
|
|
134
|
+
tocLinks.forEach((tocLink) => {
|
|
135
|
+
lastTocPage.items.push(
|
|
136
|
+
new LineItem({
|
|
137
|
+
words: [
|
|
138
|
+
new Word({
|
|
139
|
+
string: " ".repeat(tocLink.level * 3) + "-",
|
|
140
|
+
}),
|
|
141
|
+
].concat(tocLink.lineItem.words),
|
|
142
|
+
type: BlockType.TOC,
|
|
143
|
+
annotation: ADDED_ANNOTATION,
|
|
144
|
+
}),
|
|
145
|
+
);
|
|
146
|
+
});
|
|
147
|
+
|
|
148
|
+
// Add linked headers
|
|
149
|
+
const pageMapping = detectPageMappingNumber(
|
|
150
|
+
parseResult.pages.filter((page) => page.index > lastTocPage.index),
|
|
151
|
+
tocLinks,
|
|
152
|
+
);
|
|
153
|
+
tocLinks.forEach((tocLink) => {
|
|
154
|
+
var linkedPage = parseResult.pages[tocLink.pageNumber + pageMapping];
|
|
155
|
+
var foundHealineItems;
|
|
156
|
+
if (linkedPage) {
|
|
157
|
+
foundHealineItems = findHeadlineItems(
|
|
158
|
+
linkedPage,
|
|
159
|
+
tocLink.lineItem.text(),
|
|
160
|
+
);
|
|
161
|
+
if (!foundHealineItems) {
|
|
162
|
+
// pages are off by 1 ?
|
|
163
|
+
linkedPage =
|
|
164
|
+
parseResult.pages[tocLink.pageNumber + pageMapping + 1];
|
|
165
|
+
if (linkedPage) {
|
|
166
|
+
foundHealineItems = findHeadlineItems(
|
|
167
|
+
linkedPage,
|
|
168
|
+
tocLink.lineItem.text(),
|
|
169
|
+
);
|
|
170
|
+
}
|
|
171
|
+
}
|
|
172
|
+
}
|
|
173
|
+
if (foundHealineItems) {
|
|
174
|
+
addHeadlineItems(
|
|
175
|
+
linkedPage,
|
|
176
|
+
tocLink,
|
|
177
|
+
foundHealineItems,
|
|
178
|
+
headlineTypeToHeightRange,
|
|
179
|
+
);
|
|
180
|
+
} else {
|
|
181
|
+
notFoundHeadlines.push(tocLink);
|
|
182
|
+
}
|
|
183
|
+
});
|
|
184
|
+
|
|
185
|
+
// Try to find linked headers by height
|
|
186
|
+
var fromPage = lastTocPage.index + 2;
|
|
187
|
+
var lastNotFound = [];
|
|
188
|
+
const rollupLastNotFound = (currentPageNumber) => {
|
|
189
|
+
if (lastNotFound.length > 0) {
|
|
190
|
+
lastNotFound.forEach((notFoundTocLink) => {
|
|
191
|
+
const headlineType = BlockType.headlineByLevel(
|
|
192
|
+
notFoundTocLink.level + 2,
|
|
193
|
+
);
|
|
194
|
+
const heightRange = headlineTypeToHeightRange[headlineType.name];
|
|
195
|
+
if (heightRange) {
|
|
196
|
+
const [pageIndex, lineIndex] = findPageAndLineFromHeadline(
|
|
197
|
+
parseResult.pages,
|
|
198
|
+
notFoundTocLink,
|
|
199
|
+
heightRange,
|
|
200
|
+
fromPage,
|
|
201
|
+
currentPageNumber,
|
|
202
|
+
);
|
|
203
|
+
if (lineIndex > -1) {
|
|
204
|
+
const page = parseResult.pages[pageIndex];
|
|
205
|
+
page.items[lineIndex].annotation = REMOVED_ANNOTATION;
|
|
206
|
+
page.items.splice(
|
|
207
|
+
lineIndex + 1,
|
|
208
|
+
0,
|
|
209
|
+
new LineItem({
|
|
210
|
+
...notFoundTocLink.lineItem,
|
|
211
|
+
type: headlineType,
|
|
212
|
+
annotation: ADDED_ANNOTATION,
|
|
213
|
+
}),
|
|
214
|
+
);
|
|
215
|
+
foundBySize.push(notFoundTocLink);
|
|
216
|
+
}
|
|
217
|
+
}
|
|
218
|
+
});
|
|
219
|
+
lastNotFound = [];
|
|
220
|
+
}
|
|
221
|
+
};
|
|
222
|
+
if (notFoundHeadlines.length > 0) {
|
|
223
|
+
tocLinks.forEach((tocLink) => {
|
|
224
|
+
if (notFoundHeadlines.includes(tocLink)) {
|
|
225
|
+
lastNotFound.push(tocLink);
|
|
226
|
+
} else {
|
|
227
|
+
rollupLastNotFound(tocLink.pageNumber);
|
|
228
|
+
fromPage = tocLink.pageNumber;
|
|
229
|
+
}
|
|
230
|
+
});
|
|
231
|
+
if (lastNotFound.length > 0) {
|
|
232
|
+
rollupLastNotFound(parseResult.pages.length);
|
|
233
|
+
}
|
|
234
|
+
}
|
|
235
|
+
}
|
|
236
|
+
|
|
237
|
+
const messages = [];
|
|
238
|
+
messages.push("Detected " + tocPages.length + " table of content pages");
|
|
239
|
+
if (tocPages.length > 0) {
|
|
240
|
+
messages.push(
|
|
241
|
+
"TOC headline heights: " + JSON.stringify(headlineTypeToHeightRange),
|
|
242
|
+
);
|
|
243
|
+
messages.push(
|
|
244
|
+
"Found TOC headlines: " +
|
|
245
|
+
(foundHeadlines - notFoundHeadlines.length + foundBySize.length) +
|
|
246
|
+
"/" +
|
|
247
|
+
foundHeadlines,
|
|
248
|
+
);
|
|
249
|
+
}
|
|
250
|
+
if (notFoundHeadlines.length > 0) {
|
|
251
|
+
messages.push(
|
|
252
|
+
"Found TOC headlines (by size): " +
|
|
253
|
+
foundBySize.map((tocLink) => tocLink.lineItem.text()),
|
|
254
|
+
);
|
|
255
|
+
messages.push(
|
|
256
|
+
"Missing TOC headlines: " +
|
|
257
|
+
notFoundHeadlines
|
|
258
|
+
.filter((fTocLink) => !foundBySize.includes(fTocLink))
|
|
259
|
+
.map(
|
|
260
|
+
(tocLink) => tocLink.lineItem.text() + "=>" + tocLink.pageNumber,
|
|
261
|
+
),
|
|
262
|
+
);
|
|
263
|
+
}
|
|
264
|
+
return new ParseResult({
|
|
265
|
+
...parseResult,
|
|
266
|
+
globals: {
|
|
267
|
+
...parseResult.globals,
|
|
268
|
+
tocPages,
|
|
269
|
+
headlineTypeToHeightRange,
|
|
270
|
+
},
|
|
271
|
+
messages,
|
|
272
|
+
});
|
|
273
|
+
}
|
|
274
|
+
}
|
|
275
|
+
|
|
276
|
+
// Find out how the TOC page link actualy translates to the page.index
|
|
277
|
+
function detectPageMappingNumber(pages, tocLinks) {
|
|
278
|
+
for (var tocLink of tocLinks) {
|
|
279
|
+
const page = findPageWithHeadline(pages, tocLink.lineItem.text());
|
|
280
|
+
if (page) {
|
|
281
|
+
return page.index - tocLink.pageNumber;
|
|
282
|
+
}
|
|
283
|
+
}
|
|
284
|
+
return null;
|
|
285
|
+
}
|
|
286
|
+
|
|
287
|
+
function findPageWithHeadline(pages, headline) {
|
|
288
|
+
for (var page of pages) {
|
|
289
|
+
if (findHeadlineItems(page, headline)) {
|
|
290
|
+
return page;
|
|
291
|
+
}
|
|
292
|
+
}
|
|
293
|
+
return null;
|
|
294
|
+
}
|
|
295
|
+
|
|
296
|
+
function findHeadlineItems(page, headline) {
|
|
297
|
+
const headlineFinder = new HeadlineFinder({ headline });
|
|
298
|
+
var lineIndex = 0;
|
|
299
|
+
for (var line of page.items) {
|
|
300
|
+
const headlineItems = headlineFinder.consume(line);
|
|
301
|
+
if (headlineItems) {
|
|
302
|
+
return { lineIndex, headlineItems };
|
|
303
|
+
}
|
|
304
|
+
lineIndex++;
|
|
305
|
+
}
|
|
306
|
+
return null;
|
|
307
|
+
}
|
|
308
|
+
|
|
309
|
+
function addHeadlineItems(
|
|
310
|
+
page,
|
|
311
|
+
tocLink,
|
|
312
|
+
foundItems,
|
|
313
|
+
headlineTypeToHeightRange,
|
|
314
|
+
) {
|
|
315
|
+
foundItems.headlineItems.forEach(
|
|
316
|
+
(item) => (item.annotation = REMOVED_ANNOTATION),
|
|
317
|
+
);
|
|
318
|
+
const headlineType = BlockType.headlineByLevel(tocLink.level + 2);
|
|
319
|
+
const headlineHeight = foundItems.headlineItems.reduce(
|
|
320
|
+
(max, item) => Math.max(max, item.height),
|
|
321
|
+
0,
|
|
322
|
+
);
|
|
323
|
+
page.items.splice(
|
|
324
|
+
foundItems.lineIndex + 1,
|
|
325
|
+
0,
|
|
326
|
+
new LineItem({
|
|
327
|
+
...foundItems.headlineItems[0],
|
|
328
|
+
words: tocLink.lineItem.words,
|
|
329
|
+
height: headlineHeight,
|
|
330
|
+
type: headlineType,
|
|
331
|
+
annotation: ADDED_ANNOTATION,
|
|
332
|
+
}),
|
|
333
|
+
);
|
|
334
|
+
var range = headlineTypeToHeightRange[headlineType.name];
|
|
335
|
+
if (range) {
|
|
336
|
+
range.min = Math.min(range.min, headlineHeight);
|
|
337
|
+
range.max = Math.max(range.max, headlineHeight);
|
|
338
|
+
} else {
|
|
339
|
+
range = {
|
|
340
|
+
min: headlineHeight,
|
|
341
|
+
max: headlineHeight,
|
|
342
|
+
};
|
|
343
|
+
headlineTypeToHeightRange[headlineType.name] = range;
|
|
344
|
+
}
|
|
345
|
+
}
|
|
346
|
+
|
|
347
|
+
function findPageAndLineFromHeadline(
|
|
348
|
+
pages,
|
|
349
|
+
tocLink,
|
|
350
|
+
heightRange,
|
|
351
|
+
fromPage,
|
|
352
|
+
toPage,
|
|
353
|
+
) {
|
|
354
|
+
const linkText = tocLink.lineItem.text().toUpperCase();
|
|
355
|
+
for (var i = fromPage; i <= toPage; i++) {
|
|
356
|
+
const page = pages[i - 1];
|
|
357
|
+
if (page) {
|
|
358
|
+
const lineIndex = page.items.findIndex((line) => {
|
|
359
|
+
if (
|
|
360
|
+
!line.type &&
|
|
361
|
+
!line.annotation &&
|
|
362
|
+
line.height >= heightRange.min &&
|
|
363
|
+
line.height <= heightRange.max
|
|
364
|
+
) {
|
|
365
|
+
const match = wordMatch(linkText, line.text());
|
|
366
|
+
return match >= 0.5;
|
|
367
|
+
}
|
|
368
|
+
return false;
|
|
369
|
+
});
|
|
370
|
+
if (lineIndex > -1) return [i - 1, lineIndex];
|
|
371
|
+
}
|
|
372
|
+
}
|
|
373
|
+
return [-1, -1];
|
|
374
|
+
}
|
|
375
|
+
|
|
376
|
+
class LinkLeveler {
|
|
377
|
+
levelByMethod: any;
|
|
378
|
+
uniqueFonts: any[];
|
|
379
|
+
|
|
380
|
+
constructor() {
|
|
381
|
+
this.levelByMethod = null;
|
|
382
|
+
this.uniqueFonts = [];
|
|
383
|
+
}
|
|
384
|
+
|
|
385
|
+
levelPageItems(tocLinks /*: TocLink[] */) {
|
|
386
|
+
if (!this.levelByMethod) {
|
|
387
|
+
const uniqueX = this.calculateUniqueX(tocLinks);
|
|
388
|
+
if (uniqueX.length > 1) {
|
|
389
|
+
this.levelByMethod = this.levelByXDiff;
|
|
390
|
+
} else {
|
|
391
|
+
const uniqueFonts = this.calculateUniqueFonts(tocLinks);
|
|
392
|
+
if (uniqueFonts.length > 1) {
|
|
393
|
+
this.uniqueFonts = uniqueFonts;
|
|
394
|
+
this.levelByMethod = this.levelByFont;
|
|
395
|
+
} else {
|
|
396
|
+
this.levelByMethod = this.levelToZero;
|
|
397
|
+
}
|
|
398
|
+
}
|
|
399
|
+
}
|
|
400
|
+
this.levelByMethod(tocLinks);
|
|
401
|
+
}
|
|
402
|
+
|
|
403
|
+
levelByXDiff(tocLinks) {
|
|
404
|
+
const uniqueX = this.calculateUniqueX(tocLinks);
|
|
405
|
+
tocLinks.forEach((link) => {
|
|
406
|
+
link.level = uniqueX.indexOf(link.lineItem.x);
|
|
407
|
+
});
|
|
408
|
+
}
|
|
409
|
+
|
|
410
|
+
levelByFont(tocLinks) {
|
|
411
|
+
tocLinks.forEach((link) => {
|
|
412
|
+
link.level = this.uniqueFonts.indexOf(link.lineItem.font);
|
|
413
|
+
});
|
|
414
|
+
}
|
|
415
|
+
|
|
416
|
+
levelToZero(tocLinks) {
|
|
417
|
+
tocLinks.forEach((link) => {
|
|
418
|
+
link.level = 0;
|
|
419
|
+
});
|
|
420
|
+
}
|
|
421
|
+
|
|
422
|
+
calculateUniqueX(tocLinks) {
|
|
423
|
+
var uniqueX = tocLinks.reduce(function (uniquesArray, link) {
|
|
424
|
+
if (uniquesArray.indexOf(link.lineItem.x) < 0)
|
|
425
|
+
uniquesArray.push(link.lineItem.x);
|
|
426
|
+
return uniquesArray;
|
|
427
|
+
}, []);
|
|
428
|
+
|
|
429
|
+
uniqueX.sort((a, b) => {
|
|
430
|
+
return a - b;
|
|
431
|
+
});
|
|
432
|
+
|
|
433
|
+
return uniqueX;
|
|
434
|
+
}
|
|
435
|
+
|
|
436
|
+
calculateUniqueFonts(tocLinks) {
|
|
437
|
+
var uniqueFont = tocLinks.reduce(function (uniquesArray, link) {
|
|
438
|
+
if (uniquesArray.indexOf(link.lineItem.font) < 0)
|
|
439
|
+
uniquesArray.push(link.lineItem.font);
|
|
440
|
+
return uniquesArray;
|
|
441
|
+
}, []);
|
|
442
|
+
|
|
443
|
+
return uniqueFont;
|
|
444
|
+
}
|
|
445
|
+
}
|
|
446
|
+
|
|
447
|
+
class TocLink {
|
|
448
|
+
lineItem: any;
|
|
449
|
+
pageNumber: number;
|
|
450
|
+
level: number;
|
|
451
|
+
|
|
452
|
+
constructor(options: any) {
|
|
453
|
+
this.lineItem = options.lineItem;
|
|
454
|
+
this.pageNumber = options.pageNumber;
|
|
455
|
+
this.level = 0;
|
|
456
|
+
}
|
|
457
|
+
}
|
|
458
|
+
|
|
459
|
+
|