extract-pdf 0.1.0 → 0.1.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +16 -10
- package/dist/models/annotation.d.ts +20 -0
- package/dist/models/block-type.d.ts +10 -0
- package/dist/models/headline-finder.d.ts +11 -0
- package/dist/models/line-converter.d.ts +10 -0
- package/dist/models/line-item-block.d.ts +14 -0
- package/dist/models/line-item.d.ts +25 -0
- package/dist/models/metadata.d.ts +23 -0
- package/dist/models/page-item.d.ts +21 -0
- package/dist/models/page.d.ts +14 -0
- package/dist/models/parse-result.d.ts +24 -0
- package/dist/models/parsed-elements.d.ts +20 -0
- package/dist/models/stashing-stream.d.ts +23 -0
- package/dist/models/text-item-line-grouper.d.ts +8 -0
- package/dist/models/text-item.d.ts +29 -0
- package/dist/models/word.d.ts +27 -0
- package/dist/pdf-to-html.cjs.js +1 -1
- package/dist/pdf-to-html.d.ts +36 -41
- package/dist/pdf-to-html.es.js +1 -1
- package/dist/transforms/base/to-line-item-block-transform.d.ts +6 -0
- package/dist/transforms/base/to-line-item-transform.d.ts +6 -0
- package/dist/transforms/base/to-text-item-transform.d.ts +6 -0
- package/dist/transforms/base/transformation.d.ts +8 -0
- package/dist/transforms/block/detect-code-quote-blocks.d.ts +6 -0
- package/dist/transforms/block/detect-list-levels.d.ts +6 -0
- package/dist/transforms/block/gather-blocks.d.ts +6 -0
- package/dist/transforms/calculate-global-stats.d.ts +11 -0
- package/dist/transforms/line-item/compact-lines.d.ts +6 -0
- package/dist/transforms/line-item/detect-headers.d.ts +6 -0
- package/dist/transforms/line-item/detect-list-items.d.ts +6 -0
- package/dist/transforms/line-item/detect-toc.d.ts +6 -0
- package/dist/transforms/line-item/remove-repetitive-elements.d.ts +6 -0
- package/dist/transforms/line-item/vertical-to-horizontal.d.ts +6 -0
- package/dist/transforms/to-html.d.ts +6 -0
- package/dist/transforms/to-text-blocks.d.ts +6 -0
- package/dist/utils/is-url-pdf.d.ts +1 -0
- package/dist/utils/page-item-functions.d.ts +8 -0
- package/dist/utils/page-number-functions.d.ts +14 -0
- package/dist/utils/string-functions.d.ts +14 -0
- package/package.json +8 -11
- package/src/models/annotation.ts +41 -0
- package/src/models/block-type.ts +203 -0
- package/src/models/headline-finder.ts +53 -0
- package/src/models/line-converter.ts +224 -0
- package/src/models/line-item-block.ts +51 -0
- package/src/models/line-item.ts +59 -0
- package/src/models/metadata.ts +29 -0
- package/src/models/page-item.ts +36 -0
- package/src/models/page.ts +16 -0
- package/src/models/parse-result.ts +32 -0
- package/src/models/parsed-elements.ts +29 -0
- package/src/models/stashing-stream.ts +86 -0
- package/src/models/text-item-line-grouper.ts +41 -0
- package/src/models/text-item.ts +50 -0
- package/src/models/word.ts +31 -0
- package/src/pdf-to-html.ts +225 -0
- package/src/transforms/base/to-line-item-block-transform.ts +29 -0
- package/src/transforms/base/to-line-item-transform.ts +29 -0
- package/src/transforms/base/to-text-item-transform.ts +28 -0
- package/src/transforms/base/transformation.ts +35 -0
- package/src/transforms/block/detect-code-quote-blocks.ts +57 -0
- package/src/transforms/block/detect-list-levels.ts +64 -0
- package/src/transforms/block/gather-blocks.ts +113 -0
- package/src/transforms/calculate-global-stats.ts +132 -0
- package/src/transforms/line-item/compact-lines.ts +92 -0
- package/src/transforms/line-item/detect-headers.ts +173 -0
- package/src/transforms/line-item/detect-list-items.ts +68 -0
- package/src/transforms/line-item/detect-toc.ts +459 -0
- package/src/transforms/line-item/remove-repetitive-elements.ts +101 -0
- package/src/transforms/line-item/vertical-to-horizontal.ts +90 -0
- package/src/transforms/to-html.ts +46 -0
- package/src/transforms/to-text-blocks.ts +38 -0
- package/src/utils/is-url-pdf.ts +33 -0
- package/src/utils/page-item-functions.ts +35 -0
- package/src/utils/page-number-functions.ts +109 -0
- package/src/utils/string-functions.ts +124 -0
|
@@ -0,0 +1,459 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @description Line-item transformation that identifies Table of Contents pages
|
|
3
|
+
* (≥75% of lines end with a page number) and extracts TOC links with hierarchy
|
|
4
|
+
* levels determined by x-position or font differences. Then cross-references
|
|
5
|
+
* each TOC link against document pages via `HeadlineFinder` to tag the actual
|
|
6
|
+
* heading text with H2–H6 types, falling back to font-height range matching for
|
|
7
|
+
* headings that could not be located by exact text.
|
|
8
|
+
*/
|
|
9
|
+
|
|
10
|
+
import ToLineItemTransformation from "../base/to-line-item-transform";
|
|
11
|
+
import ParseResult from "../../models/parse-result";
|
|
12
|
+
import LineItem from "../../models/line-item";
|
|
13
|
+
import Word from "../../models/word";
|
|
14
|
+
import HeadlineFinder from "../../models/headline-finder";
|
|
15
|
+
import { REMOVED_ANNOTATION, ADDED_ANNOTATION } from "../../models/annotation";
|
|
16
|
+
import BlockType from "../../models/block-type";
|
|
17
|
+
import {
|
|
18
|
+
isDigit,
|
|
19
|
+
isNumber,
|
|
20
|
+
wordMatch,
|
|
21
|
+
hasOnly,
|
|
22
|
+
} from "../../utils/string-functions";
|
|
23
|
+
|
|
24
|
+
// Detect table of contents pages plus linked headlines
|
|
25
|
+
export default class DetectTOC extends ToLineItemTransformation {
|
|
26
|
+
constructor() {
|
|
27
|
+
super("Detect TOC");
|
|
28
|
+
}
|
|
29
|
+
|
|
30
|
+
transform(parseResult: ParseResult): ParseResult {
|
|
31
|
+
const tocPages = [];
|
|
32
|
+
const maxPagesToEvaluate = Math.min(20, parseResult.pages.length);
|
|
33
|
+
const linkLeveler = new LinkLeveler();
|
|
34
|
+
|
|
35
|
+
var tocLinks = [];
|
|
36
|
+
var lastTocPage;
|
|
37
|
+
var headlineItem;
|
|
38
|
+
parseResult.pages.slice(0, maxPagesToEvaluate).forEach((page) => {
|
|
39
|
+
var lineItemsWithDigits = 0;
|
|
40
|
+
const unknownLines = new Set();
|
|
41
|
+
const pageTocLinks = [];
|
|
42
|
+
var lastWordsWithoutNumber;
|
|
43
|
+
var lastLine;
|
|
44
|
+
// find lines with words containing only "." ...
|
|
45
|
+
const tocLines = page.items.filter((line) =>
|
|
46
|
+
line.words.includes((word) => hasOnly(word.string, ".")),
|
|
47
|
+
);
|
|
48
|
+
// ... and ending with a number per page
|
|
49
|
+
tocLines.forEach((line) => {
|
|
50
|
+
var words = line.words.filter((word) => !hasOnly(word.string, "."));
|
|
51
|
+
const digits = [];
|
|
52
|
+
while (words.length > 0 && isNumber(words[words.length - 1].string)) {
|
|
53
|
+
const lastWord = words.pop();
|
|
54
|
+
digits.unshift(lastWord.string);
|
|
55
|
+
}
|
|
56
|
+
|
|
57
|
+
if (digits.length === 0 && words.length > 0) {
|
|
58
|
+
const lastWord = words[words.length - 1];
|
|
59
|
+
while (
|
|
60
|
+
isDigit(lastWord.string.charCodeAt(lastWord.string.length - 1))
|
|
61
|
+
) {
|
|
62
|
+
digits.unshift(lastWord.string.charAt(lastWord.string.length - 1));
|
|
63
|
+
lastWord.string = lastWord.string.substring(
|
|
64
|
+
0,
|
|
65
|
+
lastWord.string.length - 1,
|
|
66
|
+
);
|
|
67
|
+
}
|
|
68
|
+
}
|
|
69
|
+
var endsWithDigit = digits.length > 0;
|
|
70
|
+
if (endsWithDigit) {
|
|
71
|
+
endsWithDigit = true;
|
|
72
|
+
if (lastWordsWithoutNumber) {
|
|
73
|
+
// 2-line item ?
|
|
74
|
+
words.push(...lastWordsWithoutNumber);
|
|
75
|
+
lastWordsWithoutNumber = null;
|
|
76
|
+
}
|
|
77
|
+
pageTocLinks.push(
|
|
78
|
+
new TocLink({
|
|
79
|
+
pageNumber: parseInt(digits.join("")),
|
|
80
|
+
lineItem: new LineItem({ ...line, words }),
|
|
81
|
+
}),
|
|
82
|
+
);
|
|
83
|
+
lineItemsWithDigits++;
|
|
84
|
+
} else {
|
|
85
|
+
if (!headlineItem) {
|
|
86
|
+
headlineItem = line;
|
|
87
|
+
} else {
|
|
88
|
+
if (lastWordsWithoutNumber) {
|
|
89
|
+
unknownLines.add(lastLine);
|
|
90
|
+
}
|
|
91
|
+
lastWordsWithoutNumber = words;
|
|
92
|
+
lastLine = line;
|
|
93
|
+
}
|
|
94
|
+
}
|
|
95
|
+
});
|
|
96
|
+
|
|
97
|
+
// page has been processed
|
|
98
|
+
if ((lineItemsWithDigits * 100) / page.items.length > 75) {
|
|
99
|
+
tocPages.push(page.index + 1);
|
|
100
|
+
lastTocPage = page;
|
|
101
|
+
linkLeveler.levelPageItems(pageTocLinks);
|
|
102
|
+
tocLinks.push(...pageTocLinks);
|
|
103
|
+
|
|
104
|
+
const newBlocks = [];
|
|
105
|
+
page.items.forEach((line) => {
|
|
106
|
+
if (!unknownLines.has(line)) {
|
|
107
|
+
line.annotation = REMOVED_ANNOTATION;
|
|
108
|
+
}
|
|
109
|
+
newBlocks.push(line);
|
|
110
|
+
if (line === headlineItem) {
|
|
111
|
+
newBlocks.push(
|
|
112
|
+
new LineItem({
|
|
113
|
+
...line,
|
|
114
|
+
type: BlockType.H2,
|
|
115
|
+
annotation: ADDED_ANNOTATION,
|
|
116
|
+
}),
|
|
117
|
+
);
|
|
118
|
+
}
|
|
119
|
+
});
|
|
120
|
+
page.items = newBlocks;
|
|
121
|
+
} else {
|
|
122
|
+
headlineItem = null;
|
|
123
|
+
}
|
|
124
|
+
});
|
|
125
|
+
|
|
126
|
+
// all pages have been processed
|
|
127
|
+
var foundHeadlines = tocLinks.length;
|
|
128
|
+
const notFoundHeadlines = [];
|
|
129
|
+
const foundBySize = [];
|
|
130
|
+
const headlineTypeToHeightRange = {}; // H1={min:23, max:25}
|
|
131
|
+
|
|
132
|
+
if (tocPages.length > 0) {
|
|
133
|
+
// Add TOC items
|
|
134
|
+
tocLinks.forEach((tocLink) => {
|
|
135
|
+
lastTocPage.items.push(
|
|
136
|
+
new LineItem({
|
|
137
|
+
words: [
|
|
138
|
+
new Word({
|
|
139
|
+
string: " ".repeat(tocLink.level * 3) + "-",
|
|
140
|
+
}),
|
|
141
|
+
].concat(tocLink.lineItem.words),
|
|
142
|
+
type: BlockType.TOC,
|
|
143
|
+
annotation: ADDED_ANNOTATION,
|
|
144
|
+
}),
|
|
145
|
+
);
|
|
146
|
+
});
|
|
147
|
+
|
|
148
|
+
// Add linked headers
|
|
149
|
+
const pageMapping = detectPageMappingNumber(
|
|
150
|
+
parseResult.pages.filter((page) => page.index > lastTocPage.index),
|
|
151
|
+
tocLinks,
|
|
152
|
+
);
|
|
153
|
+
tocLinks.forEach((tocLink) => {
|
|
154
|
+
var linkedPage = parseResult.pages[tocLink.pageNumber + pageMapping];
|
|
155
|
+
var foundHealineItems;
|
|
156
|
+
if (linkedPage) {
|
|
157
|
+
foundHealineItems = findHeadlineItems(
|
|
158
|
+
linkedPage,
|
|
159
|
+
tocLink.lineItem.text(),
|
|
160
|
+
);
|
|
161
|
+
if (!foundHealineItems) {
|
|
162
|
+
// pages are off by 1 ?
|
|
163
|
+
linkedPage =
|
|
164
|
+
parseResult.pages[tocLink.pageNumber + pageMapping + 1];
|
|
165
|
+
if (linkedPage) {
|
|
166
|
+
foundHealineItems = findHeadlineItems(
|
|
167
|
+
linkedPage,
|
|
168
|
+
tocLink.lineItem.text(),
|
|
169
|
+
);
|
|
170
|
+
}
|
|
171
|
+
}
|
|
172
|
+
}
|
|
173
|
+
if (foundHealineItems) {
|
|
174
|
+
addHeadlineItems(
|
|
175
|
+
linkedPage,
|
|
176
|
+
tocLink,
|
|
177
|
+
foundHealineItems,
|
|
178
|
+
headlineTypeToHeightRange,
|
|
179
|
+
);
|
|
180
|
+
} else {
|
|
181
|
+
notFoundHeadlines.push(tocLink);
|
|
182
|
+
}
|
|
183
|
+
});
|
|
184
|
+
|
|
185
|
+
// Try to find linked headers by height
|
|
186
|
+
var fromPage = lastTocPage.index + 2;
|
|
187
|
+
var lastNotFound = [];
|
|
188
|
+
const rollupLastNotFound = (currentPageNumber) => {
|
|
189
|
+
if (lastNotFound.length > 0) {
|
|
190
|
+
lastNotFound.forEach((notFoundTocLink) => {
|
|
191
|
+
const headlineType = BlockType.headlineByLevel(
|
|
192
|
+
notFoundTocLink.level + 2,
|
|
193
|
+
);
|
|
194
|
+
const heightRange = headlineTypeToHeightRange[headlineType.name];
|
|
195
|
+
if (heightRange) {
|
|
196
|
+
const [pageIndex, lineIndex] = findPageAndLineFromHeadline(
|
|
197
|
+
parseResult.pages,
|
|
198
|
+
notFoundTocLink,
|
|
199
|
+
heightRange,
|
|
200
|
+
fromPage,
|
|
201
|
+
currentPageNumber,
|
|
202
|
+
);
|
|
203
|
+
if (lineIndex > -1) {
|
|
204
|
+
const page = parseResult.pages[pageIndex];
|
|
205
|
+
page.items[lineIndex].annotation = REMOVED_ANNOTATION;
|
|
206
|
+
page.items.splice(
|
|
207
|
+
lineIndex + 1,
|
|
208
|
+
0,
|
|
209
|
+
new LineItem({
|
|
210
|
+
...notFoundTocLink.lineItem,
|
|
211
|
+
type: headlineType,
|
|
212
|
+
annotation: ADDED_ANNOTATION,
|
|
213
|
+
}),
|
|
214
|
+
);
|
|
215
|
+
foundBySize.push(notFoundTocLink);
|
|
216
|
+
}
|
|
217
|
+
}
|
|
218
|
+
});
|
|
219
|
+
lastNotFound = [];
|
|
220
|
+
}
|
|
221
|
+
};
|
|
222
|
+
if (notFoundHeadlines.length > 0) {
|
|
223
|
+
tocLinks.forEach((tocLink) => {
|
|
224
|
+
if (notFoundHeadlines.includes(tocLink)) {
|
|
225
|
+
lastNotFound.push(tocLink);
|
|
226
|
+
} else {
|
|
227
|
+
rollupLastNotFound(tocLink.pageNumber);
|
|
228
|
+
fromPage = tocLink.pageNumber;
|
|
229
|
+
}
|
|
230
|
+
});
|
|
231
|
+
if (lastNotFound.length > 0) {
|
|
232
|
+
rollupLastNotFound(parseResult.pages.length);
|
|
233
|
+
}
|
|
234
|
+
}
|
|
235
|
+
}
|
|
236
|
+
|
|
237
|
+
const messages = [];
|
|
238
|
+
messages.push("Detected " + tocPages.length + " table of content pages");
|
|
239
|
+
if (tocPages.length > 0) {
|
|
240
|
+
messages.push(
|
|
241
|
+
"TOC headline heights: " + JSON.stringify(headlineTypeToHeightRange),
|
|
242
|
+
);
|
|
243
|
+
messages.push(
|
|
244
|
+
"Found TOC headlines: " +
|
|
245
|
+
(foundHeadlines - notFoundHeadlines.length + foundBySize.length) +
|
|
246
|
+
"/" +
|
|
247
|
+
foundHeadlines,
|
|
248
|
+
);
|
|
249
|
+
}
|
|
250
|
+
if (notFoundHeadlines.length > 0) {
|
|
251
|
+
messages.push(
|
|
252
|
+
"Found TOC headlines (by size): " +
|
|
253
|
+
foundBySize.map((tocLink) => tocLink.lineItem.text()),
|
|
254
|
+
);
|
|
255
|
+
messages.push(
|
|
256
|
+
"Missing TOC headlines: " +
|
|
257
|
+
notFoundHeadlines
|
|
258
|
+
.filter((fTocLink) => !foundBySize.includes(fTocLink))
|
|
259
|
+
.map(
|
|
260
|
+
(tocLink) => tocLink.lineItem.text() + "=>" + tocLink.pageNumber,
|
|
261
|
+
),
|
|
262
|
+
);
|
|
263
|
+
}
|
|
264
|
+
return new ParseResult({
|
|
265
|
+
...parseResult,
|
|
266
|
+
globals: {
|
|
267
|
+
...parseResult.globals,
|
|
268
|
+
tocPages,
|
|
269
|
+
headlineTypeToHeightRange,
|
|
270
|
+
},
|
|
271
|
+
messages,
|
|
272
|
+
});
|
|
273
|
+
}
|
|
274
|
+
}
|
|
275
|
+
|
|
276
|
+
// Find out how the TOC page link actualy translates to the page.index
|
|
277
|
+
function detectPageMappingNumber(pages, tocLinks) {
|
|
278
|
+
for (var tocLink of tocLinks) {
|
|
279
|
+
const page = findPageWithHeadline(pages, tocLink.lineItem.text());
|
|
280
|
+
if (page) {
|
|
281
|
+
return page.index - tocLink.pageNumber;
|
|
282
|
+
}
|
|
283
|
+
}
|
|
284
|
+
return null;
|
|
285
|
+
}
|
|
286
|
+
|
|
287
|
+
function findPageWithHeadline(pages, headline) {
|
|
288
|
+
for (var page of pages) {
|
|
289
|
+
if (findHeadlineItems(page, headline)) {
|
|
290
|
+
return page;
|
|
291
|
+
}
|
|
292
|
+
}
|
|
293
|
+
return null;
|
|
294
|
+
}
|
|
295
|
+
|
|
296
|
+
function findHeadlineItems(page, headline) {
|
|
297
|
+
const headlineFinder = new HeadlineFinder({ headline });
|
|
298
|
+
var lineIndex = 0;
|
|
299
|
+
for (var line of page.items) {
|
|
300
|
+
const headlineItems = headlineFinder.consume(line);
|
|
301
|
+
if (headlineItems) {
|
|
302
|
+
return { lineIndex, headlineItems };
|
|
303
|
+
}
|
|
304
|
+
lineIndex++;
|
|
305
|
+
}
|
|
306
|
+
return null;
|
|
307
|
+
}
|
|
308
|
+
|
|
309
|
+
function addHeadlineItems(
|
|
310
|
+
page,
|
|
311
|
+
tocLink,
|
|
312
|
+
foundItems,
|
|
313
|
+
headlineTypeToHeightRange,
|
|
314
|
+
) {
|
|
315
|
+
foundItems.headlineItems.forEach(
|
|
316
|
+
(item) => (item.annotation = REMOVED_ANNOTATION),
|
|
317
|
+
);
|
|
318
|
+
const headlineType = BlockType.headlineByLevel(tocLink.level + 2);
|
|
319
|
+
const headlineHeight = foundItems.headlineItems.reduce(
|
|
320
|
+
(max, item) => Math.max(max, item.height),
|
|
321
|
+
0,
|
|
322
|
+
);
|
|
323
|
+
page.items.splice(
|
|
324
|
+
foundItems.lineIndex + 1,
|
|
325
|
+
0,
|
|
326
|
+
new LineItem({
|
|
327
|
+
...foundItems.headlineItems[0],
|
|
328
|
+
words: tocLink.lineItem.words,
|
|
329
|
+
height: headlineHeight,
|
|
330
|
+
type: headlineType,
|
|
331
|
+
annotation: ADDED_ANNOTATION,
|
|
332
|
+
}),
|
|
333
|
+
);
|
|
334
|
+
var range = headlineTypeToHeightRange[headlineType.name];
|
|
335
|
+
if (range) {
|
|
336
|
+
range.min = Math.min(range.min, headlineHeight);
|
|
337
|
+
range.max = Math.max(range.max, headlineHeight);
|
|
338
|
+
} else {
|
|
339
|
+
range = {
|
|
340
|
+
min: headlineHeight,
|
|
341
|
+
max: headlineHeight,
|
|
342
|
+
};
|
|
343
|
+
headlineTypeToHeightRange[headlineType.name] = range;
|
|
344
|
+
}
|
|
345
|
+
}
|
|
346
|
+
|
|
347
|
+
function findPageAndLineFromHeadline(
|
|
348
|
+
pages,
|
|
349
|
+
tocLink,
|
|
350
|
+
heightRange,
|
|
351
|
+
fromPage,
|
|
352
|
+
toPage,
|
|
353
|
+
) {
|
|
354
|
+
const linkText = tocLink.lineItem.text().toUpperCase();
|
|
355
|
+
for (var i = fromPage; i <= toPage; i++) {
|
|
356
|
+
const page = pages[i - 1];
|
|
357
|
+
if (page) {
|
|
358
|
+
const lineIndex = page.items.findIndex((line) => {
|
|
359
|
+
if (
|
|
360
|
+
!line.type &&
|
|
361
|
+
!line.annotation &&
|
|
362
|
+
line.height >= heightRange.min &&
|
|
363
|
+
line.height <= heightRange.max
|
|
364
|
+
) {
|
|
365
|
+
const match = wordMatch(linkText, line.text());
|
|
366
|
+
return match >= 0.5;
|
|
367
|
+
}
|
|
368
|
+
return false;
|
|
369
|
+
});
|
|
370
|
+
if (lineIndex > -1) return [i - 1, lineIndex];
|
|
371
|
+
}
|
|
372
|
+
}
|
|
373
|
+
return [-1, -1];
|
|
374
|
+
}
|
|
375
|
+
|
|
376
|
+
class LinkLeveler {
|
|
377
|
+
levelByMethod: any;
|
|
378
|
+
uniqueFonts: any[];
|
|
379
|
+
|
|
380
|
+
constructor() {
|
|
381
|
+
this.levelByMethod = null;
|
|
382
|
+
this.uniqueFonts = [];
|
|
383
|
+
}
|
|
384
|
+
|
|
385
|
+
levelPageItems(tocLinks /*: TocLink[] */) {
|
|
386
|
+
if (!this.levelByMethod) {
|
|
387
|
+
const uniqueX = this.calculateUniqueX(tocLinks);
|
|
388
|
+
if (uniqueX.length > 1) {
|
|
389
|
+
this.levelByMethod = this.levelByXDiff;
|
|
390
|
+
} else {
|
|
391
|
+
const uniqueFonts = this.calculateUniqueFonts(tocLinks);
|
|
392
|
+
if (uniqueFonts.length > 1) {
|
|
393
|
+
this.uniqueFonts = uniqueFonts;
|
|
394
|
+
this.levelByMethod = this.levelByFont;
|
|
395
|
+
} else {
|
|
396
|
+
this.levelByMethod = this.levelToZero;
|
|
397
|
+
}
|
|
398
|
+
}
|
|
399
|
+
}
|
|
400
|
+
this.levelByMethod(tocLinks);
|
|
401
|
+
}
|
|
402
|
+
|
|
403
|
+
levelByXDiff(tocLinks) {
|
|
404
|
+
const uniqueX = this.calculateUniqueX(tocLinks);
|
|
405
|
+
tocLinks.forEach((link) => {
|
|
406
|
+
link.level = uniqueX.indexOf(link.lineItem.x);
|
|
407
|
+
});
|
|
408
|
+
}
|
|
409
|
+
|
|
410
|
+
levelByFont(tocLinks) {
|
|
411
|
+
tocLinks.forEach((link) => {
|
|
412
|
+
link.level = this.uniqueFonts.indexOf(link.lineItem.font);
|
|
413
|
+
});
|
|
414
|
+
}
|
|
415
|
+
|
|
416
|
+
levelToZero(tocLinks) {
|
|
417
|
+
tocLinks.forEach((link) => {
|
|
418
|
+
link.level = 0;
|
|
419
|
+
});
|
|
420
|
+
}
|
|
421
|
+
|
|
422
|
+
calculateUniqueX(tocLinks) {
|
|
423
|
+
var uniqueX = tocLinks.reduce(function (uniquesArray, link) {
|
|
424
|
+
if (uniquesArray.indexOf(link.lineItem.x) < 0)
|
|
425
|
+
uniquesArray.push(link.lineItem.x);
|
|
426
|
+
return uniquesArray;
|
|
427
|
+
}, []);
|
|
428
|
+
|
|
429
|
+
uniqueX.sort((a, b) => {
|
|
430
|
+
return a - b;
|
|
431
|
+
});
|
|
432
|
+
|
|
433
|
+
return uniqueX;
|
|
434
|
+
}
|
|
435
|
+
|
|
436
|
+
calculateUniqueFonts(tocLinks) {
|
|
437
|
+
var uniqueFont = tocLinks.reduce(function (uniquesArray, link) {
|
|
438
|
+
if (uniquesArray.indexOf(link.lineItem.font) < 0)
|
|
439
|
+
uniquesArray.push(link.lineItem.font);
|
|
440
|
+
return uniquesArray;
|
|
441
|
+
}, []);
|
|
442
|
+
|
|
443
|
+
return uniqueFont;
|
|
444
|
+
}
|
|
445
|
+
}
|
|
446
|
+
|
|
447
|
+
class TocLink {
|
|
448
|
+
lineItem: any;
|
|
449
|
+
pageNumber: number;
|
|
450
|
+
level: number;
|
|
451
|
+
|
|
452
|
+
constructor(options: any) {
|
|
453
|
+
this.lineItem = options.lineItem;
|
|
454
|
+
this.pageNumber = options.pageNumber;
|
|
455
|
+
this.level = 0;
|
|
456
|
+
}
|
|
457
|
+
}
|
|
458
|
+
|
|
459
|
+
|
|
@@ -0,0 +1,101 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @description Line-item transformation that removes running page headers and
|
|
3
|
+
* footers. Hashes the topmost and bottommost line of each page (ignoring digits
|
|
4
|
+
* and whitespace so page numbers don't break the match), then marks lines whose
|
|
5
|
+
* hash appears on more than two-thirds of all pages with `REMOVED_ANNOTATION`.
|
|
6
|
+
*/
|
|
7
|
+
import ToLineItemTransformation from '../base/to-line-item-transform'
|
|
8
|
+
import ParseResult from '../../models/parse-result'
|
|
9
|
+
import { REMOVED_ANNOTATION } from '../../models/annotation'
|
|
10
|
+
|
|
11
|
+
import { isDigit } from '../../utils/string-functions'
|
|
12
|
+
|
|
13
|
+
function hashCodeIgnoringSpacesAndNumbers(string: string): number {
|
|
14
|
+
var hash = 0
|
|
15
|
+
if (string.trim().length === 0) return hash
|
|
16
|
+
for (var i = 0; i < string.length; i++) {
|
|
17
|
+
const charCode = string.charCodeAt(i)
|
|
18
|
+
if (!isDigit(charCode) && charCode !== 32 && charCode !== 160) {
|
|
19
|
+
hash = ((hash << 5) - hash) + charCode
|
|
20
|
+
hash |= 0 // Convert to 32bit integer
|
|
21
|
+
}
|
|
22
|
+
}
|
|
23
|
+
return hash
|
|
24
|
+
}
|
|
25
|
+
|
|
26
|
+
// Remove elements with similar content on same page positions, like page numbers, licenes information, etc...
|
|
27
|
+
export default class RemoveRepetitiveElements extends ToLineItemTransformation {
|
|
28
|
+
constructor () {
|
|
29
|
+
super('Remove Repetitive Elements')
|
|
30
|
+
}
|
|
31
|
+
|
|
32
|
+
// The idea is the following:
|
|
33
|
+
// - For each page, collect all items of the first, and all items of the last line
|
|
34
|
+
// - Calculate how often these items occur accros all pages (hash ignoring numbers, whitespace, upper/lowercase)
|
|
35
|
+
// - Delete items occuring on more then 2/3 of all pages
|
|
36
|
+
transform(parseResult: ParseResult): ParseResult {
|
|
37
|
+
// find first and last lines per page
|
|
38
|
+
const pageStore = []
|
|
39
|
+
const minLineHashRepetitions = {}
|
|
40
|
+
const maxLineHashRepetitions = {}
|
|
41
|
+
parseResult.pages.forEach(page => {
|
|
42
|
+
const minMaxItems = page.items.reduce((itemStore, item) => {
|
|
43
|
+
if (item.y < itemStore.minY) {
|
|
44
|
+
itemStore.minElements = [item]
|
|
45
|
+
itemStore.minY = item.y
|
|
46
|
+
} else if (item.y === itemStore.minY) {
|
|
47
|
+
itemStore.minElements.push(item)
|
|
48
|
+
}
|
|
49
|
+
if (item.y > itemStore.maxY) {
|
|
50
|
+
itemStore.maxElements = [item]
|
|
51
|
+
itemStore.maxY = item.y
|
|
52
|
+
} else if (item.y === itemStore.maxY) {
|
|
53
|
+
itemStore.maxElements.push(item)
|
|
54
|
+
}
|
|
55
|
+
return itemStore
|
|
56
|
+
}, {
|
|
57
|
+
minY: 999,
|
|
58
|
+
maxY: 0,
|
|
59
|
+
minElements: [],
|
|
60
|
+
maxElements: [],
|
|
61
|
+
})
|
|
62
|
+
|
|
63
|
+
const minLineHash = hashCodeIgnoringSpacesAndNumbers(minMaxItems.minElements.reduce((combinedString, item) => combinedString + item.text().toUpperCase(), ''))
|
|
64
|
+
const maxLineHash = hashCodeIgnoringSpacesAndNumbers(minMaxItems.maxElements.reduce((combinedString, item) => combinedString + item.text().toUpperCase(), ''))
|
|
65
|
+
pageStore.push({
|
|
66
|
+
minElements: minMaxItems.minElements,
|
|
67
|
+
maxElements: minMaxItems.maxElements,
|
|
68
|
+
minLineHash: minLineHash,
|
|
69
|
+
maxLineHash: maxLineHash,
|
|
70
|
+
})
|
|
71
|
+
minLineHashRepetitions[minLineHash] = minLineHashRepetitions[minLineHash] ? minLineHashRepetitions[minLineHash] + 1 : 1
|
|
72
|
+
maxLineHashRepetitions[maxLineHash] = maxLineHashRepetitions[maxLineHash] ? maxLineHashRepetitions[maxLineHash] + 1 : 1
|
|
73
|
+
})
|
|
74
|
+
|
|
75
|
+
// now annoate all removed items
|
|
76
|
+
var removedHeader = 0
|
|
77
|
+
var removedFooter = 0
|
|
78
|
+
parseResult.pages.forEach((page, i) => {
|
|
79
|
+
if (minLineHashRepetitions[pageStore[i].minLineHash] >= Math.max(3, parseResult.pages.length * 2 / 3)) {
|
|
80
|
+
pageStore[i].minElements.forEach(item => {
|
|
81
|
+
item.annotation = REMOVED_ANNOTATION
|
|
82
|
+
})
|
|
83
|
+
removedFooter++
|
|
84
|
+
}
|
|
85
|
+
if (maxLineHashRepetitions[pageStore[i].maxLineHash] >= Math.max(3, parseResult.pages.length * 2 / 3)) {
|
|
86
|
+
pageStore[i].maxElements.forEach(item => {
|
|
87
|
+
item.annotation = REMOVED_ANNOTATION
|
|
88
|
+
})
|
|
89
|
+
removedHeader++
|
|
90
|
+
}
|
|
91
|
+
})
|
|
92
|
+
|
|
93
|
+
return new ParseResult({
|
|
94
|
+
...parseResult,
|
|
95
|
+
messages: [
|
|
96
|
+
'Removed Header: ' + removedHeader,
|
|
97
|
+
'Removed Footers: ' + removedFooter,
|
|
98
|
+
],
|
|
99
|
+
})
|
|
100
|
+
}
|
|
101
|
+
}
|
|
@@ -0,0 +1,90 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @description Line-item transformation that recovers rotated sidebar text.
|
|
3
|
+
* PDF extracts vertically-oriented labels as a sequence of single-character
|
|
4
|
+
* lines. `VerticalsStream` (a `StashingStream` subclass) detects runs of 6+
|
|
5
|
+
* such lines and merges them into one horizontal `LineItem`, combining their
|
|
6
|
+
* words and computing the correct bounding box.
|
|
7
|
+
*/
|
|
8
|
+
import ToLineItemTransformation from "../base/to-line-item-transform";
|
|
9
|
+
import ParseResult from "../../models/parse-result";
|
|
10
|
+
import LineItem from "../../models/line-item";
|
|
11
|
+
import StashingStream from "../../models/stashing-stream";
|
|
12
|
+
import { REMOVED_ANNOTATION, ADDED_ANNOTATION } from "../../models/annotation";
|
|
13
|
+
|
|
14
|
+
// Converts vertical text to horizontal
|
|
15
|
+
export default class VerticalToHorizontal extends ToLineItemTransformation {
|
|
16
|
+
constructor() {
|
|
17
|
+
super("Vertical to Horizontal Text");
|
|
18
|
+
}
|
|
19
|
+
|
|
20
|
+
transform(parseResult: ParseResult): ParseResult {
|
|
21
|
+
var foundVerticals = 0;
|
|
22
|
+
parseResult.pages.forEach((page) => {
|
|
23
|
+
const stream = new VerticalsStream();
|
|
24
|
+
stream.consumeAll(page.items);
|
|
25
|
+
page.items = stream.complete();
|
|
26
|
+
foundVerticals += stream.foundVerticals;
|
|
27
|
+
});
|
|
28
|
+
|
|
29
|
+
return new ParseResult({
|
|
30
|
+
...parseResult,
|
|
31
|
+
messages: ["Converted " + foundVerticals + " verticals"],
|
|
32
|
+
});
|
|
33
|
+
}
|
|
34
|
+
}
|
|
35
|
+
|
|
36
|
+
class VerticalsStream extends StashingStream {
|
|
37
|
+
foundVerticals: number;
|
|
38
|
+
|
|
39
|
+
constructor() {
|
|
40
|
+
super();
|
|
41
|
+
this.foundVerticals = 0;
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
shouldStash(item: any): boolean {
|
|
45
|
+
return item.words.length === 1 && item.words[0].string.length === 1;
|
|
46
|
+
}
|
|
47
|
+
|
|
48
|
+
doMatchesStash(lastItem: any, item: any): boolean {
|
|
49
|
+
return (
|
|
50
|
+
lastItem.y - item.y > 5 && lastItem.words[0].type === item.words[0].type
|
|
51
|
+
);
|
|
52
|
+
}
|
|
53
|
+
|
|
54
|
+
doFlushStash(stash: any[], results: any[]): void {
|
|
55
|
+
if (stash.length > 5) {
|
|
56
|
+
// unite
|
|
57
|
+
var combinedWords = [];
|
|
58
|
+
var minX = 999;
|
|
59
|
+
var maxY = 0;
|
|
60
|
+
var sumWidth = 0;
|
|
61
|
+
var maxHeight = 0;
|
|
62
|
+
stash.forEach((oneCharacterLine) => {
|
|
63
|
+
oneCharacterLine.annotation = REMOVED_ANNOTATION;
|
|
64
|
+
results.push(oneCharacterLine);
|
|
65
|
+
combinedWords.push(oneCharacterLine.words[0]);
|
|
66
|
+
minX = Math.min(minX, oneCharacterLine.x);
|
|
67
|
+
maxY = Math.max(maxY, oneCharacterLine.y);
|
|
68
|
+
sumWidth += oneCharacterLine.width;
|
|
69
|
+
maxHeight = Math.max(maxHeight, oneCharacterLine.height);
|
|
70
|
+
});
|
|
71
|
+
results.push(
|
|
72
|
+
new LineItem({
|
|
73
|
+
...stash[0],
|
|
74
|
+
x: minX,
|
|
75
|
+
y: maxY,
|
|
76
|
+
width: sumWidth,
|
|
77
|
+
height: maxHeight,
|
|
78
|
+
words: combinedWords,
|
|
79
|
+
annotation: ADDED_ANNOTATION,
|
|
80
|
+
}),
|
|
81
|
+
);
|
|
82
|
+
this.foundVerticals++;
|
|
83
|
+
} else {
|
|
84
|
+
// add as singles
|
|
85
|
+
results.push(...stash);
|
|
86
|
+
}
|
|
87
|
+
}
|
|
88
|
+
}
|
|
89
|
+
|
|
90
|
+
|
|
@@ -0,0 +1,46 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @description Final serialization stage that converts each page's text block
|
|
3
|
+
* items into `<p>`-wrapped HTML. TOC blocks are emitted verbatim; hyphenated
|
|
4
|
+
* line-break artifacts are stripped from non-list blocks; `<code>` fencing is
|
|
5
|
+
* removed from CODE-category blocks (treated as prose in this simplified output).
|
|
6
|
+
*/
|
|
7
|
+
|
|
8
|
+
import Transformation from './base/transformation'
|
|
9
|
+
import ParseResult from '../models/parse-result'
|
|
10
|
+
|
|
11
|
+
export default class ToHTML extends Transformation {
|
|
12
|
+
constructor () {
|
|
13
|
+
super('To HTML', 'String')
|
|
14
|
+
}
|
|
15
|
+
transform(parseResult: ParseResult): ParseResult {
|
|
16
|
+
parseResult.pages.forEach(page => {
|
|
17
|
+
var text = ''
|
|
18
|
+
page.items.forEach(block => {
|
|
19
|
+
// Concatenate all words in the same block, unless it's a Table of Contents block
|
|
20
|
+
let concatText
|
|
21
|
+
if (block.category === 'TOC') {
|
|
22
|
+
concatText = block.text
|
|
23
|
+
} else {
|
|
24
|
+
concatText = block.text.replace(/(\r\n|\n|\r)/gm, '\n')
|
|
25
|
+
}
|
|
26
|
+
|
|
27
|
+
// Concatenate words that were previously broken up by newline
|
|
28
|
+
if (block.category !== 'LIST') {
|
|
29
|
+
concatText = concatText.split('- ').join('')
|
|
30
|
+
}
|
|
31
|
+
|
|
32
|
+
// Assume there are no code blocks in our documents
|
|
33
|
+
if (block.category === 'CODE') {
|
|
34
|
+
concatText = concatText.split('<code>').join('').split('</code>').join('')
|
|
35
|
+
}
|
|
36
|
+
|
|
37
|
+
text += `<p>${concatText}</p>\n\n`
|
|
38
|
+
})
|
|
39
|
+
|
|
40
|
+
page.items = [text]
|
|
41
|
+
})
|
|
42
|
+
return new ParseResult({
|
|
43
|
+
...parseResult,
|
|
44
|
+
})
|
|
45
|
+
}
|
|
46
|
+
}
|